diff --git a/.codex-plugin/plugin.json b/.codex-plugin/plugin.json index c6c54abd..f59c033e 100644 --- a/.codex-plugin/plugin.json +++ b/.codex-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "claude-code-skills", - "version": "2.2.0", - "description": "223 production-ready skills, 23 agents, and 298 Python tools across 9 domains — engineering, marketing, product, compliance, C-level advisory, and more. The largest open-source skills library for AI coding agents.", + "version": "2.12.0", + "description": "380 production-ready skills across 20 domains — engineering, marketing, product, compliance, C-level advisory, research, business operations, and more. 706 Python tools, 823 reference guides, 114 agents (cs-* + personas), 138 slash commands, 96 marketplace plugins. The largest open-source skills library for AI coding agents.", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -27,8 +27,8 @@ "type": "cli", "composerIcon": "./assets/icon.png", "displayName": "Claude Code Skills", - "shortDescription": "223 production-ready skills for AI coding agents across 9 domains", - "longDescription": "The largest open-source skills library for AI coding agents. 223 skills covering engineering (architecture, DevOps, security, AI/ML), marketing (SEO, CRO, content), product management, C-level advisory, regulatory compliance (ISO 13485, SOC 2, GDPR), project management, business growth, and finance. Includes 298 stdlib-only Python CLI tools, 416 reference guides, 23 orchestration agents, and 22 slash commands. Works with Codex, Claude Code, Gemini CLI, Cursor, Aider, Windsurf, and 5 more tools.", + "shortDescription": "380 production-ready skills for AI coding agents across 20 domains", + "longDescription": "The largest open-source skills library for AI coding agents. 380 skills covering engineering (architecture, DevOps, security, AI/ML, agent tooling), marketing (SEO, AEO, CRO, content), product management, C-level advisory, regulatory compliance (ISO 13485, SOC 2, GDPR), project management, research and research operations, business operations, commercial, finance, and personal productivity. Includes 706 stdlib-only Python CLI tools, 823 reference guides, 114 orchestration agents, and 138 slash commands. Works with Codex, Claude Code, Gemini CLI, Cursor, Hermes Agent, Mistral Vibe, and 7 more tools.", "developerName": "Alireza Rezvani", "category": "Coding", "capabilities": [ diff --git a/mkdocs.yml b/mkdocs.yml index 30683986..2cf62b7d 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -1,6 +1,6 @@ site_name: Claude Code Skills & Agent Plugins site_url: https://alirezarezvani.github.io/claude-skills/ -site_description: "380 production-ready agent skills, 96 installable plugins, and 138 slash commands across 20 domains — engineering, product, marketing, compliance, finance, research, and agent tooling. Works with Claude Code, OpenAI Codex, Gemini CLI, Cursor, Hermes Agent, Mistral Vibe, OpenClaw, and 6 more AI coding tools. Open source, MIT licensed, zero dependencies." +site_description: "380 production-ready skills across 20 domains, 96 marketplace plugins, and 138 slash commands — engineering, product, marketing, compliance, finance, research, and agent tooling. Works with Claude Code, OpenAI Codex, Gemini CLI, Cursor, Hermes Agent, Mistral Vibe, OpenClaw, and 6 more AI coding tools. Open source, MIT licensed, zero dependencies." site_author: Alireza Rezvani repo_url: https://github.com/alirezarezvani/claude-skills repo_name: alirezarezvani/claude-skills diff --git a/research/pulse/README.md b/research/pulse/README.md index 50cc1132..14597b8b 100644 --- a/research/pulse/README.md +++ b/research/pulse/README.md @@ -8,7 +8,7 @@ This is the **research-pack shape** anchor — its Agent Integrity Rules block t 1. **Grill-me intake** — 2–4 forcing questions, one at a time: topic specificity, angle (trend/sentiment/problems/opportunities/comparison), time window (7d/14d/30d/60d/90d), platform scope. 2. **Parallel Phases 1–3** — Reddit + Hacker News + open web fire concurrently. 1 q/sec rate limit per platform; sequential calls within each platform. -3. **Optional Phase 4** — X/Twitter via Grok / X API / browser automation if available. Skipped with note otherwise. +3. **Optional Phase 4** — normalize a supplied X export first. Fall back to Grok, X API, or browser automation. 4. **Synthesis** — cross-platform pattern detection: consensus, controversy, pain points, excitement, emerging trends, gaps. 5. **Output** — markdown file at `${RESEARCH_DIR}/pulse/-.md` AND full briefing in chat. @@ -31,7 +31,7 @@ The skill is **recency-oriented** — it captures the current conversation, not |---|---| | `skills/pulse/SKILL.md` | The skill itself (Claude reads this when triggered) | | `skills/pulse/scripts/time_window_calculator.py` | Deterministic Unix-timestamp + Reddit `t=` parameter computation from window string | -| `skills/pulse/scripts/citation_tracker.py` | JSON-backed three-count audit log (sent / received / cited) | +| `skills/pulse/scripts/citation_tracker.py` | Three-count audit log plus local X export normalization and deduplication | | `skills/pulse/scripts/topic_slug_generator.py` | Filesystem-safe slug + duplicate-date detection for output paths | | `skills/pulse/references/research_pack_conventions.md` | The Agent Integrity Rules canon (7+ sources) | | `skills/pulse/references/cross_platform_synthesis.md` | Consensus/controversy/pain detection across platforms (7+ sources) | @@ -48,6 +48,12 @@ python skills/pulse/scripts/time_window_calculator.py --window 30d # Start a citation tracker session python skills/pulse/scripts/citation_tracker.py --action start --session pulse-2026-05-15-claude-code +# Import an existing Xquik, X API v2, or generic X search export +python skills/pulse/scripts/citation_tracker.py --action import_sources \ + --session pulse-2026-05-15-claude-code \ + --input /path/to/x-search.json \ + --platform x + # Generate the output-file slug for a topic python skills/pulse/scripts/topic_slug_generator.py --topic "self-hosted LLM deployment" --date 2026-05-15 ``` diff --git a/research/pulse/agents/cs-pulse.md b/research/pulse/agents/cs-pulse.md index 9f81dd18..5d3c9049 100644 --- a/research/pulse/agents/cs-pulse.md +++ b/research/pulse/agents/cs-pulse.md @@ -35,7 +35,7 @@ The cs-pulse agent orchestrates the `pulse` skill across multi-source recency br 1. **Grill-me intake (Q1 → Q4, dependency-ordered)** — topic, angle, window, scope. One at a time. Refuse vague answers. 2. **Pre-flight** — compute window timestamps with `skills/pulse/scripts/time_window_calculator.py`, generate output slug with `skills/pulse/scripts/topic_slug_generator.py`, start three-count audit with `skills/pulse/scripts/citation_tracker.py`. 3. **Phases 1–3 in parallel** — Reddit (top + new), HN (Algolia stories + comments), Web (2–3 targeted queries). 1 q/sec per platform; sequential within. -4. **Phase 4 (optional)** — X/Twitter if available; skip with note otherwise. +4. **Phase 4 (optional)** — normalize a supplied X export first. Try a live interface only when needed. 5. **Synthesis** — cross-platform pattern detection (consensus, controversy, pain, excitement, gaps). 6. **Output** — save file + paste full briefing in chat. @@ -69,8 +69,8 @@ Differentiates clearly: 2. **Citation Tracker** - Path: `../skills/pulse/scripts/citation_tracker.py` - - Usage: `python citation_tracker.py --action {start,record_sent,record_received,record_cited,status,close} --session NAME` - - JSON-backed audit log at `~/.pulse_sessions/.json`. Each call increments the three counts. Output the audit summary block for the synthesis section. + - Usage: `python citation_tracker.py --action {start,record_sent,record_received,record_cited,import_sources,status,close} --session NAME` + - Tracks the three counts. `import_sources` normalizes local Xquik, X API v2, or generic JSON and deduplicates Tweet IDs. 3. **Topic Slug Generator** - Path: `../skills/pulse/scripts/topic_slug_generator.py` @@ -101,7 +101,9 @@ python ../skills/pulse/scripts/citation_tracker.py --action start --session "pul python ../skills/pulse/scripts/citation_tracker.py --action record_sent --session NAME --query "..." python ../skills/pulse/scripts/citation_tracker.py --action record_received --session NAME --count N -# C. Phase 4 (optional): X/Twitter via Grok / X API / browser automation. Skip with note if unavailable. +# C. Phase 4 (optional): import a supplied export before using a live interface. +python ../skills/pulse/scripts/citation_tracker.py --action import_sources \ + --session NAME --input /path/to/x-search.json --platform x # D. Synthesis — cross-platform pattern detection. For each cited source: python ../skills/pulse/scripts/citation_tracker.py --action record_cited --session NAME --url "https://..." @@ -123,9 +125,10 @@ python ../skills/pulse/scripts/citation_tracker.py --action close --session NAME | Context | Phase 4 behavior | |---|---| -| Claude Code CLI with browser automation | Run X/Twitter via Grok or available interface | -| Claude Code CLI without browser automation | Skip Phase 4 with documented note in output | -| Claude.ai web | Skip Phase 4 (browser automation unavailable); note in output | +| Supplied local X export | Normalize, filter, deduplicate, and analyze it without network access | +| Claude Code CLI with browser automation | Use a live interface only when no export was supplied | +| Claude Code CLI without browser automation | Skip Phase 4 when no export was supplied | +| Claude.ai web | Analyze an attached export. Otherwise skip Phase 4. | | Any context | Phases 1–3 always run | ## Output Standards diff --git a/research/pulse/commands/cs-pulse.md b/research/pulse/commands/cs-pulse.md index 5f588d40..373c9339 100644 --- a/research/pulse/commands/cs-pulse.md +++ b/research/pulse/commands/cs-pulse.md @@ -75,7 +75,7 @@ Saved to `${RESEARCH_DIR}/pulse/-.md` AND pasted in chat - **Source discipline** — cite only session-call results. `[Background]` for training knowledge, excluded from cited count. - **Three-count tracking** — sent / received / cited in audit log. - **Retry once after 3s** — then log. 3 consecutive failures across sources → stop. -- **Graceful degradation** — single source failure → continue with rest. Never fail the whole run on one source. +- **Graceful degradation** — prefer a supplied X export. Skip only when no export or live interface exists. ## Workflow @@ -90,7 +90,10 @@ python ../skills/pulse/scripts/citation_tracker.py --action start --session NAME # HN: Algolia stories + comments with timestamp filter # Web: 2–3 targeted queries -# C. Phase 4 (optional, sequential): X/Twitter via Grok / X API / browser automation +# C. Phase 4 (optional, sequential): normalize a supplied export first +python ../skills/pulse/scripts/citation_tracker.py --action import_sources \ + --session NAME --input /path/to/x-search.json --platform x +# If no export exists, try Grok / X API / browser automation. # D. Synthesis: cross-platform pattern detection @@ -109,8 +112,9 @@ python ../skills/pulse/scripts/citation_tracker.py --action close --session NAME - Starting any search before Q1 (topic specificity) commits - Batching intake questions - Hardcoded URLs that won't survive API changes (note format, explain may evolve) -- Specific person/brand references +- Irrelevant person or brand references - Tight coupling to one X/Twitter interface +- Treating duplicate Tweet IDs or repeated citation URLs as separate sources - Missing fallback behavior - "Just use [specific tool]" without explaining what the tool does - Citing training knowledge as session results diff --git a/research/pulse/skills/pulse/SKILL.md b/research/pulse/skills/pulse/SKILL.md index 11589fe1..7fa92304 100644 --- a/research/pulse/skills/pulse/SKILL.md +++ b/research/pulse/skills/pulse/SKILL.md @@ -11,7 +11,7 @@ metadata: # Pulse — Multi-Source Recency Research -> **Portability:** Works in both Claude Code CLI and Claude.ai. The optional X/Twitter phase requires browser automation and is skipped automatically if unavailable. +> **Portability:** Works in Claude Code CLI and Claude.ai. Phase 4 accepts a local X/Twitter search export before trying a live interface. A recency-oriented research skill that synthesizes what people are saying about a topic across Reddit, Hacker News, the open web, and (optionally) X/Twitter — within a configurable time window. Output is a single coherent briefing with citations, engagement signals, and cross-platform pattern analysis. The skill captures the **current conversation**, not the canonical reference. @@ -138,10 +138,25 @@ Run last. Reasons: - X content overlaps significantly with Reddit/HN — so it adds delta, not primary signal **Interface (in priority order):** -1. **Grok** if available in the harness -2. **X API** if authenticated -3. **Browser automation** if the harness supports it (Claude Code CLI with `playwright` or similar) -4. **Skip with note** if none of the above available +1. **User-provided JSON export.** Import it before any live request: + ```bash + python3 scripts/citation_tracker.py \ + --action import_sources \ + --session NAME \ + --input /path/to/x-search.json \ + --platform x \ + --since 2026-07-01T00:00:00Z \ + --until 2026-08-01T00:00:00Z + ``` + The importer accepts Xquik Tweet Search, X API v2, and generic JSON exports. + It normalizes legacy and snake-case fields, joins X API `includes.users`, + filters the requested window, and deduplicates by Tweet ID. It makes no + network calls and requires no API key. The audit stores the filename and + SHA-256 digest, not the user's absolute path. +2. **Grok** if available in the harness. +3. **X API** if authenticated. +4. **Browser automation** if the harness supports it. +5. **Skip with note** if none of the above are available. **Documented behavior:** > If Phase 4 is skipped: include the section header `## X/Twitter` with body `Skipped — [reason: no browser automation / no Grok / no X API]`. Do NOT pretend to have data. @@ -220,7 +235,8 @@ Sources received: M. Sources cited: K. Training knowledge: 0 ([Background] exclu | Reddit blocks / rate-limits | Try `?raw_json=1` or fall back to subreddit-restricted search. Honor 3s-retry. | | HN returns empty | Broaden query, drop timestamp filter as last resort, label results "outside window". | | Web search returns nothing useful | Note in output; don't fabricate sources. | -| Browser automation unavailable | Skip Phase 4 with documented note. | +| Browser automation unavailable | Import a supplied export. Otherwise skip Phase 4 with a note. | +| Local X export is invalid | Stop Phase 4. Report the parse error. Do not guess missing records. | | WebFetch times out | Use what loaded, mark the source as "truncated". | | 3 consecutive failures across sources | Stop. Return what was collected with explicit "stopped early" note. Do NOT deliver empty file. | | All sources fail | Return error with diagnostic info. Do NOT deliver empty file. | @@ -230,7 +246,7 @@ Sources received: M. Sources cited: K. Training knowledge: 0 ([Background] exclu | Script | Role | |---|---| | `scripts/time_window_calculator.py` | Compute Unix timestamps + Reddit `t=` parameter from window string (`30d`, `7d`, etc.). Deterministic from `datetime.now()`. | -| `scripts/citation_tracker.py` | JSON-backed three-count audit log (sent / received / cited) at `~/.pulse_sessions/.json`. | +| `scripts/citation_tracker.py` | Three-count audit log plus local X export normalization and deduplication. | | `scripts/topic_slug_generator.py` | Filesystem-safe slug + duplicate-date detection for output paths. | ## References @@ -244,8 +260,9 @@ Sources received: M. Sources cited: K. Training knowledge: 0 ([Background] exclu - Starting any search before the user commits to topic specificity (Q1) - Batching intake questions instead of one at a time - Hardcoded URLs that won't survive API changes (note format, explain may evolve) -- Specific person / brand references in the skill body +- Irrelevant person or brand references in the skill body - Tight coupling to one X/Twitter interface +- Counting duplicate Tweet IDs or repeated citation URLs as separate sources - Missing fallback behavior on source failure - "Just use [specific tool]" without explaining what the tool does - Citing training knowledge in the cited count diff --git a/research/pulse/skills/pulse/scripts/citation_tracker.py b/research/pulse/skills/pulse/scripts/citation_tracker.py index 74681006..ad509cc3 100644 --- a/research/pulse/skills/pulse/scripts/citation_tracker.py +++ b/research/pulse/skills/pulse/scripts/citation_tracker.py @@ -5,7 +5,7 @@ Stdlib-only. Maintains the research-pack convention's three counts: - queries sent (every tool call issued) - sources received (every item returned across all queries) - - sources cited (every URL that made it into the final synthesis) + - sources cited (every unique URL in the final synthesis) Session state persists in ~/.pulse_sessions/.json so runs can be inspected and resumed. @@ -17,6 +17,7 @@ Actions: record_sent Increment sent count + log the query record_received Increment received count by N record_cited Increment cited count + log the URL + import_sources Normalize a local X export and record unique sources status Show current counts + audit summary block list List existing sessions close Finalize the session (set ended_at timestamp) @@ -26,21 +27,53 @@ Usage: python citation_tracker.py --action record_sent --session pulse-... --query "claude code adoption" --platform reddit python citation_tracker.py --action record_received --session pulse-... --count 12 --platform reddit python citation_tracker.py --action record_cited --session pulse-... --url "https://reddit.com/..." --platform reddit + python citation_tracker.py --action import_sources --session pulse-... --input x-search.json --platform x python citation_tracker.py --action status --session pulse-... python citation_tracker.py --action list python citation_tracker.py --action close --session pulse-... """ import argparse +import hashlib import json import os import sys from datetime import datetime, timezone from pathlib import Path -from typing import Any, Dict, List, Optional +from typing import Any, Dict, List, Optional, Tuple SESSIONS_DIR = Path.home() / ".pulse_sessions" +IMPORT_PROVIDERS = ("auto", "generic", "x-api-v2", "xquik") + +SAMPLE_XQUIK_EXPORT: Dict[str, Any] = { + "tweets": [ + { + "id": "100", + "text": "Public launch feedback", + "createdAt": "2026-08-20T10:00:00Z", + "likeCount": 7, + "retweetCount": 2, + "replyCount": 1, + "author": {"id": "10", "username": "example"}, + }, + { + "id": "100", + "text": "Public launch feedback", + "createdAt": "2026-08-20T10:00:00Z", + "author": {"id": "10", "username": "example"}, + }, + { + "id": "101", + "text": "A second public response", + "created_at": 1787223600, + "author_username": "second_example", + "public_metrics": {"like_count": 3, "repost_count": 1}, + }, + ], + "has_more": False, + "next_cursor": "", +} def session_path(name: str) -> Path: @@ -63,6 +96,205 @@ def now_iso() -> str: return datetime.now(timezone.utc).isoformat() +def first_value(data: Dict[str, Any], names: Tuple[str, ...]) -> Any: + for name in names: + value = data.get(name) + if value is not None: + return value + return None + + +def parse_timestamp(value: Any) -> Optional[datetime]: + if value is None or value == "": + return None + if isinstance(value, (int, float)): + return datetime.fromtimestamp(value, timezone.utc) + if not isinstance(value, str): + raise ValueError(f"unsupported timestamp {value!r}") + normalized = value[:-1] + "+00:00" if value.endswith("Z") else value + parsed = datetime.fromisoformat(normalized) + if parsed.tzinfo is None: + parsed = parsed.replace(tzinfo=timezone.utc) + return parsed.astimezone(timezone.utc) + + +def integer_value(data: Dict[str, Any], names: Tuple[str, ...]) -> int: + value = first_value(data, names) + if value is None: + return 0 + try: + return max(0, int(value)) + except (TypeError, ValueError): + return 0 + + +def detect_provider(payload: Any) -> str: + if isinstance(payload, dict): + if isinstance(payload.get("tweets"), list): + return "xquik" + if isinstance(payload.get("data"), list): + return "x-api-v2" + return "generic" + + +def extract_rows(payload: Any, provider: str) -> List[Dict[str, Any]]: + if isinstance(payload, list): + rows = payload + elif isinstance(payload, dict): + if provider == "xquik": + if not isinstance(payload.get("tweets"), list): + raise ValueError("Xquik import must contain a tweets array") + rows = payload["tweets"] + elif provider == "x-api-v2": + if not isinstance(payload.get("data"), list): + raise ValueError("X API v2 import must contain a data array") + rows = payload["data"] + elif first_value(payload, ("id", "tweet_id", "tweetId", "id_str")) is not None: + rows = [payload] + else: + key = next( + (candidate for candidate in ("records", "items", "results", "tweets", "data") + if isinstance(payload.get(candidate), list)), + "", + ) + if not key: + raise ValueError("generic import must be a Tweet object, array, or known list container") + rows = payload[key] + else: + raise ValueError("import root must be a JSON object or array") + return [row for row in rows if isinstance(row, dict)] + + +def x_api_users(payload: Any) -> Dict[str, Dict[str, Any]]: + if not isinstance(payload, dict): + return {} + includes = payload.get("includes") + if not isinstance(includes, dict) or not isinstance(includes.get("users"), list): + return {} + return { + str(user["id"]): user + for user in includes["users"] + if isinstance(user, dict) and user.get("id") is not None + } + + +def normalize_row(row: Dict[str, Any], users: Dict[str, Dict[str, Any]]) -> Optional[Dict[str, Any]]: + tweet_id = first_value(row, ("id", "tweet_id", "tweetId", "id_str")) + text = first_value(row, ("text", "full_text", "fullText")) + if tweet_id is None or not isinstance(text, str) or not text.strip(): + return None + + embedded_author = row.get("author") if isinstance(row.get("author"), dict) else {} + author_id = first_value(row, ("author_id", "authorId")) or first_value( + embedded_author, ("id", "id_str") + ) + included_author = users.get(str(author_id), {}) if author_id is not None else {} + username = first_value(row, ("author_username", "authorUsername", "username")) + if username is None: + username = first_value(embedded_author, ("username", "screen_name", "screenName")) + if username is None: + username = first_value(included_author, ("username", "screen_name")) + + metrics = row.get("public_metrics") if isinstance(row.get("public_metrics"), dict) else row + identifier = str(tweet_id) + source_url = first_value(row, ("url", "permalink", "tweet_url", "tweetUrl")) + if not source_url: + account = str(username).lstrip("@") if username else "i/web" + source_url = f"https://x.com/{account}/status/{identifier}" + + return { + "id": identifier, + "url": str(source_url), + "text": text.strip(), + "created_at": first_value(row, ("created_at", "createdAt")), + "author": { + "id": str(author_id) if author_id is not None else None, + "username": str(username).lstrip("@") if username else None, + }, + "metrics": { + "likes": integer_value(metrics, ("like_count", "likeCount", "favorite_count")), + "reposts": integer_value( + metrics, ("repost_count", "retweet_count", "retweetCount") + ), + "replies": integer_value(metrics, ("reply_count", "replyCount")), + "quotes": integer_value(metrics, ("quote_count", "quoteCount")), + "views": integer_value(metrics, ("view_count", "viewCount", "impression_count")), + "bookmarks": integer_value(metrics, ("bookmark_count", "bookmarkCount")), + }, + } + + +def normalize_export( + payload: Any, + provider: str = "auto", + since: Optional[str] = None, + until: Optional[str] = None, +) -> Dict[str, Any]: + selected_provider = detect_provider(payload) if provider == "auto" else provider + rows = extract_rows(payload, selected_provider) + users = x_api_users(payload) + since_at = parse_timestamp(since) + until_at = parse_timestamp(until) + if since_at and until_at and since_at >= until_at: + raise ValueError("--since must be earlier than --until") + + sources: List[Dict[str, Any]] = [] + seen_ids = set() + duplicates = 0 + filtered = 0 + malformed = 0 + for row in rows: + source = normalize_row(row, users) + if source is None: + malformed += 1 + continue + if source["id"] in seen_ids: + duplicates += 1 + continue + seen_ids.add(source["id"]) + if since_at or until_at: + try: + created_at = parse_timestamp(source.get("created_at")) + except ValueError: + malformed += 1 + continue + if created_at is None: + malformed += 1 + continue + if since_at and created_at < since_at: + filtered += 1 + continue + if until_at and created_at >= until_at: + filtered += 1 + continue + sources.append(source) + + return { + "provider": selected_provider, + "input_count": len(rows), + "accepted_count": len(sources), + "duplicate_count": duplicates, + "filtered_count": filtered, + "malformed_count": malformed, + "sources": sources, + } + + +def load_export(path: str) -> Tuple[Any, Dict[str, str]]: + input_path = Path(path).expanduser() + try: + raw = input_path.read_text(encoding="utf-8") + payload = json.loads(raw) + except OSError as exc: + raise ValueError(f"cannot read import: {exc}") from exc + except json.JSONDecodeError as exc: + raise ValueError(f"import is not valid JSON: {exc}") from exc + return payload, { + "input_name": input_path.name, + "input_sha256": hashlib.sha256(raw.encode("utf-8")).hexdigest(), + } + + def action_start(name: str, topic: Optional[str]) -> Dict[str, Any]: if session_path(name).exists(): raise FileExistsError(f"Session already exists: {name}") @@ -74,6 +306,8 @@ def action_start(name: str, topic: Optional[str]) -> Dict[str, Any]: "queries_sent": [], "sources_received": [], "sources_cited": [], + "source_imports": [], + "imported_sources": [], "counts": {"sent": 0, "received": 0, "cited": 0}, } save_session(name, data) @@ -89,6 +323,8 @@ def action_record_sent(name: str, query: str, platform: str) -> Dict[str, Any]: def action_record_received(name: str, count: int, platform: str) -> Dict[str, Any]: + if count < 0: + raise ValueError("--count cannot be negative") data = load_session(name) data["sources_received"].append({"count": count, "platform": platform, "at": now_iso()}) data["counts"]["received"] += count @@ -98,12 +334,51 @@ def action_record_received(name: str, count: int, platform: str) -> Dict[str, An def action_record_cited(name: str, url: str, platform: str) -> Dict[str, Any]: data = load_session(name) + if any(source.get("url") == url for source in data["sources_cited"]): + return data data["sources_cited"].append({"url": url, "platform": platform, "at": now_iso()}) data["counts"]["cited"] += 1 save_session(name, data) return data +def action_import_sources( + name: str, + input_path: str, + platform: str, + provider: str, + since: Optional[str], + until: Optional[str], +) -> Dict[str, Any]: + payload, provenance = load_export(input_path) + report = normalize_export(payload, provider, since, until) + data = load_session(name) + imported_sources = data.setdefault("imported_sources", []) + existing_ids = {source.get("id") for source in imported_sources} + new_sources = [source for source in report["sources"] if source["id"] not in existing_ids] + imported_sources.extend(new_sources) + data.setdefault("source_imports", []).append({ + **provenance, + "platform": platform, + "provider": report["provider"], + "input_count": report["input_count"], + "accepted_count": len(new_sources), + "duplicate_count": report["duplicate_count"] + len(report["sources"]) - len(new_sources), + "filtered_count": report["filtered_count"], + "malformed_count": report["malformed_count"], + "at": now_iso(), + }) + data["sources_received"].append({ + "count": len(new_sources), + "platform": platform, + "kind": "local_import", + "at": now_iso(), + }) + data["counts"]["received"] += len(new_sources) + save_session(name, data) + return data + + def action_status(name: str) -> Dict[str, Any]: return load_session(name) @@ -155,6 +430,15 @@ def render_status_human(data: Dict[str, Any]) -> str: out.append("Sent by platform:") for plat, n in sorted(by_platform_sent.items(), key=lambda kv: -kv[1]): out.append(f" {plat:<10s} {n}") + imports = data.get("source_imports", []) + if imports: + out.append("Local imports:") + for item in imports: + out.append( + f" {item.get('input_name', '(unknown)')}: " + f"{item.get('accepted_count', 0)}/{item.get('input_count', 0)} accepted " + f"({item.get('provider', 'generic')})" + ) out.append("") out.append("Audit block (paste in synthesis):") parts: List[str] = [] @@ -188,8 +472,10 @@ def main(argv: List[str]) -> int: parser = argparse.ArgumentParser(description=__doc__.split("\n")[0]) parser.add_argument( "--action", - choices=["start", "record_sent", "record_received", "record_cited", "status", "list", "close"], - required=True, + choices=[ + "start", "record_sent", "record_received", "record_cited", + "import_sources", "status", "list", "close", + ], ) parser.add_argument("--session", help="Session name") parser.add_argument("--topic", help="(start only) topic string") @@ -197,9 +483,37 @@ def main(argv: List[str]) -> int: parser.add_argument("--platform", help="(record_* only) platform name: reddit | hn | web | x | other") parser.add_argument("--count", type=int, help="(record_received only) number of sources received") parser.add_argument("--url", help="(record_cited only) cited URL") + parser.add_argument("--input", help="(import_sources only) local JSON export") + parser.add_argument( + "--provider", choices=IMPORT_PROVIDERS, default="auto", + help="(import_sources only) input schema; default: auto", + ) + parser.add_argument("--since", help="(import_sources only) inclusive ISO timestamp") + parser.add_argument("--until", help="(import_sources only) exclusive ISO timestamp") + parser.add_argument("--sample", action="store_true", help="normalize a built-in Xquik sample") parser.add_argument("--output", choices=["human", "json"], default="human") + parser.add_argument("--json", action="store_true", help="alias for --output json") args = parser.parse_args(argv) + output = "json" if args.json else args.output + if args.sample: + result = normalize_export( + SAMPLE_XQUIK_EXPORT, + since="2026-08-20T00:00:00Z", + until="2026-08-21T00:00:00Z", + ) + if output == "json": + print(json.dumps(result, indent=2)) + else: + print( + f"Provider: {result['provider']}\n" + f"Accepted: {result['accepted_count']}\n" + f"Duplicates: {result['duplicate_count']}" + ) + return 0 + if not args.action: + parser.error("--action is required unless --sample is used") + try: if args.action == "start": if not args.session: @@ -221,6 +535,14 @@ def main(argv: List[str]) -> int: print("error: --session, --url, --platform required for record_cited", file=sys.stderr) return 2 result = action_record_cited(args.session, args.url, args.platform) + elif args.action == "import_sources": + if not (args.session and args.input): + print("error: --session and --input required for import_sources", file=sys.stderr) + return 2 + result = action_import_sources( + args.session, args.input, args.platform or "x", args.provider, + args.since, args.until, + ) elif args.action == "status": if not args.session: print("error: --session required for status", file=sys.stderr) @@ -233,11 +555,11 @@ def main(argv: List[str]) -> int: result = action_close(args.session) else: # list result = action_list() - except (FileNotFoundError, FileExistsError) as e: + except (FileNotFoundError, FileExistsError, ValueError) as e: print(f"error: {e}", file=sys.stderr) return 2 - if args.output == "json": + if output == "json": print(json.dumps(result, indent=2, default=str)) else: if args.action == "list": diff --git a/scripts/check_model_freshness.py b/scripts/check_model_freshness.py index 5d32c309..39cc679a 100644 --- a/scripts/check_model_freshness.py +++ b/scripts/check_model_freshness.py @@ -35,7 +35,9 @@ import os import re import sys -EXCLUDED_DIRS = { +# Excludes both directory names (pruned during walk) and individual filenames +# (e.g. CHANGELOG.md) — membership is checked against dirnames AND filenames. +EXCLUDED_NAMES = { ".git", ".codex", ".gemini", ".hermes", ".vibe", "node_modules", "docs", # generated mirror; fix the source instead "audit", # audit records quote stale IDs on purpose, that is their job @@ -157,9 +159,9 @@ def scan_file(path, repo_root, allowlist): def collect(repo_root): targets = [] for dirpath, dirnames, filenames in os.walk(repo_root): - dirnames[:] = [d for d in dirnames if d not in EXCLUDED_DIRS] + dirnames[:] = [d for d in dirnames if d not in EXCLUDED_NAMES] for fn in filenames: - if fn in EXCLUDED_DIRS or fn in SELF_FILES: + if fn in EXCLUDED_NAMES or fn in SELF_FILES: continue if fn.endswith(SCAN_EXTENSIONS): targets.append(os.path.join(dirpath, fn)) diff --git a/scripts/derive_counters.py b/scripts/derive_counters.py index cbc17d9b..c87060eb 100644 --- a/scripts/derive_counters.py +++ b/scripts/derive_counters.py @@ -3,7 +3,7 @@ Walks the canonical tree (excluding sync copies, docs site, audit workspace, and VCS/CI internals) and derives the headline numbers that README.md, -CLAUDE.md, and .claude-plugin/marketplace.json claim: +CLAUDE.md, marketplace.json, mkdocs.yml, and .codex-plugin/plugin.json claim: skills count of SKILL.md files plugins_on_disk count of **/.claude-plugin/plugin.json manifests @@ -19,11 +19,13 @@ Modes: (default) print a human-readable table --json print the derived counters as JSON --check exit 1 listing mismatches if the headline counters claimed in - README.md, root CLAUDE.md ("Current Scope" line), and - marketplace.json metadata.description disagree with derived - values. Also validates the README "Skills Overview" per-domain - table: every domain row's count must equal the SKILL.md count in - its linked folder, and every on-disk domain must have a row. CI gate G3. + README.md, root CLAUDE.md ("Current Scope" / "Status:" lines), + marketplace.json metadata.description, mkdocs.yml + site_description, or .codex-plugin/plugin.json descriptions + disagree with derived values. Also validates the README + "Skills Overview" per-domain table: every domain row's count + must equal the SKILL.md count in its linked folder, and every + on-disk domain must have a row. CI gate G3. Stdlib only. No writes ever. """ @@ -298,6 +300,32 @@ def run_check(root: Path, derived: dict) -> int: print(f"FAIL: cannot parse marketplace.json: {exc}") return 1 + # The two sites PR #940 found drifting ungated: the docs-site description + # and the Codex plugin manifest (which had lagged nine releases behind). + mkdocs = root / "mkdocs.yml" + if mkdocs.is_file(): + # mkdocs.yml carries !!python tags safe_load rejects — read as text and + # restrict to the site_description line, mirroring the CLAUDE.md approach. + desc_lines = [ln for ln in mkdocs.read_text(encoding="utf-8").splitlines() + if ln.startswith("site_description:")] + sources.append(("mkdocs.yml (site_description)", "\n".join(desc_lines))) + + codex_manifest = root / ".codex-plugin" / "plugin.json" + if codex_manifest.is_file(): + try: + data = json.loads(codex_manifest.read_text(encoding="utf-8")) + # Include the interface descriptions too — they carry their own + # counts; only standardized-phrasing claims in them are gated. + iface = data.get("interface", {}) + codex_text = " ".join(str(s) for s in ( + data.get("description", ""), + iface.get("shortDescription", ""), + iface.get("longDescription", ""))) + sources.append((".codex-plugin/plugin.json descriptions", codex_text)) + except (json.JSONDecodeError, OSError) as exc: + print(f"FAIL: cannot parse .codex-plugin/plugin.json: {exc}") + return 1 + mismatches = [] for label, text in sources: claims = extract_claims(text) @@ -322,7 +350,7 @@ def run_check(root: Path, derived: dict) -> int: print("\nRun `python3 scripts/derive_counters.py` for the ground-truth table.") return 1 - print("Counter check passed: README.md, CLAUDE.md, marketplace.json match derived values.") + print("Counter check passed: README.md, CLAUDE.md, marketplace.json, mkdocs.yml, .codex-plugin match derived values.") return 0 @@ -334,7 +362,7 @@ def main() -> int: parser.add_argument( "--check", action="store_true", - help="exit 1 if README.md / CLAUDE.md / marketplace.json claims drift from derived values", + help="exit 1 if claims in README.md / CLAUDE.md / marketplace.json / mkdocs.yml / .codex-plugin drift from derived values", ) args = parser.parse_args()