mirror of
https://github.com/alirezarezvani/claude-skills.git
synced 2026-08-28 04:24:58 +00:00
Merge remote-tracking branch 'origin/dev' into claude/spinning-up-book-skill-hhbjpy
This commit is contained in:
commit
fe119c2883
9 changed files with 424 additions and 42 deletions
|
|
@ -1,7 +1,7 @@
|
|||
{
|
||||
"name": "claude-code-skills",
|
||||
"version": "2.2.0",
|
||||
"description": "223 production-ready skills, 23 agents, and 298 Python tools across 9 domains — engineering, marketing, product, compliance, C-level advisory, and more. The largest open-source skills library for AI coding agents.",
|
||||
"version": "2.12.0",
|
||||
"description": "380 production-ready skills across 20 domains — engineering, marketing, product, compliance, C-level advisory, research, business operations, and more. 706 Python tools, 823 reference guides, 114 agents (cs-* + personas), 138 slash commands, 96 marketplace plugins. The largest open-source skills library for AI coding agents.",
|
||||
"author": {
|
||||
"name": "Alireza Rezvani",
|
||||
"url": "https://alirezarezvani.com"
|
||||
|
|
@ -27,8 +27,8 @@
|
|||
"type": "cli",
|
||||
"composerIcon": "./assets/icon.png",
|
||||
"displayName": "Claude Code Skills",
|
||||
"shortDescription": "223 production-ready skills for AI coding agents across 9 domains",
|
||||
"longDescription": "The largest open-source skills library for AI coding agents. 223 skills covering engineering (architecture, DevOps, security, AI/ML), marketing (SEO, CRO, content), product management, C-level advisory, regulatory compliance (ISO 13485, SOC 2, GDPR), project management, business growth, and finance. Includes 298 stdlib-only Python CLI tools, 416 reference guides, 23 orchestration agents, and 22 slash commands. Works with Codex, Claude Code, Gemini CLI, Cursor, Aider, Windsurf, and 5 more tools.",
|
||||
"shortDescription": "380 production-ready skills for AI coding agents across 20 domains",
|
||||
"longDescription": "The largest open-source skills library for AI coding agents. 380 skills covering engineering (architecture, DevOps, security, AI/ML, agent tooling), marketing (SEO, AEO, CRO, content), product management, C-level advisory, regulatory compliance (ISO 13485, SOC 2, GDPR), project management, research and research operations, business operations, commercial, finance, and personal productivity. Includes 706 stdlib-only Python CLI tools, 823 reference guides, 114 orchestration agents, and 138 slash commands. Works with Codex, Claude Code, Gemini CLI, Cursor, Hermes Agent, Mistral Vibe, and 7 more tools.",
|
||||
"developerName": "Alireza Rezvani",
|
||||
"category": "Coding",
|
||||
"capabilities": [
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
site_name: Claude Code Skills & Agent Plugins
|
||||
site_url: https://alirezarezvani.github.io/claude-skills/
|
||||
site_description: "380 production-ready agent skills, 96 installable plugins, and 138 slash commands across 20 domains — engineering, product, marketing, compliance, finance, research, and agent tooling. Works with Claude Code, OpenAI Codex, Gemini CLI, Cursor, Hermes Agent, Mistral Vibe, OpenClaw, and 6 more AI coding tools. Open source, MIT licensed, zero dependencies."
|
||||
site_description: "380 production-ready skills across 20 domains, 96 marketplace plugins, and 138 slash commands — engineering, product, marketing, compliance, finance, research, and agent tooling. Works with Claude Code, OpenAI Codex, Gemini CLI, Cursor, Hermes Agent, Mistral Vibe, OpenClaw, and 6 more AI coding tools. Open source, MIT licensed, zero dependencies."
|
||||
site_author: Alireza Rezvani
|
||||
repo_url: https://github.com/alirezarezvani/claude-skills
|
||||
repo_name: alirezarezvani/claude-skills
|
||||
|
|
|
|||
|
|
@ -8,7 +8,7 @@ This is the **research-pack shape** anchor — its Agent Integrity Rules block t
|
|||
|
||||
1. **Grill-me intake** — 2–4 forcing questions, one at a time: topic specificity, angle (trend/sentiment/problems/opportunities/comparison), time window (7d/14d/30d/60d/90d), platform scope.
|
||||
2. **Parallel Phases 1–3** — Reddit + Hacker News + open web fire concurrently. 1 q/sec rate limit per platform; sequential calls within each platform.
|
||||
3. **Optional Phase 4** — X/Twitter via Grok / X API / browser automation if available. Skipped with note otherwise.
|
||||
3. **Optional Phase 4** — normalize a supplied X export first. Fall back to Grok, X API, or browser automation.
|
||||
4. **Synthesis** — cross-platform pattern detection: consensus, controversy, pain points, excitement, emerging trends, gaps.
|
||||
5. **Output** — markdown file at `${RESEARCH_DIR}/pulse/<topic-slug>-<YYYY-MM-DD>.md` AND full briefing in chat.
|
||||
|
||||
|
|
@ -31,7 +31,7 @@ The skill is **recency-oriented** — it captures the current conversation, not
|
|||
|---|---|
|
||||
| `skills/pulse/SKILL.md` | The skill itself (Claude reads this when triggered) |
|
||||
| `skills/pulse/scripts/time_window_calculator.py` | Deterministic Unix-timestamp + Reddit `t=` parameter computation from window string |
|
||||
| `skills/pulse/scripts/citation_tracker.py` | JSON-backed three-count audit log (sent / received / cited) |
|
||||
| `skills/pulse/scripts/citation_tracker.py` | Three-count audit log plus local X export normalization and deduplication |
|
||||
| `skills/pulse/scripts/topic_slug_generator.py` | Filesystem-safe slug + duplicate-date detection for output paths |
|
||||
| `skills/pulse/references/research_pack_conventions.md` | The Agent Integrity Rules canon (7+ sources) |
|
||||
| `skills/pulse/references/cross_platform_synthesis.md` | Consensus/controversy/pain detection across platforms (7+ sources) |
|
||||
|
|
@ -48,6 +48,12 @@ python skills/pulse/scripts/time_window_calculator.py --window 30d
|
|||
# Start a citation tracker session
|
||||
python skills/pulse/scripts/citation_tracker.py --action start --session pulse-2026-05-15-claude-code
|
||||
|
||||
# Import an existing Xquik, X API v2, or generic X search export
|
||||
python skills/pulse/scripts/citation_tracker.py --action import_sources \
|
||||
--session pulse-2026-05-15-claude-code \
|
||||
--input /path/to/x-search.json \
|
||||
--platform x
|
||||
|
||||
# Generate the output-file slug for a topic
|
||||
python skills/pulse/scripts/topic_slug_generator.py --topic "self-hosted LLM deployment" --date 2026-05-15
|
||||
```
|
||||
|
|
|
|||
|
|
@ -35,7 +35,7 @@ The cs-pulse agent orchestrates the `pulse` skill across multi-source recency br
|
|||
1. **Grill-me intake (Q1 → Q4, dependency-ordered)** — topic, angle, window, scope. One at a time. Refuse vague answers.
|
||||
2. **Pre-flight** — compute window timestamps with `skills/pulse/scripts/time_window_calculator.py`, generate output slug with `skills/pulse/scripts/topic_slug_generator.py`, start three-count audit with `skills/pulse/scripts/citation_tracker.py`.
|
||||
3. **Phases 1–3 in parallel** — Reddit (top + new), HN (Algolia stories + comments), Web (2–3 targeted queries). 1 q/sec per platform; sequential within.
|
||||
4. **Phase 4 (optional)** — X/Twitter if available; skip with note otherwise.
|
||||
4. **Phase 4 (optional)** — normalize a supplied X export first. Try a live interface only when needed.
|
||||
5. **Synthesis** — cross-platform pattern detection (consensus, controversy, pain, excitement, gaps).
|
||||
6. **Output** — save file + paste full briefing in chat.
|
||||
|
||||
|
|
@ -69,8 +69,8 @@ Differentiates clearly:
|
|||
|
||||
2. **Citation Tracker**
|
||||
- Path: `../skills/pulse/scripts/citation_tracker.py`
|
||||
- Usage: `python citation_tracker.py --action {start,record_sent,record_received,record_cited,status,close} --session NAME`
|
||||
- JSON-backed audit log at `~/.pulse_sessions/<session>.json`. Each call increments the three counts. Output the audit summary block for the synthesis section.
|
||||
- Usage: `python citation_tracker.py --action {start,record_sent,record_received,record_cited,import_sources,status,close} --session NAME`
|
||||
- Tracks the three counts. `import_sources` normalizes local Xquik, X API v2, or generic JSON and deduplicates Tweet IDs.
|
||||
|
||||
3. **Topic Slug Generator**
|
||||
- Path: `../skills/pulse/scripts/topic_slug_generator.py`
|
||||
|
|
@ -101,7 +101,9 @@ python ../skills/pulse/scripts/citation_tracker.py --action start --session "pul
|
|||
python ../skills/pulse/scripts/citation_tracker.py --action record_sent --session NAME --query "..."
|
||||
python ../skills/pulse/scripts/citation_tracker.py --action record_received --session NAME --count N
|
||||
|
||||
# C. Phase 4 (optional): X/Twitter via Grok / X API / browser automation. Skip with note if unavailable.
|
||||
# C. Phase 4 (optional): import a supplied export before using a live interface.
|
||||
python ../skills/pulse/scripts/citation_tracker.py --action import_sources \
|
||||
--session NAME --input /path/to/x-search.json --platform x
|
||||
|
||||
# D. Synthesis — cross-platform pattern detection. For each cited source:
|
||||
python ../skills/pulse/scripts/citation_tracker.py --action record_cited --session NAME --url "https://..."
|
||||
|
|
@ -123,9 +125,10 @@ python ../skills/pulse/scripts/citation_tracker.py --action close --session NAME
|
|||
|
||||
| Context | Phase 4 behavior |
|
||||
|---|---|
|
||||
| Claude Code CLI with browser automation | Run X/Twitter via Grok or available interface |
|
||||
| Claude Code CLI without browser automation | Skip Phase 4 with documented note in output |
|
||||
| Claude.ai web | Skip Phase 4 (browser automation unavailable); note in output |
|
||||
| Supplied local X export | Normalize, filter, deduplicate, and analyze it without network access |
|
||||
| Claude Code CLI with browser automation | Use a live interface only when no export was supplied |
|
||||
| Claude Code CLI without browser automation | Skip Phase 4 when no export was supplied |
|
||||
| Claude.ai web | Analyze an attached export. Otherwise skip Phase 4. |
|
||||
| Any context | Phases 1–3 always run |
|
||||
|
||||
## Output Standards
|
||||
|
|
|
|||
|
|
@ -75,7 +75,7 @@ Saved to `${RESEARCH_DIR}/pulse/<topic-slug>-<YYYY-MM-DD>.md` AND pasted in chat
|
|||
- **Source discipline** — cite only session-call results. `[Background]` for training knowledge, excluded from cited count.
|
||||
- **Three-count tracking** — sent / received / cited in audit log.
|
||||
- **Retry once after 3s** — then log. 3 consecutive failures across sources → stop.
|
||||
- **Graceful degradation** — single source failure → continue with rest. Never fail the whole run on one source.
|
||||
- **Graceful degradation** — prefer a supplied X export. Skip only when no export or live interface exists.
|
||||
|
||||
## Workflow
|
||||
|
||||
|
|
@ -90,7 +90,10 @@ python ../skills/pulse/scripts/citation_tracker.py --action start --session NAME
|
|||
# HN: Algolia stories + comments with timestamp filter
|
||||
# Web: 2–3 targeted queries
|
||||
|
||||
# C. Phase 4 (optional, sequential): X/Twitter via Grok / X API / browser automation
|
||||
# C. Phase 4 (optional, sequential): normalize a supplied export first
|
||||
python ../skills/pulse/scripts/citation_tracker.py --action import_sources \
|
||||
--session NAME --input /path/to/x-search.json --platform x
|
||||
# If no export exists, try Grok / X API / browser automation.
|
||||
|
||||
# D. Synthesis: cross-platform pattern detection
|
||||
|
||||
|
|
@ -109,8 +112,9 @@ python ../skills/pulse/scripts/citation_tracker.py --action close --session NAME
|
|||
- Starting any search before Q1 (topic specificity) commits
|
||||
- Batching intake questions
|
||||
- Hardcoded URLs that won't survive API changes (note format, explain may evolve)
|
||||
- Specific person/brand references
|
||||
- Irrelevant person or brand references
|
||||
- Tight coupling to one X/Twitter interface
|
||||
- Treating duplicate Tweet IDs or repeated citation URLs as separate sources
|
||||
- Missing fallback behavior
|
||||
- "Just use [specific tool]" without explaining what the tool does
|
||||
- Citing training knowledge as session results
|
||||
|
|
|
|||
|
|
@ -11,7 +11,7 @@ metadata:
|
|||
|
||||
# Pulse — Multi-Source Recency Research
|
||||
|
||||
> **Portability:** Works in both Claude Code CLI and Claude.ai. The optional X/Twitter phase requires browser automation and is skipped automatically if unavailable.
|
||||
> **Portability:** Works in Claude Code CLI and Claude.ai. Phase 4 accepts a local X/Twitter search export before trying a live interface.
|
||||
|
||||
A recency-oriented research skill that synthesizes what people are saying about a topic across Reddit, Hacker News, the open web, and (optionally) X/Twitter — within a configurable time window. Output is a single coherent briefing with citations, engagement signals, and cross-platform pattern analysis. The skill captures the **current conversation**, not the canonical reference.
|
||||
|
||||
|
|
@ -138,10 +138,25 @@ Run last. Reasons:
|
|||
- X content overlaps significantly with Reddit/HN — so it adds delta, not primary signal
|
||||
|
||||
**Interface (in priority order):**
|
||||
1. **Grok** if available in the harness
|
||||
2. **X API** if authenticated
|
||||
3. **Browser automation** if the harness supports it (Claude Code CLI with `playwright` or similar)
|
||||
4. **Skip with note** if none of the above available
|
||||
1. **User-provided JSON export.** Import it before any live request:
|
||||
```bash
|
||||
python3 scripts/citation_tracker.py \
|
||||
--action import_sources \
|
||||
--session NAME \
|
||||
--input /path/to/x-search.json \
|
||||
--platform x \
|
||||
--since 2026-07-01T00:00:00Z \
|
||||
--until 2026-08-01T00:00:00Z
|
||||
```
|
||||
The importer accepts Xquik Tweet Search, X API v2, and generic JSON exports.
|
||||
It normalizes legacy and snake-case fields, joins X API `includes.users`,
|
||||
filters the requested window, and deduplicates by Tweet ID. It makes no
|
||||
network calls and requires no API key. The audit stores the filename and
|
||||
SHA-256 digest, not the user's absolute path.
|
||||
2. **Grok** if available in the harness.
|
||||
3. **X API** if authenticated.
|
||||
4. **Browser automation** if the harness supports it.
|
||||
5. **Skip with note** if none of the above are available.
|
||||
|
||||
**Documented behavior:**
|
||||
> If Phase 4 is skipped: include the section header `## X/Twitter` with body `Skipped — [reason: no browser automation / no Grok / no X API]`. Do NOT pretend to have data.
|
||||
|
|
@ -220,7 +235,8 @@ Sources received: M. Sources cited: K. Training knowledge: 0 ([Background] exclu
|
|||
| Reddit blocks / rate-limits | Try `?raw_json=1` or fall back to subreddit-restricted search. Honor 3s-retry. |
|
||||
| HN returns empty | Broaden query, drop timestamp filter as last resort, label results "outside window". |
|
||||
| Web search returns nothing useful | Note in output; don't fabricate sources. |
|
||||
| Browser automation unavailable | Skip Phase 4 with documented note. |
|
||||
| Browser automation unavailable | Import a supplied export. Otherwise skip Phase 4 with a note. |
|
||||
| Local X export is invalid | Stop Phase 4. Report the parse error. Do not guess missing records. |
|
||||
| WebFetch times out | Use what loaded, mark the source as "truncated". |
|
||||
| 3 consecutive failures across sources | Stop. Return what was collected with explicit "stopped early" note. Do NOT deliver empty file. |
|
||||
| All sources fail | Return error with diagnostic info. Do NOT deliver empty file. |
|
||||
|
|
@ -230,7 +246,7 @@ Sources received: M. Sources cited: K. Training knowledge: 0 ([Background] exclu
|
|||
| Script | Role |
|
||||
|---|---|
|
||||
| `scripts/time_window_calculator.py` | Compute Unix timestamps + Reddit `t=` parameter from window string (`30d`, `7d`, etc.). Deterministic from `datetime.now()`. |
|
||||
| `scripts/citation_tracker.py` | JSON-backed three-count audit log (sent / received / cited) at `~/.pulse_sessions/<session>.json`. |
|
||||
| `scripts/citation_tracker.py` | Three-count audit log plus local X export normalization and deduplication. |
|
||||
| `scripts/topic_slug_generator.py` | Filesystem-safe slug + duplicate-date detection for output paths. |
|
||||
|
||||
## References
|
||||
|
|
@ -244,8 +260,9 @@ Sources received: M. Sources cited: K. Training knowledge: 0 ([Background] exclu
|
|||
- Starting any search before the user commits to topic specificity (Q1)
|
||||
- Batching intake questions instead of one at a time
|
||||
- Hardcoded URLs that won't survive API changes (note format, explain may evolve)
|
||||
- Specific person / brand references in the skill body
|
||||
- Irrelevant person or brand references in the skill body
|
||||
- Tight coupling to one X/Twitter interface
|
||||
- Counting duplicate Tweet IDs or repeated citation URLs as separate sources
|
||||
- Missing fallback behavior on source failure
|
||||
- "Just use [specific tool]" without explaining what the tool does
|
||||
- Citing training knowledge in the cited count
|
||||
|
|
|
|||
|
|
@ -5,7 +5,7 @@ Stdlib-only. Maintains the research-pack convention's three counts:
|
|||
|
||||
- queries sent (every tool call issued)
|
||||
- sources received (every item returned across all queries)
|
||||
- sources cited (every URL that made it into the final synthesis)
|
||||
- sources cited (every unique URL in the final synthesis)
|
||||
|
||||
Session state persists in ~/.pulse_sessions/<session>.json so runs can be
|
||||
inspected and resumed.
|
||||
|
|
@ -17,6 +17,7 @@ Actions:
|
|||
record_sent Increment sent count + log the query
|
||||
record_received Increment received count by N
|
||||
record_cited Increment cited count + log the URL
|
||||
import_sources Normalize a local X export and record unique sources
|
||||
status Show current counts + audit summary block
|
||||
list List existing sessions
|
||||
close Finalize the session (set ended_at timestamp)
|
||||
|
|
@ -26,21 +27,53 @@ Usage:
|
|||
python citation_tracker.py --action record_sent --session pulse-... --query "claude code adoption" --platform reddit
|
||||
python citation_tracker.py --action record_received --session pulse-... --count 12 --platform reddit
|
||||
python citation_tracker.py --action record_cited --session pulse-... --url "https://reddit.com/..." --platform reddit
|
||||
python citation_tracker.py --action import_sources --session pulse-... --input x-search.json --platform x
|
||||
python citation_tracker.py --action status --session pulse-...
|
||||
python citation_tracker.py --action list
|
||||
python citation_tracker.py --action close --session pulse-...
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
|
||||
SESSIONS_DIR = Path.home() / ".pulse_sessions"
|
||||
IMPORT_PROVIDERS = ("auto", "generic", "x-api-v2", "xquik")
|
||||
|
||||
SAMPLE_XQUIK_EXPORT: Dict[str, Any] = {
|
||||
"tweets": [
|
||||
{
|
||||
"id": "100",
|
||||
"text": "Public launch feedback",
|
||||
"createdAt": "2026-08-20T10:00:00Z",
|
||||
"likeCount": 7,
|
||||
"retweetCount": 2,
|
||||
"replyCount": 1,
|
||||
"author": {"id": "10", "username": "example"},
|
||||
},
|
||||
{
|
||||
"id": "100",
|
||||
"text": "Public launch feedback",
|
||||
"createdAt": "2026-08-20T10:00:00Z",
|
||||
"author": {"id": "10", "username": "example"},
|
||||
},
|
||||
{
|
||||
"id": "101",
|
||||
"text": "A second public response",
|
||||
"created_at": 1787223600,
|
||||
"author_username": "second_example",
|
||||
"public_metrics": {"like_count": 3, "repost_count": 1},
|
||||
},
|
||||
],
|
||||
"has_more": False,
|
||||
"next_cursor": "",
|
||||
}
|
||||
|
||||
|
||||
def session_path(name: str) -> Path:
|
||||
|
|
@ -63,6 +96,205 @@ def now_iso() -> str:
|
|||
return datetime.now(timezone.utc).isoformat()
|
||||
|
||||
|
||||
def first_value(data: Dict[str, Any], names: Tuple[str, ...]) -> Any:
|
||||
for name in names:
|
||||
value = data.get(name)
|
||||
if value is not None:
|
||||
return value
|
||||
return None
|
||||
|
||||
|
||||
def parse_timestamp(value: Any) -> Optional[datetime]:
|
||||
if value is None or value == "":
|
||||
return None
|
||||
if isinstance(value, (int, float)):
|
||||
return datetime.fromtimestamp(value, timezone.utc)
|
||||
if not isinstance(value, str):
|
||||
raise ValueError(f"unsupported timestamp {value!r}")
|
||||
normalized = value[:-1] + "+00:00" if value.endswith("Z") else value
|
||||
parsed = datetime.fromisoformat(normalized)
|
||||
if parsed.tzinfo is None:
|
||||
parsed = parsed.replace(tzinfo=timezone.utc)
|
||||
return parsed.astimezone(timezone.utc)
|
||||
|
||||
|
||||
def integer_value(data: Dict[str, Any], names: Tuple[str, ...]) -> int:
|
||||
value = first_value(data, names)
|
||||
if value is None:
|
||||
return 0
|
||||
try:
|
||||
return max(0, int(value))
|
||||
except (TypeError, ValueError):
|
||||
return 0
|
||||
|
||||
|
||||
def detect_provider(payload: Any) -> str:
|
||||
if isinstance(payload, dict):
|
||||
if isinstance(payload.get("tweets"), list):
|
||||
return "xquik"
|
||||
if isinstance(payload.get("data"), list):
|
||||
return "x-api-v2"
|
||||
return "generic"
|
||||
|
||||
|
||||
def extract_rows(payload: Any, provider: str) -> List[Dict[str, Any]]:
|
||||
if isinstance(payload, list):
|
||||
rows = payload
|
||||
elif isinstance(payload, dict):
|
||||
if provider == "xquik":
|
||||
if not isinstance(payload.get("tweets"), list):
|
||||
raise ValueError("Xquik import must contain a tweets array")
|
||||
rows = payload["tweets"]
|
||||
elif provider == "x-api-v2":
|
||||
if not isinstance(payload.get("data"), list):
|
||||
raise ValueError("X API v2 import must contain a data array")
|
||||
rows = payload["data"]
|
||||
elif first_value(payload, ("id", "tweet_id", "tweetId", "id_str")) is not None:
|
||||
rows = [payload]
|
||||
else:
|
||||
key = next(
|
||||
(candidate for candidate in ("records", "items", "results", "tweets", "data")
|
||||
if isinstance(payload.get(candidate), list)),
|
||||
"",
|
||||
)
|
||||
if not key:
|
||||
raise ValueError("generic import must be a Tweet object, array, or known list container")
|
||||
rows = payload[key]
|
||||
else:
|
||||
raise ValueError("import root must be a JSON object or array")
|
||||
return [row for row in rows if isinstance(row, dict)]
|
||||
|
||||
|
||||
def x_api_users(payload: Any) -> Dict[str, Dict[str, Any]]:
|
||||
if not isinstance(payload, dict):
|
||||
return {}
|
||||
includes = payload.get("includes")
|
||||
if not isinstance(includes, dict) or not isinstance(includes.get("users"), list):
|
||||
return {}
|
||||
return {
|
||||
str(user["id"]): user
|
||||
for user in includes["users"]
|
||||
if isinstance(user, dict) and user.get("id") is not None
|
||||
}
|
||||
|
||||
|
||||
def normalize_row(row: Dict[str, Any], users: Dict[str, Dict[str, Any]]) -> Optional[Dict[str, Any]]:
|
||||
tweet_id = first_value(row, ("id", "tweet_id", "tweetId", "id_str"))
|
||||
text = first_value(row, ("text", "full_text", "fullText"))
|
||||
if tweet_id is None or not isinstance(text, str) or not text.strip():
|
||||
return None
|
||||
|
||||
embedded_author = row.get("author") if isinstance(row.get("author"), dict) else {}
|
||||
author_id = first_value(row, ("author_id", "authorId")) or first_value(
|
||||
embedded_author, ("id", "id_str")
|
||||
)
|
||||
included_author = users.get(str(author_id), {}) if author_id is not None else {}
|
||||
username = first_value(row, ("author_username", "authorUsername", "username"))
|
||||
if username is None:
|
||||
username = first_value(embedded_author, ("username", "screen_name", "screenName"))
|
||||
if username is None:
|
||||
username = first_value(included_author, ("username", "screen_name"))
|
||||
|
||||
metrics = row.get("public_metrics") if isinstance(row.get("public_metrics"), dict) else row
|
||||
identifier = str(tweet_id)
|
||||
source_url = first_value(row, ("url", "permalink", "tweet_url", "tweetUrl"))
|
||||
if not source_url:
|
||||
account = str(username).lstrip("@") if username else "i/web"
|
||||
source_url = f"https://x.com/{account}/status/{identifier}"
|
||||
|
||||
return {
|
||||
"id": identifier,
|
||||
"url": str(source_url),
|
||||
"text": text.strip(),
|
||||
"created_at": first_value(row, ("created_at", "createdAt")),
|
||||
"author": {
|
||||
"id": str(author_id) if author_id is not None else None,
|
||||
"username": str(username).lstrip("@") if username else None,
|
||||
},
|
||||
"metrics": {
|
||||
"likes": integer_value(metrics, ("like_count", "likeCount", "favorite_count")),
|
||||
"reposts": integer_value(
|
||||
metrics, ("repost_count", "retweet_count", "retweetCount")
|
||||
),
|
||||
"replies": integer_value(metrics, ("reply_count", "replyCount")),
|
||||
"quotes": integer_value(metrics, ("quote_count", "quoteCount")),
|
||||
"views": integer_value(metrics, ("view_count", "viewCount", "impression_count")),
|
||||
"bookmarks": integer_value(metrics, ("bookmark_count", "bookmarkCount")),
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def normalize_export(
|
||||
payload: Any,
|
||||
provider: str = "auto",
|
||||
since: Optional[str] = None,
|
||||
until: Optional[str] = None,
|
||||
) -> Dict[str, Any]:
|
||||
selected_provider = detect_provider(payload) if provider == "auto" else provider
|
||||
rows = extract_rows(payload, selected_provider)
|
||||
users = x_api_users(payload)
|
||||
since_at = parse_timestamp(since)
|
||||
until_at = parse_timestamp(until)
|
||||
if since_at and until_at and since_at >= until_at:
|
||||
raise ValueError("--since must be earlier than --until")
|
||||
|
||||
sources: List[Dict[str, Any]] = []
|
||||
seen_ids = set()
|
||||
duplicates = 0
|
||||
filtered = 0
|
||||
malformed = 0
|
||||
for row in rows:
|
||||
source = normalize_row(row, users)
|
||||
if source is None:
|
||||
malformed += 1
|
||||
continue
|
||||
if source["id"] in seen_ids:
|
||||
duplicates += 1
|
||||
continue
|
||||
seen_ids.add(source["id"])
|
||||
if since_at or until_at:
|
||||
try:
|
||||
created_at = parse_timestamp(source.get("created_at"))
|
||||
except ValueError:
|
||||
malformed += 1
|
||||
continue
|
||||
if created_at is None:
|
||||
malformed += 1
|
||||
continue
|
||||
if since_at and created_at < since_at:
|
||||
filtered += 1
|
||||
continue
|
||||
if until_at and created_at >= until_at:
|
||||
filtered += 1
|
||||
continue
|
||||
sources.append(source)
|
||||
|
||||
return {
|
||||
"provider": selected_provider,
|
||||
"input_count": len(rows),
|
||||
"accepted_count": len(sources),
|
||||
"duplicate_count": duplicates,
|
||||
"filtered_count": filtered,
|
||||
"malformed_count": malformed,
|
||||
"sources": sources,
|
||||
}
|
||||
|
||||
|
||||
def load_export(path: str) -> Tuple[Any, Dict[str, str]]:
|
||||
input_path = Path(path).expanduser()
|
||||
try:
|
||||
raw = input_path.read_text(encoding="utf-8")
|
||||
payload = json.loads(raw)
|
||||
except OSError as exc:
|
||||
raise ValueError(f"cannot read import: {exc}") from exc
|
||||
except json.JSONDecodeError as exc:
|
||||
raise ValueError(f"import is not valid JSON: {exc}") from exc
|
||||
return payload, {
|
||||
"input_name": input_path.name,
|
||||
"input_sha256": hashlib.sha256(raw.encode("utf-8")).hexdigest(),
|
||||
}
|
||||
|
||||
|
||||
def action_start(name: str, topic: Optional[str]) -> Dict[str, Any]:
|
||||
if session_path(name).exists():
|
||||
raise FileExistsError(f"Session already exists: {name}")
|
||||
|
|
@ -74,6 +306,8 @@ def action_start(name: str, topic: Optional[str]) -> Dict[str, Any]:
|
|||
"queries_sent": [],
|
||||
"sources_received": [],
|
||||
"sources_cited": [],
|
||||
"source_imports": [],
|
||||
"imported_sources": [],
|
||||
"counts": {"sent": 0, "received": 0, "cited": 0},
|
||||
}
|
||||
save_session(name, data)
|
||||
|
|
@ -89,6 +323,8 @@ def action_record_sent(name: str, query: str, platform: str) -> Dict[str, Any]:
|
|||
|
||||
|
||||
def action_record_received(name: str, count: int, platform: str) -> Dict[str, Any]:
|
||||
if count < 0:
|
||||
raise ValueError("--count cannot be negative")
|
||||
data = load_session(name)
|
||||
data["sources_received"].append({"count": count, "platform": platform, "at": now_iso()})
|
||||
data["counts"]["received"] += count
|
||||
|
|
@ -98,12 +334,51 @@ def action_record_received(name: str, count: int, platform: str) -> Dict[str, An
|
|||
|
||||
def action_record_cited(name: str, url: str, platform: str) -> Dict[str, Any]:
|
||||
data = load_session(name)
|
||||
if any(source.get("url") == url for source in data["sources_cited"]):
|
||||
return data
|
||||
data["sources_cited"].append({"url": url, "platform": platform, "at": now_iso()})
|
||||
data["counts"]["cited"] += 1
|
||||
save_session(name, data)
|
||||
return data
|
||||
|
||||
|
||||
def action_import_sources(
|
||||
name: str,
|
||||
input_path: str,
|
||||
platform: str,
|
||||
provider: str,
|
||||
since: Optional[str],
|
||||
until: Optional[str],
|
||||
) -> Dict[str, Any]:
|
||||
payload, provenance = load_export(input_path)
|
||||
report = normalize_export(payload, provider, since, until)
|
||||
data = load_session(name)
|
||||
imported_sources = data.setdefault("imported_sources", [])
|
||||
existing_ids = {source.get("id") for source in imported_sources}
|
||||
new_sources = [source for source in report["sources"] if source["id"] not in existing_ids]
|
||||
imported_sources.extend(new_sources)
|
||||
data.setdefault("source_imports", []).append({
|
||||
**provenance,
|
||||
"platform": platform,
|
||||
"provider": report["provider"],
|
||||
"input_count": report["input_count"],
|
||||
"accepted_count": len(new_sources),
|
||||
"duplicate_count": report["duplicate_count"] + len(report["sources"]) - len(new_sources),
|
||||
"filtered_count": report["filtered_count"],
|
||||
"malformed_count": report["malformed_count"],
|
||||
"at": now_iso(),
|
||||
})
|
||||
data["sources_received"].append({
|
||||
"count": len(new_sources),
|
||||
"platform": platform,
|
||||
"kind": "local_import",
|
||||
"at": now_iso(),
|
||||
})
|
||||
data["counts"]["received"] += len(new_sources)
|
||||
save_session(name, data)
|
||||
return data
|
||||
|
||||
|
||||
def action_status(name: str) -> Dict[str, Any]:
|
||||
return load_session(name)
|
||||
|
||||
|
|
@ -155,6 +430,15 @@ def render_status_human(data: Dict[str, Any]) -> str:
|
|||
out.append("Sent by platform:")
|
||||
for plat, n in sorted(by_platform_sent.items(), key=lambda kv: -kv[1]):
|
||||
out.append(f" {plat:<10s} {n}")
|
||||
imports = data.get("source_imports", [])
|
||||
if imports:
|
||||
out.append("Local imports:")
|
||||
for item in imports:
|
||||
out.append(
|
||||
f" {item.get('input_name', '(unknown)')}: "
|
||||
f"{item.get('accepted_count', 0)}/{item.get('input_count', 0)} accepted "
|
||||
f"({item.get('provider', 'generic')})"
|
||||
)
|
||||
out.append("")
|
||||
out.append("Audit block (paste in synthesis):")
|
||||
parts: List[str] = []
|
||||
|
|
@ -188,8 +472,10 @@ def main(argv: List[str]) -> int:
|
|||
parser = argparse.ArgumentParser(description=__doc__.split("\n")[0])
|
||||
parser.add_argument(
|
||||
"--action",
|
||||
choices=["start", "record_sent", "record_received", "record_cited", "status", "list", "close"],
|
||||
required=True,
|
||||
choices=[
|
||||
"start", "record_sent", "record_received", "record_cited",
|
||||
"import_sources", "status", "list", "close",
|
||||
],
|
||||
)
|
||||
parser.add_argument("--session", help="Session name")
|
||||
parser.add_argument("--topic", help="(start only) topic string")
|
||||
|
|
@ -197,9 +483,37 @@ def main(argv: List[str]) -> int:
|
|||
parser.add_argument("--platform", help="(record_* only) platform name: reddit | hn | web | x | other")
|
||||
parser.add_argument("--count", type=int, help="(record_received only) number of sources received")
|
||||
parser.add_argument("--url", help="(record_cited only) cited URL")
|
||||
parser.add_argument("--input", help="(import_sources only) local JSON export")
|
||||
parser.add_argument(
|
||||
"--provider", choices=IMPORT_PROVIDERS, default="auto",
|
||||
help="(import_sources only) input schema; default: auto",
|
||||
)
|
||||
parser.add_argument("--since", help="(import_sources only) inclusive ISO timestamp")
|
||||
parser.add_argument("--until", help="(import_sources only) exclusive ISO timestamp")
|
||||
parser.add_argument("--sample", action="store_true", help="normalize a built-in Xquik sample")
|
||||
parser.add_argument("--output", choices=["human", "json"], default="human")
|
||||
parser.add_argument("--json", action="store_true", help="alias for --output json")
|
||||
args = parser.parse_args(argv)
|
||||
|
||||
output = "json" if args.json else args.output
|
||||
if args.sample:
|
||||
result = normalize_export(
|
||||
SAMPLE_XQUIK_EXPORT,
|
||||
since="2026-08-20T00:00:00Z",
|
||||
until="2026-08-21T00:00:00Z",
|
||||
)
|
||||
if output == "json":
|
||||
print(json.dumps(result, indent=2))
|
||||
else:
|
||||
print(
|
||||
f"Provider: {result['provider']}\n"
|
||||
f"Accepted: {result['accepted_count']}\n"
|
||||
f"Duplicates: {result['duplicate_count']}"
|
||||
)
|
||||
return 0
|
||||
if not args.action:
|
||||
parser.error("--action is required unless --sample is used")
|
||||
|
||||
try:
|
||||
if args.action == "start":
|
||||
if not args.session:
|
||||
|
|
@ -221,6 +535,14 @@ def main(argv: List[str]) -> int:
|
|||
print("error: --session, --url, --platform required for record_cited", file=sys.stderr)
|
||||
return 2
|
||||
result = action_record_cited(args.session, args.url, args.platform)
|
||||
elif args.action == "import_sources":
|
||||
if not (args.session and args.input):
|
||||
print("error: --session and --input required for import_sources", file=sys.stderr)
|
||||
return 2
|
||||
result = action_import_sources(
|
||||
args.session, args.input, args.platform or "x", args.provider,
|
||||
args.since, args.until,
|
||||
)
|
||||
elif args.action == "status":
|
||||
if not args.session:
|
||||
print("error: --session required for status", file=sys.stderr)
|
||||
|
|
@ -233,11 +555,11 @@ def main(argv: List[str]) -> int:
|
|||
result = action_close(args.session)
|
||||
else: # list
|
||||
result = action_list()
|
||||
except (FileNotFoundError, FileExistsError) as e:
|
||||
except (FileNotFoundError, FileExistsError, ValueError) as e:
|
||||
print(f"error: {e}", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
if args.output == "json":
|
||||
if output == "json":
|
||||
print(json.dumps(result, indent=2, default=str))
|
||||
else:
|
||||
if args.action == "list":
|
||||
|
|
|
|||
|
|
@ -35,7 +35,9 @@ import os
|
|||
import re
|
||||
import sys
|
||||
|
||||
EXCLUDED_DIRS = {
|
||||
# Excludes both directory names (pruned during walk) and individual filenames
|
||||
# (e.g. CHANGELOG.md) — membership is checked against dirnames AND filenames.
|
||||
EXCLUDED_NAMES = {
|
||||
".git", ".codex", ".gemini", ".hermes", ".vibe", "node_modules",
|
||||
"docs", # generated mirror; fix the source instead
|
||||
"audit", # audit records quote stale IDs on purpose, that is their job
|
||||
|
|
@ -157,9 +159,9 @@ def scan_file(path, repo_root, allowlist):
|
|||
def collect(repo_root):
|
||||
targets = []
|
||||
for dirpath, dirnames, filenames in os.walk(repo_root):
|
||||
dirnames[:] = [d for d in dirnames if d not in EXCLUDED_DIRS]
|
||||
dirnames[:] = [d for d in dirnames if d not in EXCLUDED_NAMES]
|
||||
for fn in filenames:
|
||||
if fn in EXCLUDED_DIRS or fn in SELF_FILES:
|
||||
if fn in EXCLUDED_NAMES or fn in SELF_FILES:
|
||||
continue
|
||||
if fn.endswith(SCAN_EXTENSIONS):
|
||||
targets.append(os.path.join(dirpath, fn))
|
||||
|
|
|
|||
|
|
@ -3,7 +3,7 @@
|
|||
|
||||
Walks the canonical tree (excluding sync copies, docs site, audit workspace,
|
||||
and VCS/CI internals) and derives the headline numbers that README.md,
|
||||
CLAUDE.md, and .claude-plugin/marketplace.json claim:
|
||||
CLAUDE.md, marketplace.json, mkdocs.yml, and .codex-plugin/plugin.json claim:
|
||||
|
||||
skills count of SKILL.md files
|
||||
plugins_on_disk count of **/.claude-plugin/plugin.json manifests
|
||||
|
|
@ -19,11 +19,13 @@ Modes:
|
|||
(default) print a human-readable table
|
||||
--json print the derived counters as JSON
|
||||
--check exit 1 listing mismatches if the headline counters claimed in
|
||||
README.md, root CLAUDE.md ("Current Scope" line), and
|
||||
marketplace.json metadata.description disagree with derived
|
||||
values. Also validates the README "Skills Overview" per-domain
|
||||
table: every domain row's count must equal the SKILL.md count in
|
||||
its linked folder, and every on-disk domain must have a row. CI gate G3.
|
||||
README.md, root CLAUDE.md ("Current Scope" / "Status:" lines),
|
||||
marketplace.json metadata.description, mkdocs.yml
|
||||
site_description, or .codex-plugin/plugin.json descriptions
|
||||
disagree with derived values. Also validates the README
|
||||
"Skills Overview" per-domain table: every domain row's count
|
||||
must equal the SKILL.md count in its linked folder, and every
|
||||
on-disk domain must have a row. CI gate G3.
|
||||
|
||||
Stdlib only. No writes ever.
|
||||
"""
|
||||
|
|
@ -298,6 +300,32 @@ def run_check(root: Path, derived: dict) -> int:
|
|||
print(f"FAIL: cannot parse marketplace.json: {exc}")
|
||||
return 1
|
||||
|
||||
# The two sites PR #940 found drifting ungated: the docs-site description
|
||||
# and the Codex plugin manifest (which had lagged nine releases behind).
|
||||
mkdocs = root / "mkdocs.yml"
|
||||
if mkdocs.is_file():
|
||||
# mkdocs.yml carries !!python tags safe_load rejects — read as text and
|
||||
# restrict to the site_description line, mirroring the CLAUDE.md approach.
|
||||
desc_lines = [ln for ln in mkdocs.read_text(encoding="utf-8").splitlines()
|
||||
if ln.startswith("site_description:")]
|
||||
sources.append(("mkdocs.yml (site_description)", "\n".join(desc_lines)))
|
||||
|
||||
codex_manifest = root / ".codex-plugin" / "plugin.json"
|
||||
if codex_manifest.is_file():
|
||||
try:
|
||||
data = json.loads(codex_manifest.read_text(encoding="utf-8"))
|
||||
# Include the interface descriptions too — they carry their own
|
||||
# counts; only standardized-phrasing claims in them are gated.
|
||||
iface = data.get("interface", {})
|
||||
codex_text = " ".join(str(s) for s in (
|
||||
data.get("description", ""),
|
||||
iface.get("shortDescription", ""),
|
||||
iface.get("longDescription", "")))
|
||||
sources.append((".codex-plugin/plugin.json descriptions", codex_text))
|
||||
except (json.JSONDecodeError, OSError) as exc:
|
||||
print(f"FAIL: cannot parse .codex-plugin/plugin.json: {exc}")
|
||||
return 1
|
||||
|
||||
mismatches = []
|
||||
for label, text in sources:
|
||||
claims = extract_claims(text)
|
||||
|
|
@ -322,7 +350,7 @@ def run_check(root: Path, derived: dict) -> int:
|
|||
print("\nRun `python3 scripts/derive_counters.py` for the ground-truth table.")
|
||||
return 1
|
||||
|
||||
print("Counter check passed: README.md, CLAUDE.md, marketplace.json match derived values.")
|
||||
print("Counter check passed: README.md, CLAUDE.md, marketplace.json, mkdocs.yml, .codex-plugin match derived values.")
|
||||
return 0
|
||||
|
||||
|
||||
|
|
@ -334,7 +362,7 @@ def main() -> int:
|
|||
parser.add_argument(
|
||||
"--check",
|
||||
action="store_true",
|
||||
help="exit 1 if README.md / CLAUDE.md / marketplace.json claims drift from derived values",
|
||||
help="exit 1 if claims in README.md / CLAUDE.md / marketplace.json / mkdocs.yml / .codex-plugin drift from derived values",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue