From 867db4150f0f419e286c85fea533fde421636f04 Mon Sep 17 00:00:00 2001 From: mehanshbarthwal-lab Date: Wed, 20 May 2026 14:01:18 +0530 Subject: [PATCH] chore: register skill in marketplace.json --- .claude-plugin/marketplace.json | 60 +++-- .idea/.gitignore | 10 + .idea/claude-skills.iml | 24 ++ .../inspectionProfiles/profiles_settings.xml | 6 + .idea/modules.xml | 8 + .idea/vcs.xml | 6 + .../universal-scraping-architect/LICENSE | 21 -- .../universal-scraping-architect/README.md | 212 ------------------ .../{agents => }/cs-scraping-architect.md | 0 .../commands/{commands => }/cs-scrape.md | 0 .../references/parsing-and-data-extraction.md | 10 - .../scraping-ethics-security.md | 0 12 files changed, 93 insertions(+), 264 deletions(-) create mode 100644 .idea/.gitignore create mode 100644 .idea/claude-skills.iml create mode 100644 .idea/inspectionProfiles/profiles_settings.xml create mode 100644 .idea/modules.xml create mode 100644 .idea/vcs.xml delete mode 100644 engineering/universal-scraping-architect/LICENSE delete mode 100644 engineering/universal-scraping-architect/README.md rename engineering/universal-scraping-architect/agents/{agents => }/cs-scraping-architect.md (100%) rename engineering/universal-scraping-architect/commands/{commands => }/cs-scrape.md (100%) delete mode 100644 engineering/universal-scraping-architect/references/references/parsing-and-data-extraction.md rename engineering/universal-scraping-architect/references/{references/references => }/scraping-ethics-security.md (100%) diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 711a2239..54d751d5 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -4,7 +4,7 @@ "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" }, - "description": "328 production-ready skill packages for Claude AI across 14 domains: engineering advanced (76 \u2014 incl. 4 Matt Pocock-derived productivity skills + v2.7.3 security-guidance PreToolUse hook), engineering core (51), marketing (47 \u2014 incl. v2.7.3 AEO/Answer Engine Optimization), c-level advisory (66), product (17), regulatory/QMS (18), project management (9), business growth (5), finance (4), productivity (4, v2.7.0), marketing top-level (2, v2.7.0), research (8, v2.7.0), business-operations (7, v2.8.0), and commercial (8, v2.8.0). Includes ~441 Python tools, ~594 reference documents, 48+ agents, 77+ slash commands.", + "description": "328 production-ready skill packages for Claude AI across 14 domains: engineering advanced (76 — incl. 4 Matt Pocock-derived productivity skills + v2.7.3 security-guidance PreToolUse hook), engineering core (51), marketing (47 — incl. v2.7.3 AEO/Answer Engine Optimization), c-level advisory (66), product (17), regulatory/QMS (18), project management (9), business growth (5), finance (4), productivity (4, v2.7.0), marketing top-level (2, v2.7.0), research (8, v2.7.0), business-operations (7, v2.8.0), and commercial (8, v2.8.0). Includes ~441 Python tools, ~594 reference documents, 48+ agents, 77+ slash commands.", "homepage": "https://github.com/alirezarezvani/claude-skills", "repository": "https://github.com/alirezarezvani/claude-skills", "metadata": { @@ -61,7 +61,7 @@ { "name": "c-level-agents", "source": "./c-level-advisor/c-level-agents", - "description": "Founder-mode executive team plugin: 13 cs-* C-suite agents (CFO, CMO, CRO, CPO, COO, CHRO, CISO, Chief of Staff, General Counsel, Chief Data Officer, Chief AI Officer, Chief Customer Officer, VP of Engineering) with distinct cognitive voices, plus 21 /cs:* slash commands \u2014 forcing-question office hours (CFO/CMO/CPO/CRO/CTO/CISO/GC/CDO/CAIO/CCO/VPE reviews), strategic sprint pipeline (brief \u2192 boardroom \u2192 decide \u2192 execute \u2192 post-mortem), and meta routing (/cs:founder-mode auto-router, /cs:onboard, /cs:cross-eval multi-model consensus, /cs:freeze cooldown lock). Wraps the 33 c-level skills with cognitive gearing, persona voice, and artifact-driven handoffs. The business-domain answer to YC Garry Tan's gstack.", + "description": "Founder-mode executive team plugin: 13 cs-* C-suite agents (CFO, CMO, CRO, CPO, COO, CHRO, CISO, Chief of Staff, General Counsel, Chief Data Officer, Chief AI Officer, Chief Customer Officer, VP of Engineering) with distinct cognitive voices, plus 21 /cs:* slash commands — forcing-question office hours (CFO/CMO/CPO/CRO/CTO/CISO/GC/CDO/CAIO/CCO/VPE reviews), strategic sprint pipeline (brief → boardroom → decide → execute → post-mortem), and meta routing (/cs:founder-mode auto-router, /cs:onboard, /cs:cross-eval multi-model consensus, /cs:freeze cooldown lock). Wraps the 33 c-level skills with cognitive gearing, persona voice, and artifact-driven handoffs. The business-domain answer to YC Garry Tan's gstack.", "version": "1.5.0", "author": { "name": "Alireza Rezvani" @@ -111,7 +111,7 @@ { "name": "general-counsel-advisor", "source": "./c-level-advisor/general-counsel-advisor", - "description": "General Counsel advisory for startups: contract risk scanner (12 founder-killer patterns: auto-renew traps, uncapped indemnity, vague IP, MFN pricing, missing DPA, one-sided venue, broad non-solicit, perpetual license-back, etc.) and term sheet analyzer (0-100 founder-friendliness across 12 dimensions). 3 in-depth references: contracts playbook (7 startup contract types), IP + regulatory landscape mapping (HIPAA, GDPR, FDA, fintech, EU AI Act, SOC 2 \u2192 ISO sequencing), term sheet decoder (full glossary + founder-friendly defaults). Standalone-installable; also bundled in c-level-skills. Stdlib-only. NOT a substitute for licensed counsel.", + "description": "General Counsel advisory for startups: contract risk scanner (12 founder-killer patterns: auto-renew traps, uncapped indemnity, vague IP, MFN pricing, missing DPA, one-sided venue, broad non-solicit, perpetual license-back, etc.) and term sheet analyzer (0-100 founder-friendliness across 12 dimensions). 3 in-depth references: contracts playbook (7 startup contract types), IP + regulatory landscape mapping (HIPAA, GDPR, FDA, fintech, EU AI Act, SOC 2 → ISO sequencing), term sheet decoder (full glossary + founder-friendly defaults). Standalone-installable; also bundled in c-level-skills. Stdlib-only. NOT a substitute for licensed counsel.", "version": "1.0.0", "author": { "name": "Alireza Rezvani" @@ -133,7 +133,7 @@ { "name": "chief-data-officer-advisor", "source": "./c-level-advisor/chief-data-officer-advisor", - "description": "Chief Data Officer advisory for startups: AI training data audit (origin \u00d7 class \u00d7 use-case matrix with GDPR Art. 6 + EU AI Act citations), data product strategy picker (warehouse vs lakehouse vs mesh + 6-layer build-vs-buy + 12-month sequencing), data asset valuator (strategic value 0-10 + M&A multiplier with carve-out penalties + 3 ranked productization paths). 4 references answering one decision each: training rights, data product strategy, customer-data-as-asset, data team org evolution. Standalone-installable; also bundled in c-level-skills. Strategic only \u2014 does not duplicate engineering data skills.", + "description": "Chief Data Officer advisory for startups: AI training data audit (origin × class × use-case matrix with GDPR Art. 6 + EU AI Act citations), data product strategy picker (warehouse vs lakehouse vs mesh + 6-layer build-vs-buy + 12-month sequencing), data asset valuator (strategic value 0-10 + M&A multiplier with carve-out penalties + 3 ranked productization paths). 4 references answering one decision each: training rights, data product strategy, customer-data-as-asset, data team org evolution. Standalone-installable; also bundled in c-level-skills. Strategic only — does not duplicate engineering data skills.", "version": "1.0.0", "author": { "name": "Alireza Rezvani" @@ -155,7 +155,7 @@ { "name": "vpe-advisor", "source": "./c-level-advisor/vpe-advisor", - "description": "VP of Engineering advisory: delivery throughput analyzer (DORA 4 metrics + cycle-time bottleneck identification with typical fixes per stage), engineering hiring funnel calculator (7-stage conversion + pipeline gap + weakest-stage fixes from sourcing to offer-accept), engineering team structure designer (squad/tribe model + manager-trigger + director-trigger + span-of-control). 4 in-depth references citing DORA / Spotify / Conway / Google SRE / Larson / Fournier. Standalone-installable; also bundled in c-level-skills. NOT a CTO skill \u2014 VPE owns how the team ships; CTO owns what to build.", + "description": "VP of Engineering advisory: delivery throughput analyzer (DORA 4 metrics + cycle-time bottleneck identification with typical fixes per stage), engineering hiring funnel calculator (7-stage conversion + pipeline gap + weakest-stage fixes from sourcing to offer-accept), engineering team structure designer (squad/tribe model + manager-trigger + director-trigger + span-of-control). 4 in-depth references citing DORA / Spotify / Conway / Google SRE / Larson / Fournier. Standalone-installable; also bundled in c-level-skills. NOT a CTO skill — VPE owns how the team ships; CTO owns what to build.", "version": "1.0.0", "author": { "name": "Alireza Rezvani" @@ -179,7 +179,7 @@ { "name": "chief-customer-officer-advisor", "source": "./c-level-advisor/chief-customer-officer-advisor", - "description": "Chief Customer Officer advisory: retention decomposition analyzer (honest GRR vs NRR; 7-category churn taxonomy with preventable% scoring), customer segmentation designer (4-tier framework, ICP fit scoring across 7 weighted signals, kill list + upgrade candidates), CS coverage calculator (pooled vs named CSM ratio math + 12-month hiring plan with quarterly sequencing). 4 in-depth references each citing 5+ authoritative sources. Standalone-installable; also bundled in c-level-skills. Strategic only \u2014 does not duplicate business-growth tactical CS skills.", + "description": "Chief Customer Officer advisory: retention decomposition analyzer (honest GRR vs NRR; 7-category churn taxonomy with preventable% scoring), customer segmentation designer (4-tier framework, ICP fit scoring across 7 weighted signals, kill list + upgrade candidates), CS coverage calculator (pooled vs named CSM ratio math + 12-month hiring plan with quarterly sequencing). 4 in-depth references each citing 5+ authoritative sources. Standalone-installable; also bundled in c-level-skills. Strategic only — does not duplicate business-growth tactical CS skills.", "version": "1.0.0", "author": { "name": "Alireza Rezvani" @@ -201,7 +201,7 @@ { "name": "chief-ai-officer-advisor", "source": "./c-level-advisor/chief-ai-officer-advisor", - "description": "Chief AI Officer advisory for startups: model build-vs-buy calculator (API vs fine-tune vs build with 3-year TCO across 6 paths + breakeven that balances economics with practical feasibility), AI risk classifier (EU AI Act tier with 7 Article citations + US state patchwork: NYC LL 144, CO AI Act, IL HB 53, CA SB 1001, IL BIPA + industry overlays for FDA AI/ML, CFPB Circular 2023-03, NYDFS Reg 23, NAIC, ECOA, Fed SR 11-7), AI cost economics (API vs self-hosted breakeven with 2026 pricing across A100/H100, utilization reality, hidden costs). 4 in-depth references each citing 5+ authoritative sources. Standalone-installable; also bundled in c-level-skills. Strategic only \u2014 does not duplicate engineering AI/ML skills.", + "description": "Chief AI Officer advisory for startups: model build-vs-buy calculator (API vs fine-tune vs build with 3-year TCO across 6 paths + breakeven that balances economics with practical feasibility), AI risk classifier (EU AI Act tier with 7 Article citations + US state patchwork: NYC LL 144, CO AI Act, IL HB 53, CA SB 1001, IL BIPA + industry overlays for FDA AI/ML, CFPB Circular 2023-03, NYDFS Reg 23, NAIC, ECOA, Fed SR 11-7), AI cost economics (API vs self-hosted breakeven with 2026 pricing across A100/H100, utilization reality, hidden costs). 4 in-depth references each citing 5+ authoritative sources. Standalone-installable; also bundled in c-level-skills. Strategic only — does not duplicate engineering AI/ML skills.", "version": "1.0.0", "author": { "name": "Alireza Rezvani" @@ -415,7 +415,7 @@ { "name": "autoresearch-agent", "source": "./engineering/autoresearch-agent", - "description": "Autonomous experiment loop \u2014 optimize any file by a measurable metric. 5 slash commands (/ar:setup, /ar:run, /ar:loop, /ar:status, /ar:resume), 8 built-in evaluators, configurable loop intervals (10min to monthly).", + "description": "Autonomous experiment loop — optimize any file by a measurable metric. 5 slash commands (/ar:setup, /ar:run, /ar:loop, /ar:status, /ar:resume), 8 built-in evaluators, configurable loop intervals (10min to monthly).", "version": "2.2.2", "author": { "name": "Alireza Rezvani" @@ -480,7 +480,7 @@ { "name": "agenthub", "source": "./engineering/agenthub", - "description": "Multi-agent collaboration \u2014 spawn N parallel subagents that compete on code optimization, content drafts, research approaches, or any task that benefits from diverse solutions. 7 slash commands (/hub:init, /hub:spawn, /hub:status, /hub:eval, /hub:merge, /hub:board, /hub:run), agent templates, DAG-based orchestration, LLM judge mode, message board coordination.", + "description": "Multi-agent collaboration — spawn N parallel subagents that compete on code optimization, content drafts, research approaches, or any task that benefits from diverse solutions. 7 slash commands (/hub:init, /hub:spawn, /hub:status, /hub:eval, /hub:merge, /hub:board, /hub:run), agent templates, DAG-based orchestration, LLM judge mode, message board coordination.", "version": "2.2.2", "author": { "name": "Alireza Rezvani" @@ -539,7 +539,7 @@ { "name": "docker-development", "source": "./engineering/docker-development", - "description": "Docker and container development \u2014 Dockerfile optimization, docker-compose orchestration, multi-stage builds, security hardening, and CI/CD container pipelines.", + "description": "Docker and container development — Dockerfile optimization, docker-compose orchestration, multi-stage builds, security hardening, and CI/CD container pipelines.", "version": "2.2.2", "author": { "name": "Alireza Rezvani" @@ -556,7 +556,7 @@ { "name": "helm-chart-builder", "source": "./engineering/helm-chart-builder", - "description": "Helm chart development \u2014 chart scaffolding, values design, template patterns, dependency management, and Kubernetes deployment strategies.", + "description": "Helm chart development — chart scaffolding, values design, template patterns, dependency management, and Kubernetes deployment strategies.", "version": "2.2.2", "author": { "name": "Alireza Rezvani" @@ -573,7 +573,7 @@ { "name": "terraform-patterns", "source": "./engineering/terraform-patterns", - "description": "Terraform infrastructure-as-code \u2014 module design patterns, state management, provider configuration, CI/CD integration, and multi-environment strategies.", + "description": "Terraform infrastructure-as-code — module design patterns, state management, provider configuration, CI/CD integration, and multi-environment strategies.", "version": "2.2.2", "author": { "name": "Alireza Rezvani" @@ -590,7 +590,7 @@ { "name": "research-summarizer", "source": "./product-team/research-summarizer", - "description": "Structured research summarization \u2014 summarize academic papers, market research, user interviews, and competitive analysis into actionable insights.", + "description": "Structured research summarization — summarize academic papers, market research, user interviews, and competitive analysis into actionable insights.", "version": "2.2.2", "author": { "name": "Alireza Rezvani" @@ -607,7 +607,7 @@ { "name": "code-tour", "source": "./engineering/code-tour", - "description": "Create CodeTour .tour files \u2014 persona-targeted, step-by-step walkthroughs that link to real files and line numbers. 10 developer personas, all CodeTour step types, SMIG description formula.", + "description": "Create CodeTour .tour files — persona-targeted, step-by-step walkthroughs that link to real files and line numbers. 10 developer personas, all CodeTour step types, SMIG description formula.", "version": "2.2.2", "author": { "name": "Alireza Rezvani" @@ -760,7 +760,7 @@ { "name": "kubernetes-operator", "source": "./engineering/kubernetes-operator", - "description": "End-to-end Kubernetes Operator discipline: CRD design, reconcile-loop patterns, and OperatorHub Capability Levels. Ships CRD validator, reconcile-loop linter, and capability auditor (3 stdlib Python tools), 4 references on the operator pattern + CRD design + reconcile patterns + framework comparison (controller-runtime/kubebuilder/operator-sdk/metacontroller/KOPF), CRD + Go controller skeletons, and /operator-audit slash command. NOT a generic k8s skill \u2014 specifically the Operator pattern.", + "description": "End-to-end Kubernetes Operator discipline: CRD design, reconcile-loop patterns, and OperatorHub Capability Levels. Ships CRD validator, reconcile-loop linter, and capability auditor (3 stdlib Python tools), 4 references on the operator pattern + CRD design + reconcile patterns + framework comparison (controller-runtime/kubebuilder/operator-sdk/metacontroller/KOPF), CRD + Go controller skeletons, and /operator-audit slash command. NOT a generic k8s skill — specifically the Operator pattern.", "version": "2.4.0", "author": { "name": "Alireza Rezvani" @@ -825,7 +825,7 @@ { "name": "write-a-skill", "source": "./engineering/write-a-skill", - "description": "Skill-author skill: create new agent skills with proper structure, progressive disclosure, and bundled resources. Derived from Matt Pocock's MIT-licensed write-a-skill with: (1) 3 stdlib Python validation tools (description validator, structure validator, review-checklist runner \u2014 all enforcing Matt's 6-item checklist), (2) 4 references citing 7-8 authoritative sources each (progressive disclosure principles, description design patterns, quality gates, companion tooling), (3) cs-skill-author persona agent + /cs:write-a-skill slash command. Matt's voice and 3-phase workflow (Gather \u2192 Draft \u2192 Review) preserved verbatim per MIT.", + "description": "Skill-author skill: create new agent skills with proper structure, progressive disclosure, and bundled resources. Derived from Matt Pocock's MIT-licensed write-a-skill with: (1) 3 stdlib Python validation tools (description validator, structure validator, review-checklist runner — all enforcing Matt's 6-item checklist), (2) 4 references citing 7-8 authoritative sources each (progressive disclosure principles, description design patterns, quality gates, companion tooling), (3) cs-skill-author persona agent + /cs:write-a-skill slash command. Matt's voice and 3-phase workflow (Gather → Draft → Review) preserved verbatim per MIT.", "version": "2.6.0", "author": { "name": "Alireza Rezvani" @@ -881,7 +881,7 @@ { "name": "handoff", "source": "./engineering/handoff", - "description": "Conversation-handoff document generator. Compacts the current session into a markdown handoff for a fresh agent \u2014 references existing artifacts (PRDs, plans, ADRs, issues, commits) by path/URL instead of duplicating them. Derived from Matt Pocock's MIT-licensed handoff with: (1) 3 stdlib Python tools (template generator tailored to 5 next-session emphases, artifact deduplicator across 5 categories of duplication, skill recommender matching content to 14 skills in this repo), (2) 4 references citing 7-8 sources (handoff structure, deduplication discipline, next-session skill matching, companion tooling), (3) cs-handoff-author persona agent + /cs:handoff slash command. Matt's no-duplication discipline + mktemp convention preserved verbatim per MIT.", + "description": "Conversation-handoff document generator. Compacts the current session into a markdown handoff for a fresh agent — references existing artifacts (PRDs, plans, ADRs, issues, commits) by path/URL instead of duplicating them. Derived from Matt Pocock's MIT-licensed handoff with: (1) 3 stdlib Python tools (template generator tailored to 5 next-session emphases, artifact deduplicator across 5 categories of duplication, skill recommender matching content to 14 skills in this repo), (2) 4 references citing 7-8 sources (handoff structure, deduplication discipline, next-session skill matching, companion tooling), (3) cs-handoff-author persona agent + /cs:handoff slash command. Matt's no-duplication discipline + mktemp convention preserved verbatim per MIT.", "version": "2.6.0", "author": { "name": "Alireza Rezvani" @@ -919,7 +919,7 @@ { "name": "capture-skill", "source": "./productivity/capture", - "description": "Brain-dump-to-action workspace skill. Routes vague captures into discoverable actions via classify\u2192cluster\u2192connect\u2192clarify intake. Path-B from megaprompt 05.", + "description": "Brain-dump-to-action workspace skill. Routes vague captures into discoverable actions via classify→cluster→connect→clarify intake. Path-B from megaprompt 05.", "version": "2.7.0", "author": { "name": "Alireza Rezvani" @@ -1153,7 +1153,7 @@ { "name": "aeo", "source": "./marketing-skill/skills/aeo", - "description": "Answer Engine Optimization (AEO) skill \u2014 optimize content to be cited by AI language models (ChatGPT, Perplexity, Claude, Gemini, Mistral) as authoritative sources. Distinct from SEO (which optimizes for search rankings), AEO optimizes for citation in LLM-generated responses. 3 stdlib Python tools (aeo_audit, aeo_optimizer, citation_tracker), 3 references citing 8 sources each, industry-aware thresholds for 8 industries (saas/healthcare/finance/legal/ecommerce/b2b/media/education). Ported from alirezarezvani/aeo-box.", + "description": "Answer Engine Optimization (AEO) skill — optimize content to be cited by AI language models (ChatGPT, Perplexity, Claude, Gemini, Mistral) as authoritative sources. Distinct from SEO (which optimizes for search rankings), AEO optimizes for citation in LLM-generated responses. 3 stdlib Python tools (aeo_audit, aeo_optimizer, citation_tracker), 3 references citing 8 sources each, industry-aware thresholds for 8 industries (saas/healthcare/finance/legal/ecommerce/b2b/media/education). Ported from alirezarezvani/aeo-box.", "version": "2.7.3", "author": { "name": "Alireza Rezvani" @@ -1175,7 +1175,7 @@ { "name": "security-guidance", "source": "./engineering/security-guidance", - "description": "PreToolUse security reminder hook for Claude Code. Catches 12 common security anti-patterns in Edit/Write/MultiEdit operations BEFORE they happen \u2014 command injection (exec, os.system, subprocess shell=True), XSS (innerHTML, dangerouslySetInnerHTML, document.write), SQL injection (f-string queries, .format), unsafe deserialization (pickle, yaml.unsafe_load), code injection (eval, new Function), and GitHub Actions workflow injection. Session-state caching prevents duplicate warnings; 30-day auto-cleanup. Disable per-session with ENABLE_SECURITY_REMINDER=0. Ported from David Dworken at Anthropic.", + "description": "PreToolUse security reminder hook for Claude Code. Catches 12 common security anti-patterns in Edit/Write/MultiEdit operations BEFORE they happen — command injection (exec, os.system, subprocess shell=True), XSS (innerHTML, dangerouslySetInnerHTML, document.write), SQL injection (f-string queries, .format), unsafe deserialization (pickle, yaml.unsafe_load), code injection (eval, new Function), and GitHub Actions workflow injection. Session-state caching prevents duplicate warnings; 30-day auto-cleanup. Disable per-session with ENABLE_SECURITY_REMINDER=0. Ported from David Dworken at Anthropic.", "version": "2.7.3", "author": { "name": "Alireza Rezvani" @@ -1273,6 +1273,24 @@ "grill-with-docs" ], "category": "commercial" + }, + { + "name": "universal-scraping-architect", + "source": "./engineering/universal-scraping-architect", + "description": "A universal scraping skill with intelligent routing, token budget tracking, and quota awareness. Supports Firecrawl and local Python extraction.", + "version": "2.1.2", + "author": { + "name": "Mehansh Barthwal" + }, + "keywords": [ + "scraping", + "data-extraction", + "firecrawl", + "beautifulsoup4", + "pandas", + "automation" + ], + "category": "development" } ] -} +} \ No newline at end of file diff --git a/.idea/.gitignore b/.idea/.gitignore new file mode 100644 index 00000000..30cf57ed --- /dev/null +++ b/.idea/.gitignore @@ -0,0 +1,10 @@ +# Default ignored files +/shelf/ +/workspace.xml +# Editor-based HTTP Client requests +/httpRequests/ +# Ignored default folder with query files +/queries/ +# Datasource local storage ignored files +/dataSources/ +/dataSources.local.xml diff --git a/.idea/claude-skills.iml b/.idea/claude-skills.iml new file mode 100644 index 00000000..55cb08ba --- /dev/null +++ b/.idea/claude-skills.iml @@ -0,0 +1,24 @@ + + + + + + + + + + + + + + + + + \ No newline at end of file diff --git a/.idea/inspectionProfiles/profiles_settings.xml b/.idea/inspectionProfiles/profiles_settings.xml new file mode 100644 index 00000000..105ce2da --- /dev/null +++ b/.idea/inspectionProfiles/profiles_settings.xml @@ -0,0 +1,6 @@ + + + + \ No newline at end of file diff --git a/.idea/modules.xml b/.idea/modules.xml new file mode 100644 index 00000000..8210aa20 --- /dev/null +++ b/.idea/modules.xml @@ -0,0 +1,8 @@ + + + + + + + + \ No newline at end of file diff --git a/.idea/vcs.xml b/.idea/vcs.xml new file mode 100644 index 00000000..35eb1ddf --- /dev/null +++ b/.idea/vcs.xml @@ -0,0 +1,6 @@ + + + + + + \ No newline at end of file diff --git a/engineering/universal-scraping-architect/LICENSE b/engineering/universal-scraping-architect/LICENSE deleted file mode 100644 index 52def8d8..00000000 --- a/engineering/universal-scraping-architect/LICENSE +++ /dev/null @@ -1,21 +0,0 @@ -MIT License - -Copyright (c) 2026 Mehansh Barthwal - -Permission is hereby granted, free of charge, to any person obtaining a copy -of this software and associated documentation files (the "Software"), to deal -in the Software without restriction, including without limitation the rights -to use, copy, modify, merge, publish, distribute, sublicense, and/or sell -copies of the Software, and to permit persons to whom the Software is -furnished to do so, subject to the following conditions: - -The above copyright notice and this permission notice shall be included in all -copies or substantial portions of the Software. - -THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -SOFTWARE. diff --git a/engineering/universal-scraping-architect/README.md b/engineering/universal-scraping-architect/README.md deleted file mode 100644 index 53aeef04..00000000 --- a/engineering/universal-scraping-architect/README.md +++ /dev/null @@ -1,212 +0,0 @@ -# Universal Scraping Architect - -> A robust, general-purpose scraping and data extraction framework designed as a reusable **Skill** for AI Agents and LLMs. - -[![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT) -[![Python 3.8+](https://img.shields.io/badge/python-3.8+-blue.svg)](https://www.python.org/downloads/) -[![Firecrawl Ready](https://img.shields.io/badge/Firecrawl-Ready-orange.svg)](https://firecrawl.dev) - ---- - -## What This Is - -Most scraping scripts are brittle one-offs that break the moment a page layout changes, a column gets renamed, or a file format shifts slightly. This framework is built differently — it treats every extraction task as a **complete data pipeline**, with intelligent routing, validation, token tracking, and clean outputs baked in from the start. - -It was built specifically to be dropped into an AI agent's skill library (like Claude's skill system), so that any scraping or data extraction task — whether it's a live public website, a local PDF, an Excel file, or a nested JSON — gets handled with the same consistent, robust approach every time. - ---- - -## Key Features - -**Intelligent Approach Routing** — Before writing a single line of code, the framework decides whether the task calls for Firecrawl (dynamic web, search-first, or bulk crawl), traditional local scraping (local files, static HTML, private data), or a hybrid of both. It states the choice and the reason. - -**Firecrawl Integration (5 Paths)** — Full support for Firecrawl's CLI, REST API, and SDK across five clearly defined paths: live web data, app/product integration, finished deliverables, auth-only setup, and direct REST without installation. - -**Token Budget Tracking** — Every extraction that feeds an LLM estimates token volume against a configurable context limit before processing, warning early if the output is over budget. - -**Firecrawl Quota Safety** — Single-key rule enforced by default. Estimates page/credit usage before large crawl jobs and only prompts for a new key when the current one is actually exhausted or a job would genuinely exceed ~1000 requests. - -**Validation at Every Step** — Required fields are checked, row counts are logged, empty outputs are caught, and duplicate rows are flagged before anything gets saved. - -**Checkpointing for Large Jobs** — Progress is saved during multi-page or multi-file extractions so a failure mid-job doesn't mean starting from scratch. - -**Security and Ethical Scraping** — API keys are loaded from environment variables only, never hardcoded or printed. robots.txt, rate limits, and terms of service are respected. Private files are never sent to external APIs without explicit approval. - ---- - -## Supported Sources - -| Category | Examples | -|---|---| -| **Live Web** | Public URLs, dynamic pages, SPAs, paginated sites | -| **Search-First** | Topic queries, keyword discovery, entity research | -| **Bulk Crawl** | Full site sections, documentation sites, large domains | -| **Local Files** | PDF, DOCX, Excel (.xlsx/.xls), CSV, JSON, XML, ZIP | -| **Scanned Docs** | OCR pipelines for image-based PDFs | -| **APIs** | REST APIs, paginated API responses, JSON data feeds | -| **Databases** | CSV exports, SQLite, structured data dumps | - ---- - -## Output Formats - -The framework can save clean outputs as CSV, Excel, JSON, Markdown, TXT, SQLite, Parquet, or HTML depending on what the task needs. If not specified, it defaults sensibly — CSV for structured tabular data, JSON for nested structures, Markdown for clean web page text. - ---- - -## Project Structure - -``` -universal-scraping-architect/ -├── SKILL.md # Core LLM skill definition (the agent's system prompt) -├── README.md # This file -├── LICENSE # MIT License -├── .gitignore # Standard Python/project ignores -├── requirements.txt # Python dependencies -└── examples/ - ├── firecrawl_example.py # Path C workflow: Firecrawl → clean markdown output - └── local_bs4_example.py # Traditional scraping: static HTML → validated CSV -``` - ---- - -## How To Use This As An AI Skill - -Drop `SKILL.md` into your agent's system prompt or tool context. The agent will then adopt the Universal Scraping Architect approach for any extraction task — routing correctly, validating outputs, tracking tokens, and producing copy-paste-ready pipeline code rather than fragile one-offs. - -This is how it appears in Claude's skill system: - -```yaml -name: universal-scraping-architect -description: Use this skill for any scraping, crawling, extraction, parsing, web - research, document processing, dataset preparation, Firecrawl workflow, - local file extraction, API extraction, PDF/Excel/CSV/JSON/XML parsing, - validation-heavy data pipeline, or repeatable clean-output scraping task. -``` - ---- - -## Quick Start (For Developers Running the Examples) - -**Install dependencies:** - -```bash -pip install -r requirements.txt -``` - -**For Firecrawl workflows, set your API key:** - -```bash -# Linux / macOS -export FIRECRAWL_API_KEY="fc-YOUR_API_KEY_HERE" - -# Windows PowerShell -$env:FIRECRAWL_API_KEY = "fc-YOUR_API_KEY_HERE" - -# Windows Command Prompt -set FIRECRAWL_API_KEY=fc-YOUR_API_KEY_HERE -``` - -Or add it to a `.env` file in your project root (never commit this file): - -```dotenv -FIRECRAWL_API_KEY=fc-YOUR_API_KEY_HERE -``` - -**Run the Firecrawl example:** - -```bash -python examples/firecrawl_example.py -``` - -**Run the local scraping example:** - -```bash -python examples/local_bs4_example.py -``` - ---- - -## The 15-Step Pipeline - -Every task the framework handles follows this sequence: - -1. Understand the source -2. Choose the most appropriate extraction approach -3. Configure task-specific settings -4. Extract safely -5. Handle pagination, layout changes, dynamic content, or file variations -6. Clean the extracted data -7. Normalize structure and field names -8. Validate the result -9. Track token/data volume if LLM processing is involved -10. Estimate Firecrawl usage/quota before large Firecrawl jobs -11. Handle errors clearly -12. Save clean outputs -13. Save logs and checkpoints when useful -14. Print a final summary -15. Explain what was done and what can be customized - ---- - -## Firecrawl Path Reference - -| Path | When To Use | -|---|---| -| **Path A** | Need live web data right now during the current session | -| **Path B** | Building an app or product that calls Firecrawl from code | -| **Path C** | Need a finished deliverable — research brief, SEO audit, lead list, etc. | -| **Path D** | Need to set up an account or API key first | -| **Path E** | Don't want to install anything — use the REST API directly | - -Install command (covers all paths): - -```bash -npx -y firecrawl-cli@latest init --all --browser -``` - ---- - -## When To Use Firecrawl vs. Local Scraping - -**Use Firecrawl when:** -- The source is a public URL and you want clean, reliable extraction -- The page is dynamic (JavaScript-rendered, SPA, requires interaction) -- You need search-first discovery before you know the URLs -- You're crawling many pages across a domain -- You want to wire Firecrawl into an app or agentic workflow - -**Use local/traditional scraping when:** -- The source is a local file (PDF, Excel, CSV, JSON, XML) -- The data is private or sensitive and shouldn't leave your machine -- You're using official downloads/APIs where scraping isn't needed -- Simple static HTML where Firecrawl would be overkill -- Custom parsing logic with pandas, pdfplumber, openpyxl, etc. is the right fit - -**Use a hybrid when:** -- Firecrawl handles web extraction, then Python cleans and structures the output -- Firecrawl discovers URLs, then local code processes and saves the dataset -- Web content gets merged with local files - ---- - -## Security Notes - -- API keys are always loaded from environment variables — never hardcoded -- `.env` files are in `.gitignore` and should never be committed -- Real keys are never printed in logs or included in output files -- Private or sensitive files are never sent to Firecrawl without explicit user approval -- One active Firecrawl API key is used at a time (no rotation by default) -- robots.txt and rate limits are respected - ---- - -## Author - -**Mehansh Barthwal** - ---- - -## License - -This project is licensed under the MIT License — see [LICENSE](LICENSE) for details. diff --git a/engineering/universal-scraping-architect/agents/agents/cs-scraping-architect.md b/engineering/universal-scraping-architect/agents/cs-scraping-architect.md similarity index 100% rename from engineering/universal-scraping-architect/agents/agents/cs-scraping-architect.md rename to engineering/universal-scraping-architect/agents/cs-scraping-architect.md diff --git a/engineering/universal-scraping-architect/commands/commands/cs-scrape.md b/engineering/universal-scraping-architect/commands/cs-scrape.md similarity index 100% rename from engineering/universal-scraping-architect/commands/commands/cs-scrape.md rename to engineering/universal-scraping-architect/commands/cs-scrape.md diff --git a/engineering/universal-scraping-architect/references/references/parsing-and-data-extraction.md b/engineering/universal-scraping-architect/references/references/parsing-and-data-extraction.md deleted file mode 100644 index 221765d2..00000000 --- a/engineering/universal-scraping-architect/references/references/parsing-and-data-extraction.md +++ /dev/null @@ -1,10 +0,0 @@ -# Parsing and Data Extraction Standards - -Detailed strategies for BeautifulSoup4 and Pandas integration for cleaning scraped data. - -### Authoritative Sources -1. [BeautifulSoup4 Documentation](https://www.crummy.com/software/BeautifulSoup/bs4/doc/) -2. [Pandas Data Cleaning Guide](https://pandas.pydata.org/docs/user_guide/10min.html) -3. [W3C HTML Living Standard](https://html.spec.whatwg.org/multipage/) -4. [CSS Selectors Level 4 (W3C)](https://www.w3.org/TR/selectors-4/) -5. [Mozilla MDN - HTTP Headers (User-Agent)](https://developer.mozilla.org/en-US/docs/Web/HTTP/Headers/User-Agent) \ No newline at end of file diff --git a/engineering/universal-scraping-architect/references/references/references/scraping-ethics-security.md b/engineering/universal-scraping-architect/references/scraping-ethics-security.md similarity index 100% rename from engineering/universal-scraping-architect/references/references/references/scraping-ethics-security.md rename to engineering/universal-scraping-architect/references/scraping-ethics-security.md