diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 1f1f8711..77eec9ff 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -4,7 +4,7 @@ "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" }, - "description": "329 production-ready skill packages for Claude AI across 14 domains: engineering advanced (76 — incl. 4 Matt Pocock-derived productivity skills + v2.7.3 security-guidance PreToolUse hook), engineering core (51), marketing (47 — incl. v2.7.3 AEO/Answer Engine Optimization), c-level advisory (66), product (17), regulatory/QMS (18), project management (9), business growth (5), finance (4), productivity (6), marketing top-level (2, v2.7.0), research (8, v2.7.0), business-operations (7, v2.8.0), and commercial (8, v2.8.0). Includes ~444 Python tools, ~598 reference documents, 49+ agents, 79+ slash commands.", + "description": "329 production-ready skill packages for Claude AI across 14 domains: engineering advanced (76 — incl. 4 Matt Pocock-derived productivity skills + v2.7.3 security-guidance PreToolUse hook), engineering core (51), marketing (47 — incl. v2.7.3 AEO/Answer Engine Optimization), c-level advisory (66), product (17), regulatory/QMS (18), project management (9), business growth (5), finance (4), productivity (4, v2.7.0), marketing top-level (2, v2.7.0), research (8, v2.7.0), business-operations (7, v2.8.0), and commercial (8, v2.8.0). Includes ~441 Python tools, ~594 reference documents, 48+ agents, 77+ slash commands.", "homepage": "https://github.com/alirezarezvani/claude-skills", "repository": "https://github.com/alirezarezvani/claude-skills", "metadata": { @@ -1313,6 +1313,24 @@ "grill-with-docs" ], "category": "commercial" + }, + { + "name": "universal-scraping-architect", + "source": "./engineering/universal-scraping-architect", + "description": "A universal scraping skill with intelligent routing, token budget tracking, and quota awareness. Supports Firecrawl and local Python extraction.", + "version": "2.1.2", + "author": { + "name": "Mehansh Barthwal" + }, + "keywords": [ + "scraping", + "data-extraction", + "firecrawl", + "beautifulsoup4", + "pandas", + "automation" + ], + "category": "development" } ] -} +} \ No newline at end of file diff --git a/.idea/.gitignore b/.idea/.gitignore new file mode 100644 index 00000000..30cf57ed --- /dev/null +++ b/.idea/.gitignore @@ -0,0 +1,10 @@ +# Default ignored files +/shelf/ +/workspace.xml +# Editor-based HTTP Client requests +/httpRequests/ +# Ignored default folder with query files +/queries/ +# Datasource local storage ignored files +/dataSources/ +/dataSources.local.xml diff --git a/.idea/claude-skills.iml b/.idea/claude-skills.iml new file mode 100644 index 00000000..55cb08ba --- /dev/null +++ b/.idea/claude-skills.iml @@ -0,0 +1,24 @@ + + + + + + + + + + + + + + + + + \ No newline at end of file diff --git a/.idea/inspectionProfiles/profiles_settings.xml b/.idea/inspectionProfiles/profiles_settings.xml new file mode 100644 index 00000000..105ce2da --- /dev/null +++ b/.idea/inspectionProfiles/profiles_settings.xml @@ -0,0 +1,6 @@ + + + + \ No newline at end of file diff --git a/.idea/modules.xml b/.idea/modules.xml new file mode 100644 index 00000000..8210aa20 --- /dev/null +++ b/.idea/modules.xml @@ -0,0 +1,8 @@ + + + + + + + + \ No newline at end of file diff --git a/.idea/vcs.xml b/.idea/vcs.xml new file mode 100644 index 00000000..35eb1ddf --- /dev/null +++ b/.idea/vcs.xml @@ -0,0 +1,6 @@ + + + + + + \ No newline at end of file diff --git a/engineering/universal-scraping-architect/.claude-plugin/plugin.json b/engineering/universal-scraping-architect/.claude-plugin/plugin.json new file mode 100644 index 00000000..70e889ea --- /dev/null +++ b/engineering/universal-scraping-architect/.claude-plugin/plugin.json @@ -0,0 +1,16 @@ +{ + "name": "universal-scraping-architect", + "description": "A universal scraping skill with intelligent routing, token budget tracking, and quota awareness.", + "version": "2.1.2", + "author": { + "name": "Mehansh Barthwal", + "url": "https://github.com/mehanshbarthwal-lab" + }, + "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/universal-scraping-architect", + "repository": "https://github.com/alirezarezvani/claude-skills", + "license": "MIT", + "skills": "./", + "dependencies": { + "python": ["firecrawl", "pandas", "requests", "beautifulsoup4"] + } +} \ No newline at end of file diff --git a/engineering/universal-scraping-architect/.idea/.gitignore b/engineering/universal-scraping-architect/.idea/.gitignore new file mode 100644 index 00000000..30cf57ed --- /dev/null +++ b/engineering/universal-scraping-architect/.idea/.gitignore @@ -0,0 +1,10 @@ +# Default ignored files +/shelf/ +/workspace.xml +# Editor-based HTTP Client requests +/httpRequests/ +# Ignored default folder with query files +/queries/ +# Datasource local storage ignored files +/dataSources/ +/dataSources.local.xml diff --git a/engineering/universal-scraping-architect/.idea/inspectionProfiles/profiles_settings.xml b/engineering/universal-scraping-architect/.idea/inspectionProfiles/profiles_settings.xml new file mode 100644 index 00000000..105ce2da --- /dev/null +++ b/engineering/universal-scraping-architect/.idea/inspectionProfiles/profiles_settings.xml @@ -0,0 +1,6 @@ + + + + \ No newline at end of file diff --git a/engineering/universal-scraping-architect/.idea/misc.xml b/engineering/universal-scraping-architect/.idea/misc.xml new file mode 100644 index 00000000..187eec5d --- /dev/null +++ b/engineering/universal-scraping-architect/.idea/misc.xml @@ -0,0 +1,7 @@ + + + + + + \ No newline at end of file diff --git a/engineering/universal-scraping-architect/.idea/modules.xml b/engineering/universal-scraping-architect/.idea/modules.xml new file mode 100644 index 00000000..940fef20 --- /dev/null +++ b/engineering/universal-scraping-architect/.idea/modules.xml @@ -0,0 +1,8 @@ + + + + + + + + \ No newline at end of file diff --git a/engineering/universal-scraping-architect/.idea/universal-scraping-architect.iml b/engineering/universal-scraping-architect/.idea/universal-scraping-architect.iml new file mode 100644 index 00000000..3ab2d77f --- /dev/null +++ b/engineering/universal-scraping-architect/.idea/universal-scraping-architect.iml @@ -0,0 +1,14 @@ + + + + + + + + + + + + \ No newline at end of file diff --git a/engineering/universal-scraping-architect/.idea/vcs.xml b/engineering/universal-scraping-architect/.idea/vcs.xml new file mode 100644 index 00000000..b2bdec2d --- /dev/null +++ b/engineering/universal-scraping-architect/.idea/vcs.xml @@ -0,0 +1,6 @@ + + + + + + \ No newline at end of file diff --git a/engineering/universal-scraping-architect/SKILL.md b/engineering/universal-scraping-architect/SKILL.md new file mode 100644 index 00000000..32028e98 --- /dev/null +++ b/engineering/universal-scraping-architect/SKILL.md @@ -0,0 +1,58 @@ +--- +name: "universal-scraping-architect" +description: "Use for web scraping, crawling, document extraction, API parsing, or building validation-heavy data pipelines using Firecrawl or local Python scripts." +--- + +# Universal Scraping Architect + +You are an expert web scraping and data extraction engineer. Your goal is to design complete, robust data pipelines with intelligent routing, validation, and token budget tracking—not brittle one-off scripts. + +**Dependency Notice:** This skill utilizes `firecrawl`, `pandas`, `requests`, and `beautifulsoup4`. It uses a BYOK (Bring Your Own Key) pattern for Firecrawl. API keys must only be loaded via environment variables. + +## Before Starting +**Check for context first:** +If `project-context.md` exists, read it before asking questions. Determine the target data format, scale of extraction, and deployment environment before writing any code. + +## How This Skill Works + +This skill supports 3 extraction modes based on intelligent routing: + +### Mode 1: API-Driven (Firecrawl) +Use when the source is a public URL, heavily dynamic (JS/SPA), requires search-first discovery, or involves bulk crawling across a domain. +### Mode 2: Local Python (Traditional) +Use when extracting from local files (PDF, Excel, CSV), the data is private/sensitive, or the target is a simple static HTML page where Firecrawl is overkill. +### Mode 3: Hybrid Pipeline +Use when Firecrawl handles URL discovery/web extraction, but local Python (Pandas) is required to clean, normalize, and structure the output before saving. + +## The Extraction Pipeline + +When executing a scraping task, always follow this sequence: +1. **Route the Approach:** Explicitly state whether Firecrawl or Local Python is being used and why. +2. **Track Budgets:** Estimate Firecrawl API quotas or LLM token context limits before executing large jobs. +3. **Extract Safely:** Implement checkpointing for multi-page jobs. Handle pagination and dynamic layouts gracefully. +4. **Validate & Clean:** Enforce required fields, catch empty outputs, flag duplicates, and normalize field names. +5. **Format:** Default to CSV for tabular data, JSON for nested structures, and Markdown for clean text. + +## Proactive Triggers + +Surface these issues WITHOUT being asked when you notice them in context: +- **Hardcoded API Keys** → Flag immediately and rewrite to use `os.getenv('FIRECRAWL_API_KEY')`. +- **Private Data Leakage** → If the user asks to send local, sensitive files to an external API, flag the privacy risk and suggest Mode 2 (Local Python). +- **Missing Pagination** → If the target implies hundreds of records but no pagination logic is requested, flag it and add checkpointing. + +## Output Artifacts + +| When you ask for... | You get... | +|---------------------|------------| +| "Scrape this site" | A fully validated Python extraction script with routing logic and error handling. | +| "Get data from this table" | A clean CSV/JSON dataset with a summary log of row counts and empty values. | +| "Crawl these docs" | A Markdown deliverable chunked for LLM token limits. | + +## Anti-Patterns +- **Brittle Selectors:** Never use highly nested CSS selectors (e.g., `div > span > ul > li:nth-child(3)`). Use data attributes or robust structural anchors. +- **Ignoring Etiquette:** Never scrape without checking `robots.txt` or implementing sensible rate limits. +- **No Validation:** Never blindly write scraped data to a file without checking if the array is empty or missing critical keys. + +## Related Skills +- **data-cleaning**: Use when the scraped data requires complex statistical normalization or deduplication. +- **browser-automation**: Use for highly interactive scraping requiring user emulation (clicks, logins) where Firecrawl is insufficient. \ No newline at end of file diff --git a/engineering/universal-scraping-architect/agents/cs-scraping-architect.md b/engineering/universal-scraping-architect/agents/cs-scraping-architect.md new file mode 100644 index 00000000..15bf6938 --- /dev/null +++ b/engineering/universal-scraping-architect/agents/cs-scraping-architect.md @@ -0,0 +1,6 @@ +--- +name: cs-scraping-architect +description: Expert persona for web scraping and data pipeline design. +--- +# cs-scraping-architect +Use this agent when you need to design a complex extraction strategy or debug scraping scripts. \ No newline at end of file diff --git a/engineering/universal-scraping-architect/commands/cs-scrape.md b/engineering/universal-scraping-architect/commands/cs-scrape.md new file mode 100644 index 00000000..a939de83 --- /dev/null +++ b/engineering/universal-scraping-architect/commands/cs-scrape.md @@ -0,0 +1,6 @@ +--- +name: cs-scrape +description: Execute a scraping task for a specific URL. +--- +# /cs-scrape [url] +Triggers the scraping architect to analyze and extract data from the target URL. \ No newline at end of file diff --git a/engineering/universal-scraping-architect/references/firecrawl-technical-guide.md b/engineering/universal-scraping-architect/references/firecrawl-technical-guide.md new file mode 100644 index 00000000..de4d2bf7 --- /dev/null +++ b/engineering/universal-scraping-architect/references/firecrawl-technical-guide.md @@ -0,0 +1,10 @@ +# Firecrawl Technical Guide + +This document covers the technical integration patterns for the Firecrawl API within the Universal Scraping Architect. + +### Authoritative Sources +1. [Firecrawl API Documentation](https://docs.firecrawl.dev/api-reference/introduction) +2. [Firecrawl SDK for Python](https://github.com/mendableai/firecrawl-py) +3. [REST API Design Best Practices (Microsoft)](https://learn.microsoft.com/en-us/azure/architecture/best-practices/api-design) +4. [Handling API Rate Limits (Cloudflare)](https://developers.cloudflare.com/fundamentals/api/reference/rate-limits/) +5. [JSON Schema Standard](https://json-schema.org/specification.html) \ No newline at end of file diff --git a/engineering/universal-scraping-architect/references/scraping-ethics-security.md b/engineering/universal-scraping-architect/references/scraping-ethics-security.md new file mode 100644 index 00000000..bc09c71e --- /dev/null +++ b/engineering/universal-scraping-architect/references/scraping-ethics-security.md @@ -0,0 +1,10 @@ +# Scraping Ethics, Robots.txt, and Security + +Guidelines for ethical data collection and securing scraping pipelines. + +### Authoritative Sources +1. [The Robots Exclusion Protocol (RFC 9309)](https://datatracker.ietf.org/doc/rfc9309/) +2. [OWASP Automated Threats to Web Applications](https://owasp.org/www-project-automated-threats-to-web-applications/) +3. [Scraping Ethics Best Practices (Ethical Web Scraping)](https://www.ethicalwebscraping.org/) +4. [Legal Aspects of Web Scraping (Lexology)](https://www.lexology.com/library/detail.aspx?g=e6e0287a-62ad-4d6d-852a-9e535e69e6b4) +5. [Cloudflare Bot Management Overview](https://www.cloudflare.com/en-gb/pg-lp/bot-management-for-everyone/) \ No newline at end of file diff --git a/engineering/universal-scraping-architect/scripts/firecrawl_example.py b/engineering/universal-scraping-architect/scripts/firecrawl_example.py new file mode 100644 index 00000000..f7b0a9c8 --- /dev/null +++ b/engineering/universal-scraping-architect/scripts/firecrawl_example.py @@ -0,0 +1,157 @@ +""" +Universal Scraping Architect: Firecrawl Example (Path C — Repeatable Deliverable) +=================================================================================== +Extracts clean markdown content from a target URL using the Firecrawl SDK. + +Demonstrates: + - Safe API key loading from environment + - Token budget tracking before LLM processing + - Firecrawl quota awareness logging + - Structured error handling + - Clean output saving with validation + +Usage: + export FIRECRAWL_API_KEY="fc-YOUR_KEY_HERE" # Linux/macOS + $env:FIRECRAWL_API_KEY = "fc-YOUR_KEY_HERE" # Windows PowerShell + python examples/firecrawl_example.py +""" + +import os +import sys +from datetime import datetime +from firecrawl import FirecrawlApp + +# ============================================================================= +# CONFIG — Edit these for your task +# ============================================================================= +TARGET_URL = "https://example.com/research-data" +OUTPUT_FILE = "clean_extraction.md" +LOG_FILE = "firecrawl_run.log" + +# Firecrawl / Token settings +FIRECRAWL_API_KEY_ENV = "FIRECRAWL_API_KEY" +TOKEN_CONTEXT_LIMIT = 100_000 # Adjust to your model's context window +RESERVED_OUTPUT_TOKENS = 4_000 # Tokens held back for the model's response + + +# ============================================================================= +# HELPERS +# ============================================================================= + +def log(message: str, level: str = "INFO"): + """Prints a timestamped log line and appends it to the log file.""" + timestamp = datetime.now().strftime("%Y-%m-%d %H:%M:%S") + line = f"[{timestamp}] [{level}] {message}" + print(line) + with open(LOG_FILE, "a", encoding="utf-8") as f: + f.write(line + "\n") + + +def check_environment() -> str: + """Loads the Firecrawl API key from the environment. Fails fast if missing.""" + api_key = os.getenv(FIRECRAWL_API_KEY_ENV) + if not api_key: + log( + f"Missing environment variable: {FIRECRAWL_API_KEY_ENV}. " + "Set it before running this script.", + level="ERROR" + ) + sys.exit(1) + log("API key loaded successfully from environment.") + return api_key + + +def estimate_tokens(text: str) -> int: + """Rough token estimate based on character count (characters / 4).""" + return len(text) // 4 + + +def check_token_budget(text: str) -> bool: + """ + Estimates token usage and warns if over budget. + Returns True if within budget, False if over. + """ + estimated = estimate_tokens(text) + available = TOKEN_CONTEXT_LIMIT - RESERVED_OUTPUT_TOKENS + log(f"Token estimate: {estimated:,} / {available:,} available tokens.") + if estimated > available: + log( + f"OVER TOKEN BUDGET by {estimated - available:,} tokens. " + "Consider chunking the output before passing to an LLM.", + level="WARNING" + ) + return False + log("Within token budget.") + return True + + +def validate_content(content: str) -> bool: + """Basic validation — checks the response is non-empty.""" + if not content or not content.strip(): + log("Validation failed: empty content received from Firecrawl.", level="ERROR") + return False + log(f"Validation passed. Content length: {len(content):,} characters.") + return True + + +def save_output(content: str, path: str): + """Saves validated content to the output path.""" + with open(path, "w", encoding="utf-8") as f: + f.write(content) + log(f"Output saved to: {path}") + + +# ============================================================================= +# MAIN +# ============================================================================= + +def main(): + log("=" * 60) + log("Universal Scraping Architect — Firecrawl Path C") + log("=" * 60) + + # Step 1: Environment check + api_key = check_environment() + + # Step 2: Initialise Firecrawl client + app = FirecrawlApp(api_key=api_key) + log(f"Firecrawl client initialised. Target: {TARGET_URL}") + + # Step 3: Scrape + try: + log("Starting scrape...") + result = app.scrape_url(TARGET_URL, params={"formats": ["markdown"]}) + markdown_content = result.get("markdown", "") + except Exception as e: + log(f"Firecrawl scrape failed: {e}", level="ERROR") + log( + "Tip: if this is a quota or auth error, check your FIRECRAWL_API_KEY " + "and run `firecrawl --status` to inspect account state.", + level="WARNING" + ) + sys.exit(1) + + # Step 4: Validate + if not validate_content(markdown_content): + sys.exit(1) + + # Step 5: Token budget check + check_token_budget(markdown_content) + + # Step 6: Save clean output + save_output(markdown_content, OUTPUT_FILE) + + # Step 7: Final summary + log("=" * 60) + log("EXTRACTION COMPLETE") + log(f" Source URL : {TARGET_URL}") + log(f" Output file : {OUTPUT_FILE}") + log(f" Characters : {len(markdown_content):,}") + log(f" Est. tokens : {estimate_tokens(markdown_content):,}") + log(f" Log file : {LOG_FILE}") + log("=" * 60) + log("Customisable: TARGET_URL, OUTPUT_FILE, TOKEN_CONTEXT_LIMIT, formats.") + + +if __name__ == "__main__": + main() diff --git a/engineering/universal-scraping-architect/scripts/local_bs4_example.py b/engineering/universal-scraping-architect/scripts/local_bs4_example.py new file mode 100644 index 00000000..0d70c4ce --- /dev/null +++ b/engineering/universal-scraping-architect/scripts/local_bs4_example.py @@ -0,0 +1,226 @@ +""" +Universal Scraping Architect: Traditional / Local Scraping Example +================================================================== +Extracts a data table from a static HTML page using Requests and BeautifulSoup, +then validates, cleans, and saves to CSV. + +Demonstrates: + - Safe HTTP fetching with headers, timeouts, and retry logic + - HTML table parsing with pandas + - Column name normalisation to snake_case + - Required field validation before saving + - Structured error handling at every stage + - Clean, logged output + +Usage: + pip install -r requirements.txt + python examples/local_bs4_example.py +""" + +import sys +import time +from datetime import datetime + +import pandas as pd +import requests +from bs4 import BeautifulSoup + +# ============================================================================= +# CONFIG — Edit these for your task +# ============================================================================= +TARGET_URL = "https://example.com/macroeconomic-indicators/energy-prices" +OUTPUT_FILE = "energy_price_data.csv" +LOG_FILE = "local_scrape_run.log" + +# HTTP settings +USER_AGENT = "UniversalScrapingArchitect/1.0 (contact: your@email.com)" +TIMEOUT_SECONDS = 15 +MAX_RETRIES = 3 +RETRY_DELAY_SECONDS = 2 + +# Validation +REQUIRED_COLUMNS = ["date", "price_index", "yoy_change"] +TABLE_SELECTOR = {"id": "indicator-data"} # Edit to match your target table's HTML attributes + + +# ============================================================================= +# HELPERS +# ============================================================================= + +def log(message: str, level: str = "INFO"): + """Prints a timestamped log line and appends it to the log file.""" + timestamp = datetime.now().strftime("%Y-%m-%d %H:%M:%S") + line = f"[{timestamp}] [{level}] {message}" + print(line) + with open(LOG_FILE, "a", encoding="utf-8") as f: + f.write(line + "\n") + + +def safe_get(url: str) -> str: + """ + Fetches HTML from a URL with polite headers, a timeout, and retry logic. + Raises on non-2xx status. + """ + headers = {"User-Agent": USER_AGENT} + + for attempt in range(1, MAX_RETRIES + 1): + try: + log(f"Attempt {attempt}/{MAX_RETRIES}: GET {url}") + response = requests.get(url, headers=headers, timeout=TIMEOUT_SECONDS) + response.raise_for_status() + log(f"HTTP {response.status_code} — content length: {len(response.text):,} chars.") + return response.text + except requests.exceptions.HTTPError as e: + log(f"HTTP error on attempt {attempt}: {e}", level="ERROR") + except requests.exceptions.ConnectionError as e: + log(f"Connection error on attempt {attempt}: {e}", level="ERROR") + except requests.exceptions.Timeout: + log(f"Timeout on attempt {attempt} after {TIMEOUT_SECONDS}s.", level="WARNING") + except requests.exceptions.RequestException as e: + log(f"Request failed on attempt {attempt}: {e}", level="ERROR") + + if attempt < MAX_RETRIES: + log(f"Waiting {RETRY_DELAY_SECONDS}s before retry...") + time.sleep(RETRY_DELAY_SECONDS) + + raise RuntimeError(f"All {MAX_RETRIES} attempts failed for URL: {url}") + + +def find_table(html: str) -> BeautifulSoup: + """ + Parses the HTML and returns the target table element. + Adjust TABLE_SELECTOR to match the table you need. + """ + soup = BeautifulSoup(html, "html.parser") + table = soup.find("table", TABLE_SELECTOR) + if not table: + raise ValueError( + f"Target table not found. Selector used: {TABLE_SELECTOR}. " + "Inspect the page source and update TABLE_SELECTOR in CONFIG." + ) + log("Target table found in HTML.") + return table + + +def parse_table(table) -> pd.DataFrame: + """Converts a BeautifulSoup table element into a pandas DataFrame.""" + df = pd.read_html(str(table))[0] + log(f"Parsed table: {len(df)} rows, {len(df.columns)} columns.") + return df + + +def clean_column_names(df: pd.DataFrame) -> pd.DataFrame: + """Normalises all column names to snake_case.""" + df.columns = ( + df.columns + .str.strip() + .str.lower() + .str.replace(r"\s+", "_", regex=True) + .str.replace(r"[^\w]", "", regex=True) + ) + log(f"Columns after normalisation: {list(df.columns)}") + return df + + +def clean_data(df: pd.DataFrame) -> pd.DataFrame: + """ + Applies general cleaning rules. + Extend this function for your task-specific cleaning needs. + """ + # Strip whitespace from string columns + for col in df.select_dtypes(include="object").columns: + df[col] = df[col].str.strip() + + # Drop fully empty rows + before = len(df) + df = df.dropna(how="all") + dropped = before - len(df) + if dropped: + log(f"Dropped {dropped} fully empty rows.", level="WARNING") + + return df + + +def validate(df: pd.DataFrame) -> bool: + """ + Checks that the DataFrame is non-empty and contains all required columns. + Returns True if valid, False otherwise. + """ + if df.empty: + log("Validation failed: DataFrame is empty.", level="ERROR") + return False + + missing = [col for col in REQUIRED_COLUMNS if col not in df.columns] + if missing: + log( + f"Validation failed: missing required columns: {missing}. " + f"Available columns: {list(df.columns)}", + level="ERROR" + ) + return False + + log(f"Validation passed. {len(df)} rows, {len(df.columns)} columns.") + return True + + +def save_output(df: pd.DataFrame, path: str): + """Saves the validated DataFrame to CSV.""" + df.to_csv(path, index=False, encoding="utf-8") + log(f"Output saved to: {path}") + + +# ============================================================================= +# MAIN +# ============================================================================= + +def main(): + log("=" * 60) + log("Universal Scraping Architect — Traditional / Local Scraping") + log("=" * 60) + + # Step 1: Fetch + try: + html = safe_get(TARGET_URL) + except RuntimeError as e: + log(str(e), level="ERROR") + sys.exit(1) + + # Step 2: Find table + try: + table = find_table(html) + except ValueError as e: + log(str(e), level="ERROR") + sys.exit(1) + + # Step 3: Parse + try: + df = parse_table(table) + except Exception as e: + log(f"Failed to parse table into DataFrame: {e}", level="ERROR") + sys.exit(1) + + # Step 4: Clean + df = clean_column_names(df) + df = clean_data(df) + + # Step 5: Validate + if not validate(df): + sys.exit(1) + + # Step 6: Save + save_output(df, OUTPUT_FILE) + + # Step 7: Final summary + log("=" * 60) + log("EXTRACTION COMPLETE") + log(f" Source URL : {TARGET_URL}") + log(f" Output file : {OUTPUT_FILE}") + log(f" Rows saved : {len(df)}") + log(f" Columns : {list(df.columns)}") + log(f" Log file : {LOG_FILE}") + log("=" * 60) + log("Customisable: TARGET_URL, OUTPUT_FILE, TABLE_SELECTOR, REQUIRED_COLUMNS.") + + +if __name__ == "__main__": + main() diff --git a/engineering/universal-scraping-architect/scripts/scripts/validate_extraction.py b/engineering/universal-scraping-architect/scripts/scripts/validate_extraction.py new file mode 100644 index 00000000..97dd4e5a --- /dev/null +++ b/engineering/universal-scraping-architect/scripts/scripts/validate_extraction.py @@ -0,0 +1,44 @@ +#!/usr/bin/env python3 +""" +validate_extraction.py +Intelligent stdlib-only validation for JSON structures. +""" +import argparse +import json +import sys +import os + + +def validate_json(file_path): + if not os.path.exists(file_path): + return {"status": "error", "message": f"File {file_path} not found."} + + try: + with open(file_path, 'r', encoding='utf-8') as f: + data = json.load(f) + # Basic validation: ensure it's not empty and is a list/dict + if not data: + return {"status": "warning", "message": "JSON file is empty."} + return {"status": "ok", "message": "JSON is valid and well-formed."} + except json.JSONDecodeError as e: + return {"status": "error", "message": f"Invalid JSON: {str(e)}"} + + +def main(): + parser = argparse.ArgumentParser(description="Standard Library JSON Validator") + parser.add_argument("file", help="Path to JSON file to validate") + parser.add_argument("--json", action="store_true", help="Output results in JSON format") + + args = parser.parse_args() + result = validate_json(args.file) + + if args.json: + print(json.dumps(result, indent=2)) + else: + print(f"[{result['status'].upper()}] {result['message']}") + + sys.exit(0 if result['status'] == "ok" else 1) + + +if __name__ == "__main__": + main() \ No newline at end of file