+
+
\ No newline at end of file
diff --git a/.idea/inspectionProfiles/profiles_settings.xml b/.idea/inspectionProfiles/profiles_settings.xml
new file mode 100644
index 00000000..105ce2da
--- /dev/null
+++ b/.idea/inspectionProfiles/profiles_settings.xml
@@ -0,0 +1,6 @@
+
+
+
+
+
+
\ No newline at end of file
diff --git a/.idea/modules.xml b/.idea/modules.xml
new file mode 100644
index 00000000..8210aa20
--- /dev/null
+++ b/.idea/modules.xml
@@ -0,0 +1,8 @@
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/.idea/vcs.xml b/.idea/vcs.xml
new file mode 100644
index 00000000..35eb1ddf
--- /dev/null
+++ b/.idea/vcs.xml
@@ -0,0 +1,6 @@
+
+
+
+
+
+
\ No newline at end of file
diff --git a/engineering/universal-scraping-architect/.claude-plugin/plugin.json b/engineering/universal-scraping-architect/.claude-plugin/plugin.json
new file mode 100644
index 00000000..70e889ea
--- /dev/null
+++ b/engineering/universal-scraping-architect/.claude-plugin/plugin.json
@@ -0,0 +1,16 @@
+{
+ "name": "universal-scraping-architect",
+ "description": "A universal scraping skill with intelligent routing, token budget tracking, and quota awareness.",
+ "version": "2.1.2",
+ "author": {
+ "name": "Mehansh Barthwal",
+ "url": "https://github.com/mehanshbarthwal-lab"
+ },
+ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/universal-scraping-architect",
+ "repository": "https://github.com/alirezarezvani/claude-skills",
+ "license": "MIT",
+ "skills": "./",
+ "dependencies": {
+ "python": ["firecrawl", "pandas", "requests", "beautifulsoup4"]
+ }
+}
\ No newline at end of file
diff --git a/engineering/universal-scraping-architect/.idea/.gitignore b/engineering/universal-scraping-architect/.idea/.gitignore
new file mode 100644
index 00000000..30cf57ed
--- /dev/null
+++ b/engineering/universal-scraping-architect/.idea/.gitignore
@@ -0,0 +1,10 @@
+# Default ignored files
+/shelf/
+/workspace.xml
+# Editor-based HTTP Client requests
+/httpRequests/
+# Ignored default folder with query files
+/queries/
+# Datasource local storage ignored files
+/dataSources/
+/dataSources.local.xml
diff --git a/engineering/universal-scraping-architect/.idea/inspectionProfiles/profiles_settings.xml b/engineering/universal-scraping-architect/.idea/inspectionProfiles/profiles_settings.xml
new file mode 100644
index 00000000..105ce2da
--- /dev/null
+++ b/engineering/universal-scraping-architect/.idea/inspectionProfiles/profiles_settings.xml
@@ -0,0 +1,6 @@
+
+
+
+
+
+
\ No newline at end of file
diff --git a/engineering/universal-scraping-architect/.idea/misc.xml b/engineering/universal-scraping-architect/.idea/misc.xml
new file mode 100644
index 00000000..187eec5d
--- /dev/null
+++ b/engineering/universal-scraping-architect/.idea/misc.xml
@@ -0,0 +1,7 @@
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/engineering/universal-scraping-architect/.idea/modules.xml b/engineering/universal-scraping-architect/.idea/modules.xml
new file mode 100644
index 00000000..940fef20
--- /dev/null
+++ b/engineering/universal-scraping-architect/.idea/modules.xml
@@ -0,0 +1,8 @@
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/engineering/universal-scraping-architect/.idea/universal-scraping-architect.iml b/engineering/universal-scraping-architect/.idea/universal-scraping-architect.iml
new file mode 100644
index 00000000..3ab2d77f
--- /dev/null
+++ b/engineering/universal-scraping-architect/.idea/universal-scraping-architect.iml
@@ -0,0 +1,14 @@
+
+
+
+
+
+
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/engineering/universal-scraping-architect/.idea/vcs.xml b/engineering/universal-scraping-architect/.idea/vcs.xml
new file mode 100644
index 00000000..b2bdec2d
--- /dev/null
+++ b/engineering/universal-scraping-architect/.idea/vcs.xml
@@ -0,0 +1,6 @@
+
+
+
+
+
+
\ No newline at end of file
diff --git a/engineering/universal-scraping-architect/SKILL.md b/engineering/universal-scraping-architect/SKILL.md
new file mode 100644
index 00000000..32028e98
--- /dev/null
+++ b/engineering/universal-scraping-architect/SKILL.md
@@ -0,0 +1,58 @@
+---
+name: "universal-scraping-architect"
+description: "Use for web scraping, crawling, document extraction, API parsing, or building validation-heavy data pipelines using Firecrawl or local Python scripts."
+---
+
+# Universal Scraping Architect
+
+You are an expert web scraping and data extraction engineer. Your goal is to design complete, robust data pipelines with intelligent routing, validation, and token budget tracking—not brittle one-off scripts.
+
+**Dependency Notice:** This skill utilizes `firecrawl`, `pandas`, `requests`, and `beautifulsoup4`. It uses a BYOK (Bring Your Own Key) pattern for Firecrawl. API keys must only be loaded via environment variables.
+
+## Before Starting
+**Check for context first:**
+If `project-context.md` exists, read it before asking questions. Determine the target data format, scale of extraction, and deployment environment before writing any code.
+
+## How This Skill Works
+
+This skill supports 3 extraction modes based on intelligent routing:
+
+### Mode 1: API-Driven (Firecrawl)
+Use when the source is a public URL, heavily dynamic (JS/SPA), requires search-first discovery, or involves bulk crawling across a domain.
+### Mode 2: Local Python (Traditional)
+Use when extracting from local files (PDF, Excel, CSV), the data is private/sensitive, or the target is a simple static HTML page where Firecrawl is overkill.
+### Mode 3: Hybrid Pipeline
+Use when Firecrawl handles URL discovery/web extraction, but local Python (Pandas) is required to clean, normalize, and structure the output before saving.
+
+## The Extraction Pipeline
+
+When executing a scraping task, always follow this sequence:
+1. **Route the Approach:** Explicitly state whether Firecrawl or Local Python is being used and why.
+2. **Track Budgets:** Estimate Firecrawl API quotas or LLM token context limits before executing large jobs.
+3. **Extract Safely:** Implement checkpointing for multi-page jobs. Handle pagination and dynamic layouts gracefully.
+4. **Validate & Clean:** Enforce required fields, catch empty outputs, flag duplicates, and normalize field names.
+5. **Format:** Default to CSV for tabular data, JSON for nested structures, and Markdown for clean text.
+
+## Proactive Triggers
+
+Surface these issues WITHOUT being asked when you notice them in context:
+- **Hardcoded API Keys** → Flag immediately and rewrite to use `os.getenv('FIRECRAWL_API_KEY')`.
+- **Private Data Leakage** → If the user asks to send local, sensitive files to an external API, flag the privacy risk and suggest Mode 2 (Local Python).
+- **Missing Pagination** → If the target implies hundreds of records but no pagination logic is requested, flag it and add checkpointing.
+
+## Output Artifacts
+
+| When you ask for... | You get... |
+|---------------------|------------|
+| "Scrape this site" | A fully validated Python extraction script with routing logic and error handling. |
+| "Get data from this table" | A clean CSV/JSON dataset with a summary log of row counts and empty values. |
+| "Crawl these docs" | A Markdown deliverable chunked for LLM token limits. |
+
+## Anti-Patterns
+- **Brittle Selectors:** Never use highly nested CSS selectors (e.g., `div > span > ul > li:nth-child(3)`). Use data attributes or robust structural anchors.
+- **Ignoring Etiquette:** Never scrape without checking `robots.txt` or implementing sensible rate limits.
+- **No Validation:** Never blindly write scraped data to a file without checking if the array is empty or missing critical keys.
+
+## Related Skills
+- **data-cleaning**: Use when the scraped data requires complex statistical normalization or deduplication.
+- **browser-automation**: Use for highly interactive scraping requiring user emulation (clicks, logins) where Firecrawl is insufficient.
\ No newline at end of file
diff --git a/engineering/universal-scraping-architect/agents/cs-scraping-architect.md b/engineering/universal-scraping-architect/agents/cs-scraping-architect.md
new file mode 100644
index 00000000..15bf6938
--- /dev/null
+++ b/engineering/universal-scraping-architect/agents/cs-scraping-architect.md
@@ -0,0 +1,6 @@
+---
+name: cs-scraping-architect
+description: Expert persona for web scraping and data pipeline design.
+---
+# cs-scraping-architect
+Use this agent when you need to design a complex extraction strategy or debug scraping scripts.
\ No newline at end of file
diff --git a/engineering/universal-scraping-architect/commands/cs-scrape.md b/engineering/universal-scraping-architect/commands/cs-scrape.md
new file mode 100644
index 00000000..a939de83
--- /dev/null
+++ b/engineering/universal-scraping-architect/commands/cs-scrape.md
@@ -0,0 +1,6 @@
+---
+name: cs-scrape
+description: Execute a scraping task for a specific URL.
+---
+# /cs-scrape [url]
+Triggers the scraping architect to analyze and extract data from the target URL.
\ No newline at end of file
diff --git a/engineering/universal-scraping-architect/references/firecrawl-technical-guide.md b/engineering/universal-scraping-architect/references/firecrawl-technical-guide.md
new file mode 100644
index 00000000..de4d2bf7
--- /dev/null
+++ b/engineering/universal-scraping-architect/references/firecrawl-technical-guide.md
@@ -0,0 +1,10 @@
+# Firecrawl Technical Guide
+
+This document covers the technical integration patterns for the Firecrawl API within the Universal Scraping Architect.
+
+### Authoritative Sources
+1. [Firecrawl API Documentation](https://docs.firecrawl.dev/api-reference/introduction)
+2. [Firecrawl SDK for Python](https://github.com/mendableai/firecrawl-py)
+3. [REST API Design Best Practices (Microsoft)](https://learn.microsoft.com/en-us/azure/architecture/best-practices/api-design)
+4. [Handling API Rate Limits (Cloudflare)](https://developers.cloudflare.com/fundamentals/api/reference/rate-limits/)
+5. [JSON Schema Standard](https://json-schema.org/specification.html)
\ No newline at end of file
diff --git a/engineering/universal-scraping-architect/references/scraping-ethics-security.md b/engineering/universal-scraping-architect/references/scraping-ethics-security.md
new file mode 100644
index 00000000..bc09c71e
--- /dev/null
+++ b/engineering/universal-scraping-architect/references/scraping-ethics-security.md
@@ -0,0 +1,10 @@
+# Scraping Ethics, Robots.txt, and Security
+
+Guidelines for ethical data collection and securing scraping pipelines.
+
+### Authoritative Sources
+1. [The Robots Exclusion Protocol (RFC 9309)](https://datatracker.ietf.org/doc/rfc9309/)
+2. [OWASP Automated Threats to Web Applications](https://owasp.org/www-project-automated-threats-to-web-applications/)
+3. [Scraping Ethics Best Practices (Ethical Web Scraping)](https://www.ethicalwebscraping.org/)
+4. [Legal Aspects of Web Scraping (Lexology)](https://www.lexology.com/library/detail.aspx?g=e6e0287a-62ad-4d6d-852a-9e535e69e6b4)
+5. [Cloudflare Bot Management Overview](https://www.cloudflare.com/en-gb/pg-lp/bot-management-for-everyone/)
\ No newline at end of file
diff --git a/engineering/universal-scraping-architect/scripts/firecrawl_example.py b/engineering/universal-scraping-architect/scripts/firecrawl_example.py
new file mode 100644
index 00000000..f7b0a9c8
--- /dev/null
+++ b/engineering/universal-scraping-architect/scripts/firecrawl_example.py
@@ -0,0 +1,157 @@
+"""
+Universal Scraping Architect: Firecrawl Example (Path C — Repeatable Deliverable)
+===================================================================================
+Extracts clean markdown content from a target URL using the Firecrawl SDK.
+
+Demonstrates:
+ - Safe API key loading from environment
+ - Token budget tracking before LLM processing
+ - Firecrawl quota awareness logging
+ - Structured error handling
+ - Clean output saving with validation
+
+Usage:
+ export FIRECRAWL_API_KEY="fc-YOUR_KEY_HERE" # Linux/macOS
+ $env:FIRECRAWL_API_KEY = "fc-YOUR_KEY_HERE" # Windows PowerShell
+ python examples/firecrawl_example.py
+"""
+
+import os
+import sys
+from datetime import datetime
+from firecrawl import FirecrawlApp
+
+# =============================================================================
+# CONFIG — Edit these for your task
+# =============================================================================
+TARGET_URL = "https://example.com/research-data"
+OUTPUT_FILE = "clean_extraction.md"
+LOG_FILE = "firecrawl_run.log"
+
+# Firecrawl / Token settings
+FIRECRAWL_API_KEY_ENV = "FIRECRAWL_API_KEY"
+TOKEN_CONTEXT_LIMIT = 100_000 # Adjust to your model's context window
+RESERVED_OUTPUT_TOKENS = 4_000 # Tokens held back for the model's response
+
+
+# =============================================================================
+# HELPERS
+# =============================================================================
+
+def log(message: str, level: str = "INFO"):
+ """Prints a timestamped log line and appends it to the log file."""
+ timestamp = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
+ line = f"[{timestamp}] [{level}] {message}"
+ print(line)
+ with open(LOG_FILE, "a", encoding="utf-8") as f:
+ f.write(line + "\n")
+
+
+def check_environment() -> str:
+ """Loads the Firecrawl API key from the environment. Fails fast if missing."""
+ api_key = os.getenv(FIRECRAWL_API_KEY_ENV)
+ if not api_key:
+ log(
+ f"Missing environment variable: {FIRECRAWL_API_KEY_ENV}. "
+ "Set it before running this script.",
+ level="ERROR"
+ )
+ sys.exit(1)
+ log("API key loaded successfully from environment.")
+ return api_key
+
+
+def estimate_tokens(text: str) -> int:
+ """Rough token estimate based on character count (characters / 4)."""
+ return len(text) // 4
+
+
+def check_token_budget(text: str) -> bool:
+ """
+ Estimates token usage and warns if over budget.
+ Returns True if within budget, False if over.
+ """
+ estimated = estimate_tokens(text)
+ available = TOKEN_CONTEXT_LIMIT - RESERVED_OUTPUT_TOKENS
+ log(f"Token estimate: {estimated:,} / {available:,} available tokens.")
+ if estimated > available:
+ log(
+ f"OVER TOKEN BUDGET by {estimated - available:,} tokens. "
+ "Consider chunking the output before passing to an LLM.",
+ level="WARNING"
+ )
+ return False
+ log("Within token budget.")
+ return True
+
+
+def validate_content(content: str) -> bool:
+ """Basic validation — checks the response is non-empty."""
+ if not content or not content.strip():
+ log("Validation failed: empty content received from Firecrawl.", level="ERROR")
+ return False
+ log(f"Validation passed. Content length: {len(content):,} characters.")
+ return True
+
+
+def save_output(content: str, path: str):
+ """Saves validated content to the output path."""
+ with open(path, "w", encoding="utf-8") as f:
+ f.write(content)
+ log(f"Output saved to: {path}")
+
+
+# =============================================================================
+# MAIN
+# =============================================================================
+
+def main():
+ log("=" * 60)
+ log("Universal Scraping Architect — Firecrawl Path C")
+ log("=" * 60)
+
+ # Step 1: Environment check
+ api_key = check_environment()
+
+ # Step 2: Initialise Firecrawl client
+ app = FirecrawlApp(api_key=api_key)
+ log(f"Firecrawl client initialised. Target: {TARGET_URL}")
+
+ # Step 3: Scrape
+ try:
+ log("Starting scrape...")
+ result = app.scrape_url(TARGET_URL, params={"formats": ["markdown"]})
+ markdown_content = result.get("markdown", "")
+ except Exception as e:
+ log(f"Firecrawl scrape failed: {e}", level="ERROR")
+ log(
+ "Tip: if this is a quota or auth error, check your FIRECRAWL_API_KEY "
+ "and run `firecrawl --status` to inspect account state.",
+ level="WARNING"
+ )
+ sys.exit(1)
+
+ # Step 4: Validate
+ if not validate_content(markdown_content):
+ sys.exit(1)
+
+ # Step 5: Token budget check
+ check_token_budget(markdown_content)
+
+ # Step 6: Save clean output
+ save_output(markdown_content, OUTPUT_FILE)
+
+ # Step 7: Final summary
+ log("=" * 60)
+ log("EXTRACTION COMPLETE")
+ log(f" Source URL : {TARGET_URL}")
+ log(f" Output file : {OUTPUT_FILE}")
+ log(f" Characters : {len(markdown_content):,}")
+ log(f" Est. tokens : {estimate_tokens(markdown_content):,}")
+ log(f" Log file : {LOG_FILE}")
+ log("=" * 60)
+ log("Customisable: TARGET_URL, OUTPUT_FILE, TOKEN_CONTEXT_LIMIT, formats.")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/engineering/universal-scraping-architect/scripts/local_bs4_example.py b/engineering/universal-scraping-architect/scripts/local_bs4_example.py
new file mode 100644
index 00000000..0d70c4ce
--- /dev/null
+++ b/engineering/universal-scraping-architect/scripts/local_bs4_example.py
@@ -0,0 +1,226 @@
+"""
+Universal Scraping Architect: Traditional / Local Scraping Example
+==================================================================
+Extracts a data table from a static HTML page using Requests and BeautifulSoup,
+then validates, cleans, and saves to CSV.
+
+Demonstrates:
+ - Safe HTTP fetching with headers, timeouts, and retry logic
+ - HTML table parsing with pandas
+ - Column name normalisation to snake_case
+ - Required field validation before saving
+ - Structured error handling at every stage
+ - Clean, logged output
+
+Usage:
+ pip install -r requirements.txt
+ python examples/local_bs4_example.py
+"""
+
+import sys
+import time
+from datetime import datetime
+
+import pandas as pd
+import requests
+from bs4 import BeautifulSoup
+
+# =============================================================================
+# CONFIG — Edit these for your task
+# =============================================================================
+TARGET_URL = "https://example.com/macroeconomic-indicators/energy-prices"
+OUTPUT_FILE = "energy_price_data.csv"
+LOG_FILE = "local_scrape_run.log"
+
+# HTTP settings
+USER_AGENT = "UniversalScrapingArchitect/1.0 (contact: your@email.com)"
+TIMEOUT_SECONDS = 15
+MAX_RETRIES = 3
+RETRY_DELAY_SECONDS = 2
+
+# Validation
+REQUIRED_COLUMNS = ["date", "price_index", "yoy_change"]
+TABLE_SELECTOR = {"id": "indicator-data"} # Edit to match your target table's HTML attributes
+
+
+# =============================================================================
+# HELPERS
+# =============================================================================
+
+def log(message: str, level: str = "INFO"):
+ """Prints a timestamped log line and appends it to the log file."""
+ timestamp = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
+ line = f"[{timestamp}] [{level}] {message}"
+ print(line)
+ with open(LOG_FILE, "a", encoding="utf-8") as f:
+ f.write(line + "\n")
+
+
+def safe_get(url: str) -> str:
+ """
+ Fetches HTML from a URL with polite headers, a timeout, and retry logic.
+ Raises on non-2xx status.
+ """
+ headers = {"User-Agent": USER_AGENT}
+
+ for attempt in range(1, MAX_RETRIES + 1):
+ try:
+ log(f"Attempt {attempt}/{MAX_RETRIES}: GET {url}")
+ response = requests.get(url, headers=headers, timeout=TIMEOUT_SECONDS)
+ response.raise_for_status()
+ log(f"HTTP {response.status_code} — content length: {len(response.text):,} chars.")
+ return response.text
+ except requests.exceptions.HTTPError as e:
+ log(f"HTTP error on attempt {attempt}: {e}", level="ERROR")
+ except requests.exceptions.ConnectionError as e:
+ log(f"Connection error on attempt {attempt}: {e}", level="ERROR")
+ except requests.exceptions.Timeout:
+ log(f"Timeout on attempt {attempt} after {TIMEOUT_SECONDS}s.", level="WARNING")
+ except requests.exceptions.RequestException as e:
+ log(f"Request failed on attempt {attempt}: {e}", level="ERROR")
+
+ if attempt < MAX_RETRIES:
+ log(f"Waiting {RETRY_DELAY_SECONDS}s before retry...")
+ time.sleep(RETRY_DELAY_SECONDS)
+
+ raise RuntimeError(f"All {MAX_RETRIES} attempts failed for URL: {url}")
+
+
+def find_table(html: str) -> BeautifulSoup:
+ """
+ Parses the HTML and returns the target table element.
+ Adjust TABLE_SELECTOR to match the table you need.
+ """
+ soup = BeautifulSoup(html, "html.parser")
+ table = soup.find("table", TABLE_SELECTOR)
+ if not table:
+ raise ValueError(
+ f"Target table not found. Selector used: {TABLE_SELECTOR}. "
+ "Inspect the page source and update TABLE_SELECTOR in CONFIG."
+ )
+ log("Target table found in HTML.")
+ return table
+
+
+def parse_table(table) -> pd.DataFrame:
+ """Converts a BeautifulSoup table element into a pandas DataFrame."""
+ df = pd.read_html(str(table))[0]
+ log(f"Parsed table: {len(df)} rows, {len(df.columns)} columns.")
+ return df
+
+
+def clean_column_names(df: pd.DataFrame) -> pd.DataFrame:
+ """Normalises all column names to snake_case."""
+ df.columns = (
+ df.columns
+ .str.strip()
+ .str.lower()
+ .str.replace(r"\s+", "_", regex=True)
+ .str.replace(r"[^\w]", "", regex=True)
+ )
+ log(f"Columns after normalisation: {list(df.columns)}")
+ return df
+
+
+def clean_data(df: pd.DataFrame) -> pd.DataFrame:
+ """
+ Applies general cleaning rules.
+ Extend this function for your task-specific cleaning needs.
+ """
+ # Strip whitespace from string columns
+ for col in df.select_dtypes(include="object").columns:
+ df[col] = df[col].str.strip()
+
+ # Drop fully empty rows
+ before = len(df)
+ df = df.dropna(how="all")
+ dropped = before - len(df)
+ if dropped:
+ log(f"Dropped {dropped} fully empty rows.", level="WARNING")
+
+ return df
+
+
+def validate(df: pd.DataFrame) -> bool:
+ """
+ Checks that the DataFrame is non-empty and contains all required columns.
+ Returns True if valid, False otherwise.
+ """
+ if df.empty:
+ log("Validation failed: DataFrame is empty.", level="ERROR")
+ return False
+
+ missing = [col for col in REQUIRED_COLUMNS if col not in df.columns]
+ if missing:
+ log(
+ f"Validation failed: missing required columns: {missing}. "
+ f"Available columns: {list(df.columns)}",
+ level="ERROR"
+ )
+ return False
+
+ log(f"Validation passed. {len(df)} rows, {len(df.columns)} columns.")
+ return True
+
+
+def save_output(df: pd.DataFrame, path: str):
+ """Saves the validated DataFrame to CSV."""
+ df.to_csv(path, index=False, encoding="utf-8")
+ log(f"Output saved to: {path}")
+
+
+# =============================================================================
+# MAIN
+# =============================================================================
+
+def main():
+ log("=" * 60)
+ log("Universal Scraping Architect — Traditional / Local Scraping")
+ log("=" * 60)
+
+ # Step 1: Fetch
+ try:
+ html = safe_get(TARGET_URL)
+ except RuntimeError as e:
+ log(str(e), level="ERROR")
+ sys.exit(1)
+
+ # Step 2: Find table
+ try:
+ table = find_table(html)
+ except ValueError as e:
+ log(str(e), level="ERROR")
+ sys.exit(1)
+
+ # Step 3: Parse
+ try:
+ df = parse_table(table)
+ except Exception as e:
+ log(f"Failed to parse table into DataFrame: {e}", level="ERROR")
+ sys.exit(1)
+
+ # Step 4: Clean
+ df = clean_column_names(df)
+ df = clean_data(df)
+
+ # Step 5: Validate
+ if not validate(df):
+ sys.exit(1)
+
+ # Step 6: Save
+ save_output(df, OUTPUT_FILE)
+
+ # Step 7: Final summary
+ log("=" * 60)
+ log("EXTRACTION COMPLETE")
+ log(f" Source URL : {TARGET_URL}")
+ log(f" Output file : {OUTPUT_FILE}")
+ log(f" Rows saved : {len(df)}")
+ log(f" Columns : {list(df.columns)}")
+ log(f" Log file : {LOG_FILE}")
+ log("=" * 60)
+ log("Customisable: TARGET_URL, OUTPUT_FILE, TABLE_SELECTOR, REQUIRED_COLUMNS.")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/engineering/universal-scraping-architect/scripts/scripts/validate_extraction.py b/engineering/universal-scraping-architect/scripts/scripts/validate_extraction.py
new file mode 100644
index 00000000..97dd4e5a
--- /dev/null
+++ b/engineering/universal-scraping-architect/scripts/scripts/validate_extraction.py
@@ -0,0 +1,44 @@
+#!/usr/bin/env python3
+"""
+validate_extraction.py
+Intelligent stdlib-only validation for JSON structures.
+"""
+import argparse
+import json
+import sys
+import os
+
+
+def validate_json(file_path):
+ if not os.path.exists(file_path):
+ return {"status": "error", "message": f"File {file_path} not found."}
+
+ try:
+ with open(file_path, 'r', encoding='utf-8') as f:
+ data = json.load(f)
+ # Basic validation: ensure it's not empty and is a list/dict
+ if not data:
+ return {"status": "warning", "message": "JSON file is empty."}
+ return {"status": "ok", "message": "JSON is valid and well-formed."}
+ except json.JSONDecodeError as e:
+ return {"status": "error", "message": f"Invalid JSON: {str(e)}"}
+
+
+def main():
+ parser = argparse.ArgumentParser(description="Standard Library JSON Validator")
+ parser.add_argument("file", help="Path to JSON file to validate")
+ parser.add_argument("--json", action="store_true", help="Output results in JSON format")
+
+ args = parser.parse_args()
+ result = validate_json(args.file)
+
+ if args.json:
+ print(json.dumps(result, indent=2))
+ else:
+ print(f"[{result['status'].upper()}] {result['message']}")
+
+ sys.exit(0 if result['status'] == "ok" else 1)
+
+
+if __name__ == "__main__":
+ main()
\ No newline at end of file