mirror of
https://github.com/alirezarezvani/claude-skills.git
synced 2026-10-10 03:27:56 +00:00
Merge 1b240027c6 into 852e7da786
This commit is contained in:
commit
07441d139f
21 changed files with 658 additions and 2 deletions
|
|
@ -4,7 +4,7 @@
|
|||
"name": "Alireza Rezvani",
|
||||
"url": "https://alirezarezvani.com"
|
||||
},
|
||||
"description": "329 production-ready skill packages for Claude AI across 14 domains: engineering advanced (76 — incl. 4 Matt Pocock-derived productivity skills + v2.7.3 security-guidance PreToolUse hook), engineering core (51), marketing (47 — incl. v2.7.3 AEO/Answer Engine Optimization), c-level advisory (66), product (17), regulatory/QMS (18), project management (9), business growth (5), finance (4), productivity (6), marketing top-level (2, v2.7.0), research (8, v2.7.0), business-operations (7, v2.8.0), and commercial (8, v2.8.0). Includes ~444 Python tools, ~598 reference documents, 49+ agents, 79+ slash commands.",
|
||||
"description": "329 production-ready skill packages for Claude AI across 14 domains: engineering advanced (76 — incl. 4 Matt Pocock-derived productivity skills + v2.7.3 security-guidance PreToolUse hook), engineering core (51), marketing (47 — incl. v2.7.3 AEO/Answer Engine Optimization), c-level advisory (66), product (17), regulatory/QMS (18), project management (9), business growth (5), finance (4), productivity (4, v2.7.0), marketing top-level (2, v2.7.0), research (8, v2.7.0), business-operations (7, v2.8.0), and commercial (8, v2.8.0). Includes ~441 Python tools, ~594 reference documents, 48+ agents, 77+ slash commands.",
|
||||
"homepage": "https://github.com/alirezarezvani/claude-skills",
|
||||
"repository": "https://github.com/alirezarezvani/claude-skills",
|
||||
"metadata": {
|
||||
|
|
@ -1313,6 +1313,24 @@
|
|||
"grill-with-docs"
|
||||
],
|
||||
"category": "commercial"
|
||||
},
|
||||
{
|
||||
"name": "universal-scraping-architect",
|
||||
"source": "./engineering/universal-scraping-architect",
|
||||
"description": "A universal scraping skill with intelligent routing, token budget tracking, and quota awareness. Supports Firecrawl and local Python extraction.",
|
||||
"version": "2.1.2",
|
||||
"author": {
|
||||
"name": "Mehansh Barthwal"
|
||||
},
|
||||
"keywords": [
|
||||
"scraping",
|
||||
"data-extraction",
|
||||
"firecrawl",
|
||||
"beautifulsoup4",
|
||||
"pandas",
|
||||
"automation"
|
||||
],
|
||||
"category": "development"
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
10
.idea/.gitignore
generated
vendored
Normal file
10
.idea/.gitignore
generated
vendored
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
# Default ignored files
|
||||
/shelf/
|
||||
/workspace.xml
|
||||
# Editor-based HTTP Client requests
|
||||
/httpRequests/
|
||||
# Ignored default folder with query files
|
||||
/queries/
|
||||
# Datasource local storage ignored files
|
||||
/dataSources/
|
||||
/dataSources.local.xml
|
||||
24
.idea/claude-skills.iml
generated
Normal file
24
.idea/claude-skills.iml
generated
Normal file
|
|
@ -0,0 +1,24 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<module type="PYTHON_MODULE" version="4">
|
||||
<component name="NewModuleRootManager">
|
||||
<content url="file://$MODULE_DIR$">
|
||||
<excludeFolder url="file://$MODULE_DIR$/engineering/universal-scraping-architect/.venv" />
|
||||
</content>
|
||||
<orderEntry type="inheritedJdk" />
|
||||
<orderEntry type="sourceFolder" forTests="false" />
|
||||
</component>
|
||||
<component name="PyDocumentationSettings">
|
||||
<option name="format" value="PLAIN" />
|
||||
<option name="myDocStringFormat" value="Plain" />
|
||||
</component>
|
||||
<component name="TemplatesService">
|
||||
<option name="TEMPLATE_FOLDERS">
|
||||
<list>
|
||||
<option value="$MODULE_DIR$/c-level-advisor/skills/board-deck-builder/templates" />
|
||||
</list>
|
||||
</option>
|
||||
</component>
|
||||
<component name="TestRunnerService">
|
||||
<option name="PROJECT_TEST_RUNNER" value="py.test" />
|
||||
</component>
|
||||
</module>
|
||||
6
.idea/inspectionProfiles/profiles_settings.xml
generated
Normal file
6
.idea/inspectionProfiles/profiles_settings.xml
generated
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
<component name="InspectionProjectProfileManager">
|
||||
<settings>
|
||||
<option name="USE_PROJECT_PROFILE" value="false" />
|
||||
<version value="1.0" />
|
||||
</settings>
|
||||
</component>
|
||||
8
.idea/modules.xml
generated
Normal file
8
.idea/modules.xml
generated
Normal file
|
|
@ -0,0 +1,8 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project version="4">
|
||||
<component name="ProjectModuleManager">
|
||||
<modules>
|
||||
<module fileurl="file://$PROJECT_DIR$/.idea/claude-skills.iml" filepath="$PROJECT_DIR$/.idea/claude-skills.iml" />
|
||||
</modules>
|
||||
</component>
|
||||
</project>
|
||||
6
.idea/vcs.xml
generated
Normal file
6
.idea/vcs.xml
generated
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project version="4">
|
||||
<component name="VcsDirectoryMappings">
|
||||
<mapping directory="" vcs="Git" />
|
||||
</component>
|
||||
</project>
|
||||
|
|
@ -0,0 +1,16 @@
|
|||
{
|
||||
"name": "universal-scraping-architect",
|
||||
"description": "A universal scraping skill with intelligent routing, token budget tracking, and quota awareness.",
|
||||
"version": "2.1.2",
|
||||
"author": {
|
||||
"name": "Mehansh Barthwal",
|
||||
"url": "https://github.com/mehanshbarthwal-lab"
|
||||
},
|
||||
"homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/universal-scraping-architect",
|
||||
"repository": "https://github.com/alirezarezvani/claude-skills",
|
||||
"license": "MIT",
|
||||
"skills": "./",
|
||||
"dependencies": {
|
||||
"python": ["firecrawl", "pandas", "requests", "beautifulsoup4"]
|
||||
}
|
||||
}
|
||||
10
engineering/universal-scraping-architect/.idea/.gitignore
generated
vendored
Normal file
10
engineering/universal-scraping-architect/.idea/.gitignore
generated
vendored
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
# Default ignored files
|
||||
/shelf/
|
||||
/workspace.xml
|
||||
# Editor-based HTTP Client requests
|
||||
/httpRequests/
|
||||
# Ignored default folder with query files
|
||||
/queries/
|
||||
# Datasource local storage ignored files
|
||||
/dataSources/
|
||||
/dataSources.local.xml
|
||||
6
engineering/universal-scraping-architect/.idea/inspectionProfiles/profiles_settings.xml
generated
Normal file
6
engineering/universal-scraping-architect/.idea/inspectionProfiles/profiles_settings.xml
generated
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
<component name="InspectionProjectProfileManager">
|
||||
<settings>
|
||||
<option name="USE_PROJECT_PROFILE" value="false" />
|
||||
<version value="1.0" />
|
||||
</settings>
|
||||
</component>
|
||||
7
engineering/universal-scraping-architect/.idea/misc.xml
generated
Normal file
7
engineering/universal-scraping-architect/.idea/misc.xml
generated
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project version="4">
|
||||
<component name="Black">
|
||||
<option name="sdkName" value="Python 3.14 (universal-scraping-architect)" />
|
||||
</component>
|
||||
<component name="ProjectRootManager" version="2" project-jdk-name="Python 3.14 (universal-scraping-architect)" project-jdk-type="Python SDK" />
|
||||
</project>
|
||||
8
engineering/universal-scraping-architect/.idea/modules.xml
generated
Normal file
8
engineering/universal-scraping-architect/.idea/modules.xml
generated
Normal file
|
|
@ -0,0 +1,8 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project version="4">
|
||||
<component name="ProjectModuleManager">
|
||||
<modules>
|
||||
<module fileurl="file://$PROJECT_DIR$/.idea/universal-scraping-architect.iml" filepath="$PROJECT_DIR$/.idea/universal-scraping-architect.iml" />
|
||||
</modules>
|
||||
</component>
|
||||
</project>
|
||||
14
engineering/universal-scraping-architect/.idea/universal-scraping-architect.iml
generated
Normal file
14
engineering/universal-scraping-architect/.idea/universal-scraping-architect.iml
generated
Normal file
|
|
@ -0,0 +1,14 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<module type="PYTHON_MODULE" version="4">
|
||||
<component name="NewModuleRootManager">
|
||||
<content url="file://$MODULE_DIR$">
|
||||
<excludeFolder url="file://$MODULE_DIR$/.venv" />
|
||||
</content>
|
||||
<orderEntry type="jdk" jdkName="Python 3.14 (universal-scraping-architect)" jdkType="Python SDK" />
|
||||
<orderEntry type="sourceFolder" forTests="false" />
|
||||
</component>
|
||||
<component name="PyDocumentationSettings">
|
||||
<option name="format" value="PLAIN" />
|
||||
<option name="myDocStringFormat" value="Plain" />
|
||||
</component>
|
||||
</module>
|
||||
6
engineering/universal-scraping-architect/.idea/vcs.xml
generated
Normal file
6
engineering/universal-scraping-architect/.idea/vcs.xml
generated
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project version="4">
|
||||
<component name="VcsDirectoryMappings">
|
||||
<mapping directory="$PROJECT_DIR$/../.." vcs="Git" />
|
||||
</component>
|
||||
</project>
|
||||
58
engineering/universal-scraping-architect/SKILL.md
Normal file
58
engineering/universal-scraping-architect/SKILL.md
Normal file
|
|
@ -0,0 +1,58 @@
|
|||
---
|
||||
name: "universal-scraping-architect"
|
||||
description: "Use for web scraping, crawling, document extraction, API parsing, or building validation-heavy data pipelines using Firecrawl or local Python scripts."
|
||||
---
|
||||
|
||||
# Universal Scraping Architect
|
||||
|
||||
You are an expert web scraping and data extraction engineer. Your goal is to design complete, robust data pipelines with intelligent routing, validation, and token budget tracking—not brittle one-off scripts.
|
||||
|
||||
**Dependency Notice:** This skill utilizes `firecrawl`, `pandas`, `requests`, and `beautifulsoup4`. It uses a BYOK (Bring Your Own Key) pattern for Firecrawl. API keys must only be loaded via environment variables.
|
||||
|
||||
## Before Starting
|
||||
**Check for context first:**
|
||||
If `project-context.md` exists, read it before asking questions. Determine the target data format, scale of extraction, and deployment environment before writing any code.
|
||||
|
||||
## How This Skill Works
|
||||
|
||||
This skill supports 3 extraction modes based on intelligent routing:
|
||||
|
||||
### Mode 1: API-Driven (Firecrawl)
|
||||
Use when the source is a public URL, heavily dynamic (JS/SPA), requires search-first discovery, or involves bulk crawling across a domain.
|
||||
### Mode 2: Local Python (Traditional)
|
||||
Use when extracting from local files (PDF, Excel, CSV), the data is private/sensitive, or the target is a simple static HTML page where Firecrawl is overkill.
|
||||
### Mode 3: Hybrid Pipeline
|
||||
Use when Firecrawl handles URL discovery/web extraction, but local Python (Pandas) is required to clean, normalize, and structure the output before saving.
|
||||
|
||||
## The Extraction Pipeline
|
||||
|
||||
When executing a scraping task, always follow this sequence:
|
||||
1. **Route the Approach:** Explicitly state whether Firecrawl or Local Python is being used and why.
|
||||
2. **Track Budgets:** Estimate Firecrawl API quotas or LLM token context limits before executing large jobs.
|
||||
3. **Extract Safely:** Implement checkpointing for multi-page jobs. Handle pagination and dynamic layouts gracefully.
|
||||
4. **Validate & Clean:** Enforce required fields, catch empty outputs, flag duplicates, and normalize field names.
|
||||
5. **Format:** Default to CSV for tabular data, JSON for nested structures, and Markdown for clean text.
|
||||
|
||||
## Proactive Triggers
|
||||
|
||||
Surface these issues WITHOUT being asked when you notice them in context:
|
||||
- **Hardcoded API Keys** → Flag immediately and rewrite to use `os.getenv('FIRECRAWL_API_KEY')`.
|
||||
- **Private Data Leakage** → If the user asks to send local, sensitive files to an external API, flag the privacy risk and suggest Mode 2 (Local Python).
|
||||
- **Missing Pagination** → If the target implies hundreds of records but no pagination logic is requested, flag it and add checkpointing.
|
||||
|
||||
## Output Artifacts
|
||||
|
||||
| When you ask for... | You get... |
|
||||
|---------------------|------------|
|
||||
| "Scrape this site" | A fully validated Python extraction script with routing logic and error handling. |
|
||||
| "Get data from this table" | A clean CSV/JSON dataset with a summary log of row counts and empty values. |
|
||||
| "Crawl these docs" | A Markdown deliverable chunked for LLM token limits. |
|
||||
|
||||
## Anti-Patterns
|
||||
- **Brittle Selectors:** Never use highly nested CSS selectors (e.g., `div > span > ul > li:nth-child(3)`). Use data attributes or robust structural anchors.
|
||||
- **Ignoring Etiquette:** Never scrape without checking `robots.txt` or implementing sensible rate limits.
|
||||
- **No Validation:** Never blindly write scraped data to a file without checking if the array is empty or missing critical keys.
|
||||
|
||||
## Related Skills
|
||||
- **data-cleaning**: Use when the scraped data requires complex statistical normalization or deduplication.
|
||||
- **browser-automation**: Use for highly interactive scraping requiring user emulation (clicks, logins) where Firecrawl is insufficient.
|
||||
|
|
@ -0,0 +1,6 @@
|
|||
---
|
||||
name: cs-scraping-architect
|
||||
description: Expert persona for web scraping and data pipeline design.
|
||||
---
|
||||
# cs-scraping-architect
|
||||
Use this agent when you need to design a complex extraction strategy or debug scraping scripts.
|
||||
|
|
@ -0,0 +1,6 @@
|
|||
---
|
||||
name: cs-scrape
|
||||
description: Execute a scraping task for a specific URL.
|
||||
---
|
||||
# /cs-scrape [url]
|
||||
Triggers the scraping architect to analyze and extract data from the target URL.
|
||||
|
|
@ -0,0 +1,10 @@
|
|||
# Firecrawl Technical Guide
|
||||
|
||||
This document covers the technical integration patterns for the Firecrawl API within the Universal Scraping Architect.
|
||||
|
||||
### Authoritative Sources
|
||||
1. [Firecrawl API Documentation](https://docs.firecrawl.dev/api-reference/introduction)
|
||||
2. [Firecrawl SDK for Python](https://github.com/mendableai/firecrawl-py)
|
||||
3. [REST API Design Best Practices (Microsoft)](https://learn.microsoft.com/en-us/azure/architecture/best-practices/api-design)
|
||||
4. [Handling API Rate Limits (Cloudflare)](https://developers.cloudflare.com/fundamentals/api/reference/rate-limits/)
|
||||
5. [JSON Schema Standard](https://json-schema.org/specification.html)
|
||||
|
|
@ -0,0 +1,10 @@
|
|||
# Scraping Ethics, Robots.txt, and Security
|
||||
|
||||
Guidelines for ethical data collection and securing scraping pipelines.
|
||||
|
||||
### Authoritative Sources
|
||||
1. [The Robots Exclusion Protocol (RFC 9309)](https://datatracker.ietf.org/doc/rfc9309/)
|
||||
2. [OWASP Automated Threats to Web Applications](https://owasp.org/www-project-automated-threats-to-web-applications/)
|
||||
3. [Scraping Ethics Best Practices (Ethical Web Scraping)](https://www.ethicalwebscraping.org/)
|
||||
4. [Legal Aspects of Web Scraping (Lexology)](https://www.lexology.com/library/detail.aspx?g=e6e0287a-62ad-4d6d-852a-9e535e69e6b4)
|
||||
5. [Cloudflare Bot Management Overview](https://www.cloudflare.com/en-gb/pg-lp/bot-management-for-everyone/)
|
||||
|
|
@ -0,0 +1,157 @@
|
|||
"""
|
||||
Universal Scraping Architect: Firecrawl Example (Path C — Repeatable Deliverable)
|
||||
===================================================================================
|
||||
Extracts clean markdown content from a target URL using the Firecrawl SDK.
|
||||
|
||||
Demonstrates:
|
||||
- Safe API key loading from environment
|
||||
- Token budget tracking before LLM processing
|
||||
- Firecrawl quota awareness logging
|
||||
- Structured error handling
|
||||
- Clean output saving with validation
|
||||
|
||||
Usage:
|
||||
export FIRECRAWL_API_KEY="fc-YOUR_KEY_HERE" # Linux/macOS
|
||||
$env:FIRECRAWL_API_KEY = "fc-YOUR_KEY_HERE" # Windows PowerShell
|
||||
python examples/firecrawl_example.py
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
from datetime import datetime
|
||||
from firecrawl import FirecrawlApp
|
||||
|
||||
# =============================================================================
|
||||
# CONFIG — Edit these for your task
|
||||
# =============================================================================
|
||||
TARGET_URL = "https://example.com/research-data"
|
||||
OUTPUT_FILE = "clean_extraction.md"
|
||||
LOG_FILE = "firecrawl_run.log"
|
||||
|
||||
# Firecrawl / Token settings
|
||||
FIRECRAWL_API_KEY_ENV = "FIRECRAWL_API_KEY"
|
||||
TOKEN_CONTEXT_LIMIT = 100_000 # Adjust to your model's context window
|
||||
RESERVED_OUTPUT_TOKENS = 4_000 # Tokens held back for the model's response
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# HELPERS
|
||||
# =============================================================================
|
||||
|
||||
def log(message: str, level: str = "INFO"):
|
||||
"""Prints a timestamped log line and appends it to the log file."""
|
||||
timestamp = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
|
||||
line = f"[{timestamp}] [{level}] {message}"
|
||||
print(line)
|
||||
with open(LOG_FILE, "a", encoding="utf-8") as f:
|
||||
f.write(line + "\n")
|
||||
|
||||
|
||||
def check_environment() -> str:
|
||||
"""Loads the Firecrawl API key from the environment. Fails fast if missing."""
|
||||
api_key = os.getenv(FIRECRAWL_API_KEY_ENV)
|
||||
if not api_key:
|
||||
log(
|
||||
f"Missing environment variable: {FIRECRAWL_API_KEY_ENV}. "
|
||||
"Set it before running this script.",
|
||||
level="ERROR"
|
||||
)
|
||||
sys.exit(1)
|
||||
log("API key loaded successfully from environment.")
|
||||
return api_key
|
||||
|
||||
|
||||
def estimate_tokens(text: str) -> int:
|
||||
"""Rough token estimate based on character count (characters / 4)."""
|
||||
return len(text) // 4
|
||||
|
||||
|
||||
def check_token_budget(text: str) -> bool:
|
||||
"""
|
||||
Estimates token usage and warns if over budget.
|
||||
Returns True if within budget, False if over.
|
||||
"""
|
||||
estimated = estimate_tokens(text)
|
||||
available = TOKEN_CONTEXT_LIMIT - RESERVED_OUTPUT_TOKENS
|
||||
log(f"Token estimate: {estimated:,} / {available:,} available tokens.")
|
||||
if estimated > available:
|
||||
log(
|
||||
f"OVER TOKEN BUDGET by {estimated - available:,} tokens. "
|
||||
"Consider chunking the output before passing to an LLM.",
|
||||
level="WARNING"
|
||||
)
|
||||
return False
|
||||
log("Within token budget.")
|
||||
return True
|
||||
|
||||
|
||||
def validate_content(content: str) -> bool:
|
||||
"""Basic validation — checks the response is non-empty."""
|
||||
if not content or not content.strip():
|
||||
log("Validation failed: empty content received from Firecrawl.", level="ERROR")
|
||||
return False
|
||||
log(f"Validation passed. Content length: {len(content):,} characters.")
|
||||
return True
|
||||
|
||||
|
||||
def save_output(content: str, path: str):
|
||||
"""Saves validated content to the output path."""
|
||||
with open(path, "w", encoding="utf-8") as f:
|
||||
f.write(content)
|
||||
log(f"Output saved to: {path}")
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# MAIN
|
||||
# =============================================================================
|
||||
|
||||
def main():
|
||||
log("=" * 60)
|
||||
log("Universal Scraping Architect — Firecrawl Path C")
|
||||
log("=" * 60)
|
||||
|
||||
# Step 1: Environment check
|
||||
api_key = check_environment()
|
||||
|
||||
# Step 2: Initialise Firecrawl client
|
||||
app = FirecrawlApp(api_key=api_key)
|
||||
log(f"Firecrawl client initialised. Target: {TARGET_URL}")
|
||||
|
||||
# Step 3: Scrape
|
||||
try:
|
||||
log("Starting scrape...")
|
||||
result = app.scrape_url(TARGET_URL, params={"formats": ["markdown"]})
|
||||
markdown_content = result.get("markdown", "")
|
||||
except Exception as e:
|
||||
log(f"Firecrawl scrape failed: {e}", level="ERROR")
|
||||
log(
|
||||
"Tip: if this is a quota or auth error, check your FIRECRAWL_API_KEY "
|
||||
"and run `firecrawl --status` to inspect account state.",
|
||||
level="WARNING"
|
||||
)
|
||||
sys.exit(1)
|
||||
|
||||
# Step 4: Validate
|
||||
if not validate_content(markdown_content):
|
||||
sys.exit(1)
|
||||
|
||||
# Step 5: Token budget check
|
||||
check_token_budget(markdown_content)
|
||||
|
||||
# Step 6: Save clean output
|
||||
save_output(markdown_content, OUTPUT_FILE)
|
||||
|
||||
# Step 7: Final summary
|
||||
log("=" * 60)
|
||||
log("EXTRACTION COMPLETE")
|
||||
log(f" Source URL : {TARGET_URL}")
|
||||
log(f" Output file : {OUTPUT_FILE}")
|
||||
log(f" Characters : {len(markdown_content):,}")
|
||||
log(f" Est. tokens : {estimate_tokens(markdown_content):,}")
|
||||
log(f" Log file : {LOG_FILE}")
|
||||
log("=" * 60)
|
||||
log("Customisable: TARGET_URL, OUTPUT_FILE, TOKEN_CONTEXT_LIMIT, formats.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
@ -0,0 +1,226 @@
|
|||
"""
|
||||
Universal Scraping Architect: Traditional / Local Scraping Example
|
||||
==================================================================
|
||||
Extracts a data table from a static HTML page using Requests and BeautifulSoup,
|
||||
then validates, cleans, and saves to CSV.
|
||||
|
||||
Demonstrates:
|
||||
- Safe HTTP fetching with headers, timeouts, and retry logic
|
||||
- HTML table parsing with pandas
|
||||
- Column name normalisation to snake_case
|
||||
- Required field validation before saving
|
||||
- Structured error handling at every stage
|
||||
- Clean, logged output
|
||||
|
||||
Usage:
|
||||
pip install -r requirements.txt
|
||||
python examples/local_bs4_example.py
|
||||
"""
|
||||
|
||||
import sys
|
||||
import time
|
||||
from datetime import datetime
|
||||
|
||||
import pandas as pd
|
||||
import requests
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
# =============================================================================
|
||||
# CONFIG — Edit these for your task
|
||||
# =============================================================================
|
||||
TARGET_URL = "https://example.com/macroeconomic-indicators/energy-prices"
|
||||
OUTPUT_FILE = "energy_price_data.csv"
|
||||
LOG_FILE = "local_scrape_run.log"
|
||||
|
||||
# HTTP settings
|
||||
USER_AGENT = "UniversalScrapingArchitect/1.0 (contact: your@email.com)"
|
||||
TIMEOUT_SECONDS = 15
|
||||
MAX_RETRIES = 3
|
||||
RETRY_DELAY_SECONDS = 2
|
||||
|
||||
# Validation
|
||||
REQUIRED_COLUMNS = ["date", "price_index", "yoy_change"]
|
||||
TABLE_SELECTOR = {"id": "indicator-data"} # Edit to match your target table's HTML attributes
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# HELPERS
|
||||
# =============================================================================
|
||||
|
||||
def log(message: str, level: str = "INFO"):
|
||||
"""Prints a timestamped log line and appends it to the log file."""
|
||||
timestamp = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
|
||||
line = f"[{timestamp}] [{level}] {message}"
|
||||
print(line)
|
||||
with open(LOG_FILE, "a", encoding="utf-8") as f:
|
||||
f.write(line + "\n")
|
||||
|
||||
|
||||
def safe_get(url: str) -> str:
|
||||
"""
|
||||
Fetches HTML from a URL with polite headers, a timeout, and retry logic.
|
||||
Raises on non-2xx status.
|
||||
"""
|
||||
headers = {"User-Agent": USER_AGENT}
|
||||
|
||||
for attempt in range(1, MAX_RETRIES + 1):
|
||||
try:
|
||||
log(f"Attempt {attempt}/{MAX_RETRIES}: GET {url}")
|
||||
response = requests.get(url, headers=headers, timeout=TIMEOUT_SECONDS)
|
||||
response.raise_for_status()
|
||||
log(f"HTTP {response.status_code} — content length: {len(response.text):,} chars.")
|
||||
return response.text
|
||||
except requests.exceptions.HTTPError as e:
|
||||
log(f"HTTP error on attempt {attempt}: {e}", level="ERROR")
|
||||
except requests.exceptions.ConnectionError as e:
|
||||
log(f"Connection error on attempt {attempt}: {e}", level="ERROR")
|
||||
except requests.exceptions.Timeout:
|
||||
log(f"Timeout on attempt {attempt} after {TIMEOUT_SECONDS}s.", level="WARNING")
|
||||
except requests.exceptions.RequestException as e:
|
||||
log(f"Request failed on attempt {attempt}: {e}", level="ERROR")
|
||||
|
||||
if attempt < MAX_RETRIES:
|
||||
log(f"Waiting {RETRY_DELAY_SECONDS}s before retry...")
|
||||
time.sleep(RETRY_DELAY_SECONDS)
|
||||
|
||||
raise RuntimeError(f"All {MAX_RETRIES} attempts failed for URL: {url}")
|
||||
|
||||
|
||||
def find_table(html: str) -> BeautifulSoup:
|
||||
"""
|
||||
Parses the HTML and returns the target table element.
|
||||
Adjust TABLE_SELECTOR to match the table you need.
|
||||
"""
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
table = soup.find("table", TABLE_SELECTOR)
|
||||
if not table:
|
||||
raise ValueError(
|
||||
f"Target table not found. Selector used: {TABLE_SELECTOR}. "
|
||||
"Inspect the page source and update TABLE_SELECTOR in CONFIG."
|
||||
)
|
||||
log("Target table found in HTML.")
|
||||
return table
|
||||
|
||||
|
||||
def parse_table(table) -> pd.DataFrame:
|
||||
"""Converts a BeautifulSoup table element into a pandas DataFrame."""
|
||||
df = pd.read_html(str(table))[0]
|
||||
log(f"Parsed table: {len(df)} rows, {len(df.columns)} columns.")
|
||||
return df
|
||||
|
||||
|
||||
def clean_column_names(df: pd.DataFrame) -> pd.DataFrame:
|
||||
"""Normalises all column names to snake_case."""
|
||||
df.columns = (
|
||||
df.columns
|
||||
.str.strip()
|
||||
.str.lower()
|
||||
.str.replace(r"\s+", "_", regex=True)
|
||||
.str.replace(r"[^\w]", "", regex=True)
|
||||
)
|
||||
log(f"Columns after normalisation: {list(df.columns)}")
|
||||
return df
|
||||
|
||||
|
||||
def clean_data(df: pd.DataFrame) -> pd.DataFrame:
|
||||
"""
|
||||
Applies general cleaning rules.
|
||||
Extend this function for your task-specific cleaning needs.
|
||||
"""
|
||||
# Strip whitespace from string columns
|
||||
for col in df.select_dtypes(include="object").columns:
|
||||
df[col] = df[col].str.strip()
|
||||
|
||||
# Drop fully empty rows
|
||||
before = len(df)
|
||||
df = df.dropna(how="all")
|
||||
dropped = before - len(df)
|
||||
if dropped:
|
||||
log(f"Dropped {dropped} fully empty rows.", level="WARNING")
|
||||
|
||||
return df
|
||||
|
||||
|
||||
def validate(df: pd.DataFrame) -> bool:
|
||||
"""
|
||||
Checks that the DataFrame is non-empty and contains all required columns.
|
||||
Returns True if valid, False otherwise.
|
||||
"""
|
||||
if df.empty:
|
||||
log("Validation failed: DataFrame is empty.", level="ERROR")
|
||||
return False
|
||||
|
||||
missing = [col for col in REQUIRED_COLUMNS if col not in df.columns]
|
||||
if missing:
|
||||
log(
|
||||
f"Validation failed: missing required columns: {missing}. "
|
||||
f"Available columns: {list(df.columns)}",
|
||||
level="ERROR"
|
||||
)
|
||||
return False
|
||||
|
||||
log(f"Validation passed. {len(df)} rows, {len(df.columns)} columns.")
|
||||
return True
|
||||
|
||||
|
||||
def save_output(df: pd.DataFrame, path: str):
|
||||
"""Saves the validated DataFrame to CSV."""
|
||||
df.to_csv(path, index=False, encoding="utf-8")
|
||||
log(f"Output saved to: {path}")
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# MAIN
|
||||
# =============================================================================
|
||||
|
||||
def main():
|
||||
log("=" * 60)
|
||||
log("Universal Scraping Architect — Traditional / Local Scraping")
|
||||
log("=" * 60)
|
||||
|
||||
# Step 1: Fetch
|
||||
try:
|
||||
html = safe_get(TARGET_URL)
|
||||
except RuntimeError as e:
|
||||
log(str(e), level="ERROR")
|
||||
sys.exit(1)
|
||||
|
||||
# Step 2: Find table
|
||||
try:
|
||||
table = find_table(html)
|
||||
except ValueError as e:
|
||||
log(str(e), level="ERROR")
|
||||
sys.exit(1)
|
||||
|
||||
# Step 3: Parse
|
||||
try:
|
||||
df = parse_table(table)
|
||||
except Exception as e:
|
||||
log(f"Failed to parse table into DataFrame: {e}", level="ERROR")
|
||||
sys.exit(1)
|
||||
|
||||
# Step 4: Clean
|
||||
df = clean_column_names(df)
|
||||
df = clean_data(df)
|
||||
|
||||
# Step 5: Validate
|
||||
if not validate(df):
|
||||
sys.exit(1)
|
||||
|
||||
# Step 6: Save
|
||||
save_output(df, OUTPUT_FILE)
|
||||
|
||||
# Step 7: Final summary
|
||||
log("=" * 60)
|
||||
log("EXTRACTION COMPLETE")
|
||||
log(f" Source URL : {TARGET_URL}")
|
||||
log(f" Output file : {OUTPUT_FILE}")
|
||||
log(f" Rows saved : {len(df)}")
|
||||
log(f" Columns : {list(df.columns)}")
|
||||
log(f" Log file : {LOG_FILE}")
|
||||
log("=" * 60)
|
||||
log("Customisable: TARGET_URL, OUTPUT_FILE, TABLE_SELECTOR, REQUIRED_COLUMNS.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
@ -0,0 +1,44 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
validate_extraction.py
|
||||
Intelligent stdlib-only validation for JSON structures.
|
||||
"""
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
import os
|
||||
|
||||
|
||||
def validate_json(file_path):
|
||||
if not os.path.exists(file_path):
|
||||
return {"status": "error", "message": f"File {file_path} not found."}
|
||||
|
||||
try:
|
||||
with open(file_path, 'r', encoding='utf-8') as f:
|
||||
data = json.load(f)
|
||||
# Basic validation: ensure it's not empty and is a list/dict
|
||||
if not data:
|
||||
return {"status": "warning", "message": "JSON file is empty."}
|
||||
return {"status": "ok", "message": "JSON is valid and well-formed."}
|
||||
except json.JSONDecodeError as e:
|
||||
return {"status": "error", "message": f"Invalid JSON: {str(e)}"}
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="Standard Library JSON Validator")
|
||||
parser.add_argument("file", help="Path to JSON file to validate")
|
||||
parser.add_argument("--json", action="store_true", help="Output results in JSON format")
|
||||
|
||||
args = parser.parse_args()
|
||||
result = validate_json(args.file)
|
||||
|
||||
if args.json:
|
||||
print(json.dumps(result, indent=2))
|
||||
else:
|
||||
print(f"[{result['status'].upper()}] {result['message']}")
|
||||
|
||||
sys.exit(0 if result['status'] == "ok" else 1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Loading…
Add table
Reference in a new issue