{"id":"1af0e1ee-403b-4d62-87f0-7c6e5871fdd9","entityType":"agent","slug":"clawhub-skills-1kalin-afrexai-web-scraping-engine","name":"Web Scraping & Data Extraction Engine","canonicalUrl":"https://www.xpersona.co/agent/clawhub-skills-1kalin-afrexai-web-scraping-engine","canonicalPath":"/agent/clawhub-skills-1kalin-afrexai-web-scraping-engine","generatedAt":"2026-10-10T04:56:08.762Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":null},"description":"Complete web scraping methodology — legal compliance, architecture design, anti-detection, data pipelines, and production operations. Use when building scrapers, extracting web data, monitoring competitors, or automating data collection at scale. --- name: Web Scraping & Data Extraction Engine description: Complete web scraping methodology — legal compliance, architecture design, anti-detection, data pipelines, and production operations. Use when building scrapers, extracting web data, monitoring competitors, or automating data collection at scale. --- Web Scraping & Data Extraction Engine Quick Health Check (Run First) Score your scraping operation (2 points","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. Last updated 4/15/2026.","installCommand":"clawhub skill install skills:1kalin:afrexai-web-scraping-engine","sourceUrl":"https://github.com/openclaw/skills/tree/main/skills/1kalin/afrexai-web-scraping-engine","homepage":null,"primaryLinks":[{"label":"View on ClawHub","url":"https://github.com/openclaw/skills/tree/main/skills/1kalin/afrexai-web-scraping-engine","kind":"source"}],"safetyScore":84,"overallRank":62,"popularityScore":50,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Complete web scraping methodology — legal compliance, architecture design, anti-detection, data pipelines, and production operations. Use when building scrapers"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[{"label":"change","status":"self-declared"},{"label":"scrape","status":"self-declared"}],"verifiedCount":0,"selfDeclaredCount":3,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"},{"key":"change","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"},{"key":"scrape","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile capability:change|supported|profile capability:scrape|supported|profile"}},"adoption":{"evidence":{"source":"no-adoption-signals","verified":false,"confidence":"low","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":"No source adoption metrics were available."},"stars":null,"forks":null,"downloads":null,"packageName":null,"latestVersion":null,"tractionLabel":null},"release":{"evidence":{"source":"agent-index","verified":false,"confidence":"medium","updatedAt":"2026-02-25T05:53:01.942Z","emptyReason":null},"lastUpdatedAt":"2026-04-15T00:45:39.800Z","lastCrawledAt":"2026-02-25T05:53:01.942Z","lastIndexedAt":null,"nextCrawlAt":"2026-02-26T05:53:01.942Z","lastVerifiedAt":null,"highlights":[]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install skills:1kalin:afrexai-web-scraping-engine","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-web-scraping-engine/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-web-scraping-engine/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-web-scraping-engine/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-web-scraping-engine/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-web-scraping-engine/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-web-scraping-engine/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T04:56:08.761Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-web-scraping-engine/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-web-scraping-engine/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-web-scraping-engine/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-web-scraping-engine/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":null},"readme":"---\nname: Web Scraping & Data Extraction Engine\ndescription: Complete web scraping methodology — legal compliance, architecture design, anti-detection, data pipelines, and production operations. Use when building scrapers, extracting web data, monitoring competitors, or automating data collection at scale.\n---\n\n# Web Scraping & Data Extraction Engine\n\n## Quick Health Check (Run First)\n\nScore your scraping operation (2 points each):\n\n| Signal | Healthy | Unhealthy |\n|--------|---------|-----------|\n| Legal compliance | robots.txt checked, ToS reviewed | Scraping blindly |\n| Architecture | Tool matches site complexity | Using Puppeteer for static HTML |\n| Anti-detection | Rotation, delays, fingerprint diversity | Single IP, no delays |\n| Data quality | Validation + dedup pipeline | Raw dumps, no cleaning |\n| Error handling | Retry logic, circuit breakers | Crashes on first 403 |\n| Monitoring | Success rates tracked, alerts set | No visibility |\n| Storage | Structured, deduplicated, versioned | Flat files, duplicates |\n| Scheduling | Appropriate frequency, off-peak | Hammering during business hours |\n\n**Score: /16** → 12+: Production-ready | 8-11: Needs work | <8: Stop and redesign\n\n---\n\n## Phase 1: Legal & Ethical Foundation\n\n### Pre-Scrape Compliance Checklist\n\n```yaml\ncompliance_brief:\n  target_domain: \"\"\n  date_assessed: \"\"\n  \n  robots_txt:\n    checked: false\n    target_paths_allowed: false\n    crawl_delay_specified: \"\"\n    ai_bot_rules: \"\"  # Many sites now block AI crawlers specifically\n    \n  terms_of_service:\n    reviewed: false\n    scraping_mentioned: false\n    scraping_prohibited: false\n    api_available: false\n    api_sufficient: false\n    \n  data_classification:\n    type: \"\"  # public-factual | public-personal | behind-auth | copyrighted\n    contains_pii: false\n    pii_types: []  # name, email, phone, address, photo\n    gdpr_applies: false  # EU residents' data\n    ccpa_applies: false  # California residents' data\n    \n  legal_risk: \"\"  # low | medium | high | do-not-scrape\n  decision: \"\"  # proceed | use-api | request-permission | abandon\n  justification: \"\"\n```\n\n### Legal Landscape Quick Reference\n\n| Scenario | Risk Level | Key Case Law |\n|----------|-----------|--------------|\n| Public data, no login, robots.txt allows | LOW | hiQ v. LinkedIn (2022) |\n| Public data, robots.txt disallows | MEDIUM | Meta v. Bright Data (2024) |\n| Behind authentication | HIGH | Van Buren v. US (2021), CFAA |\n| Personal data without consent | HIGH | GDPR Art. 6, CCPA §1798.100 |\n| Republishing copyrighted content | HIGH | Copyright Act §106 |\n| Price/product comparison | LOW | eBay v. Bidder's Edge (fair use) |\n| Academic/research use | LOW-MEDIUM | Varies by jurisdiction |\n| Bypassing anti-bot measures | HIGH | CFAA \"exceeds authorized access\" |\n\n### Decision Rules\n\n1. **API exists and covers your needs?** → Use the API. Always.\n2. **robots.txt disallows your target?** → Respect it unless you have written permission.\n3. **Data behind login?** → Do not scrape without explicit authorization.\n4. **Contains PII?** → GDPR/CCPA compliance required before collection.\n5. **Copyrighted content?** → Extract facts/data points only, never full content.\n6. **Site explicitly prohibits scraping?** → Request permission or find alternative source.\n\n### AI Crawler Considerations (2025+)\n\nMany sites now specifically block AI-related crawlers:\n\n```\n# Common AI bot blocks in robots.txt\nUser-agent: GPTBot\nUser-agent: ChatGPT-User\nUser-agent: Google-Extended\nUser-agent: CCBot\nUser-agent: anthropic-ai\nUser-agent: ClaudeBot\nUser-agent: Bytespider\nUser-agent: PerplexityBot\n```\n\n**Rule**: If collecting data for AI training, check for these specific blocks.\n\n---\n\n## Phase 2: Architecture Decision\n\n### Tool Selection Matrix\n\n| Tool/Approach | Best For | Speed | JS Support | Complexity | Cost |\n|---------------|----------|-------|------------|------------|------|\n| HTTP client (requests/axios) | Static HTML, APIs | ⚡⚡⚡ | ❌ | Low | Free |\n| Beautiful Soup / Cheerio | Static HTML parsing | ⚡⚡⚡ | ❌ | Low | Free |\n| Scrapy | Large-scale structured crawling | ⚡⚡⚡ | Plugin | Medium | Free |\n| Playwright / Puppeteer | JS-rendered, SPAs, interactions | ⚡ | ✅ | Medium | Free |\n| Selenium | Legacy, browser automation | ⚡ | ✅ | High | Free |\n| Crawlee | Hybrid (HTTP + browser fallback) | ⚡⚡ | ✅ | Medium | Free |\n| Firecrawl / ScrapingBee | Managed, anti-bot bypass | ⚡⚡ | ✅ | Low | Paid |\n| Bright Data / Oxylabs | Enterprise, proxy + browser | ⚡⚡ | ✅ | Low | Paid |\n\n### Decision Tree\n\n```\nIs the content in the initial HTML source?\n├── YES → Is the site structure consistent?\n│   ├── YES → Static scraper (requests + BeautifulSoup/Cheerio)\n│   └── NO → Scrapy with custom parsers\n└── NO → Does the page require user interaction?\n    ├── YES → Playwright/Puppeteer with interaction scripts\n    └── NO → Playwright in non-interactive mode\n        └── At scale (>10K pages)? → Crawlee (hybrid mode)\n            └── Heavy anti-bot? → Managed service (Firecrawl/ScrapingBee)\n```\n\n### Architecture Brief YAML\n\n```yaml\nscraping_project:\n  name: \"\"\n  objective: \"\"  # What data, why, how often\n  \n  targets:\n    - domain: \"\"\n      pages_estimated: 0\n      rendering: \"static\" | \"javascript\" | \"spa\"\n      anti_bot: \"none\" | \"basic\" | \"cloudflare\" | \"advanced\"\n      rate_limit: \"\"  # requests per second safe limit\n      \n  tool_selected: \"\"\n  justification: \"\"\n  \n  data_schema:\n    fields: []\n    output_format: \"\"  # json | csv | database\n    \n  schedule:\n    frequency: \"\"  # once | hourly | daily | weekly\n    preferred_time: \"\"  # off-peak for target timezone\n    \n  infrastructure:\n    proxy_needed: false\n    proxy_type: \"\"  # residential | datacenter | mobile\n    storage: \"\"\n    monitoring: \"\"\n```\n\n---\n\n## Phase 3: Request Engineering\n\n### HTTP Request Best Practices\n\n```python\n# Python example — production request pattern\nimport requests\nfrom requests.adapters import HTTPAdapter\nfrom urllib3.util.retry import Retry\n\nsession = requests.Session()\n\n# Retry strategy\nretry = Retry(\n    total=3,\n    backoff_factor=1,      # 1s, 2s, 4s\n    status_forcelist=[429, 500, 502, 503, 504],\n    respect_retry_after_header=True\n)\nsession.mount(\"https://\", HTTPAdapter(max_retries=retry))\n\n# Realistic headers\nsession.headers.update({\n    \"User-Agent\": \"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36\",\n    \"Accept\": \"text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8\",\n    \"Accept-Language\": \"en-US,en;q=0.9\",\n    \"Accept-Encoding\": \"gzip, deflate, br\",\n    \"Connection\": \"keep-alive\",\n    \"Cache-Control\": \"no-cache\",\n})\n```\n\n### Header Rotation Strategy\n\nRotate these to avoid fingerprinting:\n\n| Header | Rotation Pool Size | Notes |\n|--------|-------------------|-------|\n| User-Agent | 20-50 real browser UAs | Match OS distribution |\n| Accept-Language | 5-10 locale combos | Match proxy geo |\n| Sec-Ch-Ua | Match User-Agent | Chrome/Edge/Brave |\n| Referer | Vary per request | Previous page or search engine |\n\n### Rate Limiting Rules\n\n| Site Type | Safe Delay | Aggressive (risky) |\n|-----------|-----------|-------------------|\n| Small business site | 5-10 seconds | 2-3 seconds |\n| Medium site | 2-5 seconds | 1-2 seconds |\n| Large platform (Amazon, etc.) | 3-5 seconds | 1 second |\n| API endpoint | Per API docs | Never exceed |\n| robots.txt crawl-delay | Respect exactly | Never below |\n\n**Rules:**\n1. Always respect `Crawl-delay` in robots.txt\n2. Add random jitter (±30%) to avoid pattern detection\n3. Slow down during business hours for smaller sites\n4. Respect `Retry-After` headers — they mean it\n5. Watch for 429s — back off exponentially (2x each time)\n\n---\n\n## Phase 4: Parsing & Extraction\n\n### CSS Selector Strategy (Priority Order)\n\n1. **Data attributes** → `[data-product-id]`, `[data-price]` (most stable)\n2. **Semantic IDs** → `#product-title`, `#price` (stable but can change)\n3. **ARIA attributes** → `[aria-label=\"Price\"]` (accessibility, fairly stable)\n4. **Semantic HTML** → `article`, `main`, `nav` (structural, stable)\n5. **Class names** → `.product-card` (can change with redesigns)\n6. **XPath position** → `//div[3]/span[2]` (FRAGILE — last resort)\n\n### Extraction Patterns\n\n**Structured data first** — Check before writing CSS selectors:\n\n```python\n# 1. Check JSON-LD (best source — structured, clean)\nimport json\nfrom bs4 import BeautifulSoup\n\nsoup = BeautifulSoup(html, 'html.parser')\nfor script in soup.find_all('script', type='application/ld+json'):\n    data = json.loads(script.string)\n    # Often contains: Product, Article, Organization, etc.\n\n# 2. Check Open Graph meta tags\nog_title = soup.find('meta', property='og:title')\nog_price = soup.find('meta', property='product:price:amount')\n\n# 3. Check microdata\nitems = soup.find_all(itemtype=True)\n\n# 4. Fall back to CSS selectors only if above are empty\n```\n\n**Table extraction pattern:**\n\n```python\nimport pandas as pd\n\n# Quick table extraction\ntables = pd.read_html(html)  # Returns list of DataFrames\n\n# For complex tables with merged cells\ndef extract_table(soup, selector):\n    table = soup.select_one(selector)\n    headers = [th.get_text(strip=True) for th in table.select('thead th')]\n    rows = []\n    for tr in table.select('tbody tr'):\n        cells = [td.get_text(strip=True) for td in tr.select('td')]\n        rows.append(dict(zip(headers, cells)))\n    return rows\n```\n\n**Pagination handling:**\n\n```python\n# Pattern 1: Next button\nwhile True:\n    # ... scrape current page ...\n    next_link = soup.select_one('a.next-page, [rel=\"next\"], .pagination .next a')\n    if not next_link or not next_link.get('href'):\n        break\n    url = urljoin(base_url, next_link['href'])\n    \n# Pattern 2: API pagination (infinite scroll sites)\npage = 1\nwhile True:\n    resp = session.get(f\"{api_url}?page={page}&limit=50\")\n    data = resp.json()\n    if not data.get('results'):\n        break\n    # ... process results ...\n    page += 1\n\n# Pattern 3: Cursor-based\ncursor = None\nwhile True:\n    params = {\"limit\": 50}\n    if cursor:\n        params[\"cursor\"] = cursor\n    resp = session.get(api_url, params=params)\n    data = resp.json()\n    # ... process ...\n    cursor = data.get('next_cursor')\n    if not cursor:\n        break\n```\n\n### JavaScript-Rendered Content\n\n```python\n# Playwright pattern for JS-rendered pages\nfrom playwright.sync_api import sync_playwright\n\nwith sync_playwright() as p:\n    browser = p.chromium.launch(headless=True)\n    context = browser.new_context(\n        viewport={\"width\": 1920, \"height\": 1080},\n        user_agent=\"Mozilla/5.0 ...\",\n    )\n    page = context.new_page()\n    \n    # Block unnecessary resources (speed + stealth)\n    page.route(\"**/*.{png,jpg,jpeg,gif,svg,woff,woff2}\", \n               lambda route: route.abort())\n    \n    page.goto(url, wait_until=\"networkidle\")\n    \n    # Wait for specific content (better than arbitrary sleep)\n    page.wait_for_selector('[data-product-id]', timeout=10000)\n    \n    # Extract after JS rendering\n    content = page.content()\n    # ... parse with BeautifulSoup/Cheerio ...\n    \n    browser.close()\n```\n\n---\n\n## Phase 5: Anti-Detection & Stealth\n\n### Detection Signals (What Sites Check)\n\n| Signal | Detection Method | Mitigation |\n|--------|-----------------|------------|\n| IP reputation | IP blacklists, datacenter ranges | Residential proxies |\n| Request rate | Requests/min from same IP | Rate limiting + jitter |\n| TLS fingerprint | JA3/JA4 hash matching | Use real browser or curl-impersonate |\n| Browser fingerprint | Canvas, WebGL, fonts | Playwright with stealth plugin |\n| JavaScript challenges | Cloudflare Turnstile, hCaptcha | Managed browser services |\n| Cookie/session behavior | Missing cookies, no history | Full session management |\n| Navigation pattern | Direct URL hits, no referrer | Simulate natural browsing |\n| Mouse/keyboard events | No interaction telemetry | Event simulation (Playwright) |\n| Header consistency | Mismatched headers vs UA | Header sets that match |\n\n### Proxy Strategy\n\n```yaml\nproxy_strategy:\n  # Tier 1: Free/Datacenter (for non-protected sites)\n  basic:\n    type: \"datacenter\"\n    cost: \"$1-5/GB\"\n    success_rate: \"60-80%\"\n    use_for: \"APIs, small sites, no anti-bot\"\n    \n  # Tier 2: Residential (for most protected sites)\n  standard:\n    type: \"residential\"\n    cost: \"$5-15/GB\"\n    success_rate: \"90-95%\"\n    use_for: \"Cloudflare, major platforms\"\n    rotation: \"per-request or sticky 10min\"\n    \n  # Tier 3: Mobile/ISP (for maximum stealth)\n  premium:\n    type: \"mobile\"\n    cost: \"$15-30/GB\"\n    success_rate: \"95-99%\"\n    use_for: \"Aggressive anti-bot, social media\"\n    \n  rules:\n    - Start with cheapest tier, escalate only on blocks\n    - Match proxy geo to target audience geo\n    - Rotate on 403/429, not every request\n    - Use sticky sessions for multi-page scrapes\n    - Monitor proxy health — remove slow/blocked IPs\n```\n\n### Playwright Stealth Configuration\n\n```python\n# Essential stealth for Playwright\nfrom playwright.sync_api import sync_playwright\n\nwith sync_playwright() as p:\n    browser = p.chromium.launch(\n        headless=True,\n        args=[\n            '--disable-blink-features=AutomationControlled',\n            '--disable-features=IsolateOrigins,site-per-process',\n        ]\n    )\n    context = browser.new_context(\n        viewport={\"width\": 1920, \"height\": 1080},\n        locale=\"en-US\",\n        timezone_id=\"America/New_York\",\n        geolocation={\"latitude\": 40.7128, \"longitude\": -74.0060},\n        permissions=[\"geolocation\"],\n    )\n    \n    # Remove automation indicators\n    page = context.new_page()\n    page.add_init_script(\"\"\"\n        Object.defineProperty(navigator, 'webdriver', {get: () => undefined});\n        Object.defineProperty(navigator, 'plugins', {get: () => [1, 2, 3]});\n    \"\"\")\n```\n\n### Cloudflare Bypass Decision\n\n```\nCloudflare detected?\n├── JS Challenge only → Playwright with stealth + residential proxy\n├── Turnstile CAPTCHA → Managed service (ScrapingBee/Bright Data)\n├── Under Attack Mode → Wait, try later, or managed service\n└── WAF blocking → Different approach needed\n    ├── Check for API endpoints (network tab)\n    ├── Check for mobile app API\n    └── Consider if data is available elsewhere\n```\n\n---\n\n## Phase 6: Data Pipeline & Quality\n\n### Data Validation Rules\n\n```python\n# Validation pattern — validate BEFORE storing\nfrom dataclasses import dataclass, field\nfrom typing import Optional\nimport re\nfrom datetime import datetime\n\n@dataclass\nclass ScrapedProduct:\n    url: str\n    title: str\n    price: Optional[float]\n    currency: str = \"USD\"\n    scraped_at: str = field(default_factory=lambda: datetime.utcnow().isoformat())\n    \n    def validate(self) -> list[str]:\n        errors = []\n        if not self.url.startswith('http'):\n            errors.append(\"Invalid URL\")\n        if not self.title or len(self.title) < 3:\n            errors.append(\"Title too short or missing\")\n        if self.price is not None and self.price < 0:\n            errors.append(\"Negative price\")\n        if self.price is not None and self.price > 1_000_000:\n            errors.append(\"Price suspiciously high — verify\")\n        if self.currency not in (\"USD\", \"EUR\", \"GBP\", \"BTC\"):\n            errors.append(f\"Unknown currency: {self.currency}\")\n        return errors\n```\n\n### Deduplication Strategy\n\n| Method | When to Use | Implementation |\n|--------|------------|----------------|\n| URL-based | Pages with unique URLs | Hash the canonical URL |\n| Content hash | Same URL, changing content | MD5/SHA256 of key fields |\n| Fuzzy matching | Near-duplicate detection | Jaccard similarity > 0.85 |\n| Composite key | Multi-field uniqueness | Hash(domain + product_id + variant) |\n\n```python\nimport hashlib\n\ndef dedup_key(item: dict, fields: list[str]) -> str:\n    \"\"\"Generate dedup key from selected fields.\"\"\"\n    values = \"|\".join(str(item.get(f, \"\")) for f in fields)\n    return hashlib.sha256(values.encode()).hexdigest()\n\n# Usage\nseen = set()\nfor item in scraped_items:\n    key = dedup_key(item, [\"url\", \"product_id\"])\n    if key not in seen:\n        seen.add(key)\n        clean_items.append(item)\n```\n\n### Data Cleaning Pipeline\n\n```\nRaw HTML → Parse → Extract → Validate → Clean → Deduplicate → Store\n                                ↓\n                          Quarantine (failed validation)\n```\n\n**Common cleaning operations:**\n\n| Problem | Solution |\n|---------|----------|\n| HTML entities (`&amp;`) | `html.unescape()` |\n| Extra whitespace | `\" \".join(text.split())` |\n| Unicode issues | `unicodedata.normalize('NFKD', text)` |\n| Price in text (\"$49.99\") | Regex: `r'[\\$£€]?([\\d,]+\\.?\\d*)'` |\n| Date formats vary | `dateutil.parser.parse()` with `dayfirst` flag |\n| Relative URLs | `urllib.parse.urljoin(base, relative)` |\n| Encoding issues | `chardet.detect()` then decode |\n\n---\n\n## Phase 7: Storage & Export\n\n### Storage Decision Guide\n\n| Volume | Frequency | Query Needs | Recommendation |\n|--------|-----------|-------------|----------------|\n| <10K records | One-time | None | JSON/CSV files |\n| <10K records | Recurring | Simple lookups | SQLite |\n| 10K-1M records | Recurring | Complex queries | PostgreSQL |\n| 1M+ records | Continuous | Analytics | PostgreSQL + partitioning |\n| Append-only logs | Continuous | Time-series | ClickHouse / TimescaleDB |\n\n### SQLite Pattern (Most Common)\n\n```python\nimport sqlite3\nimport json\nfrom datetime import datetime\n\ndef init_db(path=\"scraper_data.db\"):\n    conn = sqlite3.connect(path)\n    conn.execute(\"\"\"\n        CREATE TABLE IF NOT EXISTS items (\n            id INTEGER PRIMARY KEY,\n            url TEXT UNIQUE,\n            data JSON NOT NULL,\n            scraped_at TEXT DEFAULT (datetime('now')),\n            updated_at TEXT,\n            checksum TEXT\n        )\n    \"\"\")\n    conn.execute(\"CREATE INDEX IF NOT EXISTS idx_url ON items(url)\")\n    conn.execute(\"CREATE INDEX IF NOT EXISTS idx_scraped ON items(scraped_at)\")\n    return conn\n\ndef upsert(conn, url, data, checksum):\n    conn.execute(\"\"\"\n        INSERT INTO items (url, data, checksum) VALUES (?, ?, ?)\n        ON CONFLICT(url) DO UPDATE SET\n            data = excluded.data,\n            updated_at = datetime('now'),\n            checksum = excluded.checksum\n        WHERE items.checksum != excluded.checksum\n    \"\"\", (url, json.dumps(data), checksum))\n    conn.commit()\n```\n\n### Export Formats\n\n```python\n# CSV export\nimport csv\ndef to_csv(items, path, fields):\n    with open(path, 'w', newline='') as f:\n        writer = csv.DictWriter(f, fieldnames=fields)\n        writer.writeheader()\n        writer.writerows(items)\n\n# JSON Lines (best for large datasets — streaming)\ndef to_jsonl(items, path):\n    with open(path, 'w') as f:\n        for item in items:\n            f.write(json.dumps(item) + '\\n')\n\n# Incremental export (only new/changed since last export)\ndef export_since(conn, last_export_time):\n    cursor = conn.execute(\n        \"SELECT data FROM items WHERE scraped_at > ? OR updated_at > ?\",\n        (last_export_time, last_export_time)\n    )\n    return [json.loads(row[0]) for row in cursor]\n```\n\n---\n\n## Phase 8: Error Handling & Resilience\n\n### Error Classification\n\n| HTTP Code | Meaning | Action |\n|-----------|---------|--------|\n| 200 | Success | Process normally |\n| 301/302 | Redirect | Follow (max 5 hops) |\n| 403 | Forbidden/blocked | Rotate proxy, slow down |\n| 404 | Not found | Log, skip, mark URL dead |\n| 429 | Rate limited | Respect Retry-After, back off 2x |\n| 500-504 | Server error | Retry 3x with backoff |\n| Connection timeout | Network issue | Retry with different proxy |\n| SSL error | Certificate issue | Log, investigate, skip |\n\n### Circuit Breaker Pattern\n\n```python\nclass CircuitBreaker:\n    def __init__(self, failure_threshold=5, reset_timeout=300):\n        self.failures = 0\n        self.threshold = failure_threshold\n        self.reset_timeout = reset_timeout\n        self.last_failure = 0\n        self.state = \"closed\"  # closed | open | half-open\n    \n    def record_failure(self):\n        self.failures += 1\n        self.last_failure = time.time()\n        if self.failures >= self.threshold:\n            self.state = \"open\"\n            # Alert: \"Circuit open — too many failures\"\n    \n    def record_success(self):\n        self.failures = 0\n        self.state = \"closed\"\n    \n    def can_proceed(self):\n        if self.state == \"closed\":\n            return True\n        if self.state == \"open\":\n            if time.time() - self.last_failure > self.reset_timeout:\n                self.state = \"half-open\"\n                return True  # Try one request\n            return False\n        return True  # half-open: allow attempt\n```\n\n### Checkpoint & Resume\n\n```python\nimport json\nfrom pathlib import Path\n\nclass Checkpointer:\n    def __init__(self, path=\"checkpoint.json\"):\n        self.path = Path(path)\n        self.state = self._load()\n    \n    def _load(self):\n        if self.path.exists():\n            return json.loads(self.path.read_text())\n        return {\"completed_urls\": [], \"last_page\": 0, \"cursor\": None}\n    \n    def save(self):\n        self.path.write_text(json.dumps(self.state))\n    \n    def is_done(self, url):\n        return url in self.state[\"completed_urls\"]\n    \n    def mark_done(self, url):\n        self.state[\"completed_urls\"].append(url)\n        if len(self.state[\"completed_urls\"]) % 50 == 0:\n            self.save()  # Periodic save\n```\n\n---\n\n## Phase 9: Monitoring & Operations\n\n### Scraper Health Dashboard\n\n```yaml\ndashboard:\n  real_time:\n    - metric: \"requests_per_minute\"\n      alert_if: \"> 60 for small sites\"\n    - metric: \"success_rate\"\n      alert_if: \"< 90%\"\n    - metric: \"avg_response_time_ms\"\n      alert_if: \"> 5000\"\n    - metric: \"blocked_rate\"\n      alert_if: \"> 10%\"\n      \n  per_run:\n    - metric: \"pages_scraped\"\n    - metric: \"items_extracted\"\n    - metric: \"items_validated\"\n    - metric: \"items_deduplicated\"\n    - metric: \"new_items\"\n    - metric: \"updated_items\"\n    - metric: \"errors_by_type\"\n    - metric: \"run_duration\"\n    - metric: \"proxy_cost\"\n    \n  weekly:\n    - metric: \"data_freshness\"\n      description: \"% of records updated in last 7 days\"\n    - metric: \"site_structure_changes\"\n      description: \"Selectors that stopped matching\"\n    - metric: \"total_cost\"\n      description: \"Proxy + compute + storage\"\n```\n\n### Breakage Detection\n\nSites redesign. Selectors break. Detect it early:\n\n```python\ndef health_check(results: list[dict], expected_fields: list[str]) -> dict:\n    \"\"\"Check if scraper is still extracting correctly.\"\"\"\n    total = len(results)\n    if total == 0:\n        return {\"status\": \"CRITICAL\", \"message\": \"Zero results — likely broken\"}\n    \n    field_coverage = {}\n    for field in expected_fields:\n        filled = sum(1 for r in results if r.get(field))\n        coverage = filled / total\n        field_coverage[field] = coverage\n        \n    issues = []\n    for field, coverage in field_coverage.items():\n        if coverage < 0.5:\n            issues.append(f\"{field}: {coverage:.0%} fill rate (expected >50%)\")\n    \n    if issues:\n        return {\"status\": \"WARNING\", \"issues\": issues}\n    return {\"status\": \"OK\", \"field_coverage\": field_coverage}\n```\n\n### Operational Runbook\n\n**Daily:**\n- Check success rate per target domain\n- Review error logs for new patterns\n- Verify data freshness\n\n**Weekly:**\n- Compare extraction counts vs baseline (>20% drop = investigate)\n- Review proxy spend\n- Spot-check 10 random records for accuracy\n\n**Monthly:**\n- Full selector validation against live pages\n- Review legal compliance (robots.txt changes, ToS updates)\n- Cost optimization review\n- Prune dead URLs from queue\n\n---\n\n## Phase 10: Common Scraping Patterns\n\n### Pattern 1: E-commerce Price Monitor\n\n```yaml\nuse_case: \"Track competitor prices daily\"\ntool: \"requests + BeautifulSoup\"\nschedule: \"Daily at 03:00 UTC (off-peak)\"\ntargets: [\"competitor-a.com/products\", \"competitor-b.com/api\"]\ndata:\n  - product_id\n  - product_name\n  - price\n  - currency\n  - in_stock\n  - scraped_at\nstorage: \"SQLite with price history\"\nalerts: \"Price change > 10% → notify\"\n```\n\n### Pattern 2: Job Board Aggregator\n\n```yaml\nuse_case: \"Aggregate job listings from multiple boards\"\ntool: \"Scrapy with per-site spiders\"\nschedule: \"Every 6 hours\"\ntargets: [\"board-a.com\", \"board-b.com\", \"board-c.com\"]\ndata:\n  - title\n  - company\n  - location\n  - salary_range\n  - posted_date\n  - url\n  - source\ndedup: \"Hash(title + company + location)\"\nstorage: \"PostgreSQL\"\n```\n\n### Pattern 3: News & Content Monitor\n\n```yaml\nuse_case: \"Monitor industry news mentions\"\ntool: \"requests + RSS feeds (preferred) + web fallback\"\nschedule: \"Every 30 minutes\"\napproach:\n  1: \"RSS/Atom feeds (fastest, cleanest)\"\n  2: \"Google News RSS for topic\"\n  3: \"Direct scraping if no feed\"\ndata:\n  - headline\n  - source\n  - url\n  - published_at\n  - snippet\n  - sentiment\nalerts: \"Keyword match → immediate notification\"\n```\n\n### Pattern 4: Social Media Intelligence\n\n```yaml\nuse_case: \"Track brand mentions and sentiment\"\ntool: \"Official APIs (always) + web search fallback\"\nrules:\n  - NEVER scrape social platforms directly — use APIs\n  - Twitter/X: Official API ($100/mo basic)\n  - Reddit: Official API (free tier available)\n  - LinkedIn: No scraping (aggressive legal action)\n  - Instagram: Official API only (Meta Business)\nfallback: \"Brave/Google search for public mentions\"\n```\n\n### Pattern 5: Real Estate Listings\n\n```yaml\nuse_case: \"Track property listings and prices\"\ntool: \"Playwright (most listing sites are JS-heavy)\"\nschedule: \"Daily\"\nchallenges:\n  - Heavy JavaScript rendering\n  - Anti-bot measures (Cloudflare common)\n  - Frequent layout changes\n  - Map-based results\napproach: \"API endpoint discovery via network tab first\"\n```\n\n---\n\n## Phase 11: Scaling Strategies\n\n### Concurrency Architecture\n\n```\nSingle machine (small scale):\n├── asyncio + aiohttp (Python) → 50-200 concurrent requests\n├── Worker pool (ThreadPoolExecutor) → 10-50 threads\n└── Scrapy reactor → Built-in concurrency\n\nMulti-machine (large scale):\n├── URL queue: Redis / RabbitMQ / SQS\n├── Workers: Multiple Scrapy/custom workers\n├── Results: Shared PostgreSQL / S3\n└── Coordinator: Celery / custom scheduler\n```\n\n### Cost Optimization\n\n| Lever | Impact | How |\n|-------|--------|-----|\n| Static > Browser | 10-50x cheaper | Always try HTTP first |\n| Block images/CSS/fonts | 60-80% bandwidth saved | Route filtering |\n| Cache DNS | Minor but cumulative | Local DNS cache |\n| Compress responses | 50-70% bandwidth | Accept-Encoding: gzip, br |\n| Smart scheduling | Avoid redundant scrapes | Change detection before full re-scrape |\n| Proxy tier matching | 3-10x cost difference | Don't use residential for easy sites |\n\n---\n\n## Phase 12: Advanced Patterns\n\n### API Discovery (Network Tab Mining)\n\nBefore building a scraper, check if the site has hidden API endpoints:\n\n1. Open DevTools → Network tab\n2. Filter by XHR/Fetch\n3. Navigate the site, click load-more, filter/sort\n4. Look for JSON responses — these are your goldmine\n5. Most SPAs load data via REST/GraphQL APIs\n\n**Common hidden API patterns:**\n- `/api/v1/products?page=1&limit=20`\n- `/graphql` with query parameters\n- `/_next/data/...` (Next.js data routes)\n- `/wp-json/wp/v2/posts` (WordPress)\n\n### Headless Browser Optimization\n\n```python\n# Minimize browser resource usage\ncontext = browser.new_context(\n    viewport={\"width\": 1280, \"height\": 720},\n    java_script_enabled=True,  # Only if needed\n    has_touch=False,\n    is_mobile=False,\n)\n\n# Block resource types you don't need\npage.route(\"**/*\", lambda route: (\n    route.abort() if route.request.resource_type in \n    [\"image\", \"stylesheet\", \"font\", \"media\"] \n    else route.continue_()\n))\n```\n\n### Scraping Behind Authentication\n\n```python\n# When authorized to scrape behind login\n# ALWAYS use session-based auth, never store passwords in code\n\n# Pattern: Login once, reuse session\nsession = requests.Session()\nlogin_resp = session.post(\"https://example.com/login\", data={\n    \"username\": os.environ[\"SCRAPE_USER\"],\n    \"password\": os.environ[\"SCRAPE_PASS\"],\n})\nassert login_resp.ok, \"Login failed\"\n\n# Session cookies are now stored — use for subsequent requests\ndata_resp = session.get(\"https://example.com/api/data\")\n```\n\n### Change Detection (Avoid Redundant Scrapes)\n\n```python\nimport hashlib\n\ndef has_changed(url, session, last_etag=None, last_modified=None):\n    \"\"\"Check if page changed without downloading full content.\"\"\"\n    headers = {}\n    if last_etag:\n        headers[\"If-None-Match\"] = last_etag\n    if last_modified:\n        headers[\"If-Modified-Since\"] = last_modified\n    \n    resp = session.head(url, headers=headers)\n    \n    if resp.status_code == 304:\n        return False, resp.headers.get(\"ETag\"), resp.headers.get(\"Last-Modified\")\n    \n    return True, resp.headers.get(\"ETag\"), resp.headers.get(\"Last-Modified\")\n```\n\n---\n\n## Quality Scoring Rubric (0-100)\n\n| Dimension | Weight | What to Assess |\n|-----------|--------|---------------|\n| Legal compliance | 20% | robots.txt, ToS, PII handling, audit trail |\n| Data quality | 20% | Validation, accuracy, completeness, freshness |\n| Resilience | 15% | Error handling, retries, circuit breakers, checkpointing |\n| Anti-detection | 15% | Proxy rotation, fingerprint diversity, rate limiting |\n| Architecture | 10% | Right tool selection, clean code, modularity |\n| Monitoring | 10% | Success rates, breakage detection, alerting |\n| Performance | 5% | Speed, cost efficiency, resource usage |\n| Documentation | 5% | Runbook, schema docs, legal assessment |\n\n**Grading:** 90+ Excellent | 75-89 Good | 60-74 Needs work | <60 Redesign\n\n---\n\n## 10 Common Mistakes\n\n| # | Mistake | Fix |\n|---|---------|-----|\n| 1 | No robots.txt check | Always check first — it's your legal defense |\n| 2 | Fixed delays (no jitter) | Add ±30% random jitter to all delays |\n| 3 | No data validation | Validate every field before storing |\n| 4 | Using browser for static HTML | HTTP client is 10-50x faster and cheaper |\n| 5 | Single IP, no rotation | Proxy rotation for any serious scraping |\n| 6 | No breakage detection | Monitor extraction counts and field fill rates |\n| 7 | Storing raw HTML only | Extract + structure immediately |\n| 8 | No checkpoint/resume | Long scrapes must be resumable |\n| 9 | Ignoring structured data | JSON-LD/microdata is cleaner than CSS selectors |\n| 10 | Scraping when API exists | Always check for API first |\n\n---\n\n## 5 Edge Cases\n\n1. **Single-page apps (React/Vue/Angular)**: Must use browser rendering OR find the underlying API (network tab). Prefer API discovery — it's faster and more reliable.\n\n2. **Infinite scroll**: Intercept the XHR/fetch calls that load more content. Simulate scrolling only as last resort. The API endpoint usually accepts `page` or `offset` params.\n\n3. **CAPTCHAs**: If you're hitting CAPTCHAs, you're scraping too aggressively. Slow down first. If CAPTCHAs persist: managed services (2Captcha, Anti-Captcha) or rethink approach.\n\n4. **Dynamic class names** (CSS modules, Tailwind): Use data attributes, ARIA labels, or text content selectors instead. `[data-testid=\"price\"]` survives redesigns. `.sc-bdVTJa` does not.\n\n5. **Multi-language sites**: Detect language via `html[lang]` attribute. Set `Accept-Language` header to get desired locale. Watch for different URL structures (`/en/`, `/de/`, subdomains).\n\n---\n\n## Natural Language Commands\n\n1. **\"Check if I can scrape [URL]\"** → Run compliance checklist (robots.txt, ToS, data type)\n2. **\"What tool should I use for [site]?\"** → Analyze site rendering, anti-bot, recommend tool\n3. **\"Build a scraper for [description]\"** → Full architecture brief + code pattern\n4. **\"My scraper is getting blocked\"** → Anti-detection diagnostic + proxy/stealth recommendations\n5. **\"Extract [data] from [URL]\"** → Check structured data first, then CSS selectors\n6. **\"Monitor [site] for changes\"** → Change detection + scheduling + alerting setup\n7. **\"How do I handle pagination on [site]?\"** → Identify pagination type + code pattern\n8. **\"Scrape at scale ([N] pages)\"** → Concurrency architecture + cost estimate\n9. **\"Clean and store this scraped data\"** → Validation + dedup + storage recommendation\n10. **\"Is my scraper healthy?\"** → Run health check + breakage detection\n11. **\"Find the API behind [site]\"** → Network tab mining guide + common patterns\n12. **\"Set up price monitoring for [competitors]\"** → Full e-commerce monitor pattern\n","readmeExcerpt":"--- name: Web Scraping & Data Extraction Engine description: Complete web scraping methodology — legal compliance, architecture design, anti-detection, data pipelines, and production operations. Use when building scrapers, extracting web data, monitoring competitors, or automating data collection at scale. --- Web Scraping & Data Extraction Engine Quick Health Check (Run First) Score your scraping operation (2 points","codeSnippets":[],"executableExamples":[{"language":"yaml","snippet":"compliance_brief:\n  target_domain: \"\"\n  date_assessed: \"\"\n  \n  robots_txt:\n    checked: false\n    target_paths_allowed: false\n    crawl_delay_specified: \"\"\n    ai_bot_rules: \"\"  # Many sites now block AI crawlers specifically\n    \n  terms_of_service:\n    reviewed: false\n    scraping_mentioned: false\n    scraping_prohibited: false\n    api_available: false\n    api_sufficient: false\n    \n  data_classification:\n    type: \"\"  # public-factual | public-personal | behind-auth | copyrighted\n    contains_pii: false\n    pii_types: []  # name, email, phone, address, photo\n    gdpr_applies: false  # EU residents' data\n    ccpa_applies: false  # California residents' data\n    \n  legal_risk: \"\"  # low | medium | high | do-not-scrape\n  decision: \"\"  # proceed | use-api | request-permission | abandon\n  justification: \"\""},{"language":"text","snippet":"# Common AI bot blocks in robots.txt\nUser-agent: GPTBot\nUser-agent: ChatGPT-User\nUser-agent: Google-Extended\nUser-agent: CCBot\nUser-agent: anthropic-ai\nUser-agent: ClaudeBot\nUser-agent: Bytespider\nUser-agent: PerplexityBot"},{"language":"text","snippet":"Is the content in the initial HTML source?\n├── YES → Is the site structure consistent?\n│   ├── YES → Static scraper (requests + BeautifulSoup/Cheerio)\n│   └── NO → Scrapy with custom parsers\n└── NO → Does the page require user interaction?\n    ├── YES → Playwright/Puppeteer with interaction scripts\n    └── NO → Playwright in non-interactive mode\n        └── At scale (>10K pages)? → Crawlee (hybrid mode)\n            └── Heavy anti-bot? → Managed service (Firecrawl/ScrapingBee)"},{"language":"yaml","snippet":"scraping_project:\n  name: \"\"\n  objective: \"\"  # What data, why, how often\n  \n  targets:\n    - domain: \"\"\n      pages_estimated: 0\n      rendering: \"static\" | \"javascript\" | \"spa\"\n      anti_bot: \"none\" | \"basic\" | \"cloudflare\" | \"advanced\"\n      rate_limit: \"\"  # requests per second safe limit\n      \n  tool_selected: \"\"\n  justification: \"\"\n  \n  data_schema:\n    fields: []\n    output_format: \"\"  # json | csv | database\n    \n  schedule:\n    frequency: \"\"  # once | hourly | daily | weekly\n    preferred_time: \"\"  # off-peak for target timezone\n    \n  infrastructure:\n    proxy_needed: false\n    proxy_type: \"\"  # residential | datacenter | mobile\n    storage: \"\"\n    monitoring: \"\""},{"language":"python","snippet":"# Python example — production request pattern\nimport requests\nfrom requests.adapters import HTTPAdapter\nfrom urllib3.util.retry import Retry\n\nsession = requests.Session()\n\n# Retry strategy\nretry = Retry(\n    total=3,\n    backoff_factor=1,      # 1s, 2s, 4s\n    status_forcelist=[429, 500, 502, 503, 504],\n    respect_retry_after_header=True\n)\nsession.mount(\"https://\", HTTPAdapter(max_retries=retry))\n\n# Realistic headers\nsession.headers.update({\n    \"User-Agent\": \"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36\",\n    \"Accept\": \"text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8\",\n    \"Accept-Language\": \"en-US,en;q=0.9\",\n    \"Accept-Encoding\": \"gzip, deflate, br\",\n    \"Connection\": \"keep-alive\",\n    \"Cache-Control\": \"no-cache\",\n})"},{"language":"python","snippet":"# 1. Check JSON-LD (best source — structured, clean)\nimport json\nfrom bs4 import BeautifulSoup\n\nsoup = BeautifulSoup(html, 'html.parser')\nfor script in soup.find_all('script', type='application/ld+json'):\n    data = json.loads(script.string)\n    # Often contains: Product, Article, Organization, etc.\n\n# 2. Check Open Graph meta tags\nog_title = soup.find('meta', property='og:title')\nog_price = soup.find('meta', property='product:price:amount')\n\n# 3. Check microdata\nitems = soup.find_all(itemtype=True)\n\n# 4. Fall back to CSS selectors only if above are empty"}],"parameters":{},"dependencies":[],"permissions":[],"extractedFiles":[],"languages":["typescript"],"docsSourceLabel":"CLAWHUB","editorialOverview":"Complete web scraping methodology — legal compliance, architecture design, anti-detection, data pipelines, and production operations. Use when building scrapers, extracting web data, monitoring competitors, or automating data collection at scale. --- name: Web Scraping & Data Extraction Engine description: Complete web scraping methodology — legal compliance, architecture design, anti-detection, data pipelines, and production operations. Use when building scrapers, extracting web data, monitoring competitors, or automating data collection at scale. --- Web Scraping & Data Extraction Engine Quick Health Check (Run First) Score your scraping operation (2 points","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":416,"uniquenessScore":58,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T04:56:08.762Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}