{"id":"1f006935-4f06-46c0-bf25-dc982a00fbdd","entityType":"agent","slug":"clawhub-jeminay-scrapling-fetcher","name":"Scrapling - Stealth Web Scraper","canonicalUrl":"https://www.xpersona.co/agent/clawhub-jeminay-scrapling-fetcher","canonicalPath":"/agent/clawhub-jeminay-scrapling-fetcher","generatedAt":"2026-10-09T17:23:13.885Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-09T10:38:41.911Z","emptyReason":null},"description":"Web scraping using Scrapling — a Python framework with anti-bot bypass (Cloudflare Turnstile, fingerprint spoofing), adaptive element tracking, stealth headl... Skill: Scrapling - Stealth Web Scraper Owner: jeminay Summary: Web scraping using Scrapling — a Python framework with anti-bot bypass (Cloudflare Turnstile, fingerprint spoofing), adaptive element tracking, stealth headl... Tags: latest:1.0.3 Version history: v1.0.3 | 2026-02-25T07:47:52.031Z | user Added license (MIT) and metadata.source/pypi fields to frontmatter so registry shows verified provenance instead of 'So","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 2.9K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s170dx4nw2adr8scv090bb28dd84pbe2:scrapling-fetcher","sourceUrl":"https://clawhub.ai/jeminay/scrapling-fetcher","homepage":"https://clawhub.ai/jeminay/skills/scrapling-fetcher","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/jeminay/scrapling-fetcher","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/jeminay/skills/scrapling-fetcher","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":69,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Web scraping using Scrapling — a Python framework with anti-bot bypass (Cloudflare Turnstile, fingerprint spoofing), adaptive element tracking, stealth headl..."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T10:38:41.911Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T10:38:41.911Z","emptyReason":null},"stars":null,"forks":null,"downloads":2942,"packageName":null,"latestVersion":"1.0.3","tractionLabel":"2.9K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T10:38:41.911Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T10:38:41.911Z","lastCrawledAt":"2026-10-09T10:38:41.911Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T10:38:41.911Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.3","createdAt":"2026-02-25T07:47:52.031Z","changelog":"Added license (MIT) and metadata.source/pypi fields to frontmatter so registry shows verified provenance instead of 'Source: unknown'.","fileCount":5,"zipByteSize":5882},{"version":"1.0.2","createdAt":"2026-02-25T07:44:43.172Z","changelog":"Clarified that patchright is a legit stealth-Playwright fork bundled by scrapling[all] (not a typo). Added install size note (~200 MB). No behavior changes.","fileCount":4,"zipByteSize":4393},{"version":"1.0.1","createdAt":"2026-02-25T07:41:47.037Z","changelog":"Added provenance links, install confirmation requirement, MCP server warning, auto_save note, ethical use disclaimer.","fileCount":4,"zipByteSize":4275},{"version":"1.0.0","createdAt":"2026-02-25T07:39:27.783Z","changelog":"Anti-bot web scraping with Cloudflare bypass, adaptive element tracking, CSS/XPath extraction. Supports http/stealth/dynamic modes. Includes CLI script and patterns reference.","fileCount":4,"zipByteSize":3870}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s170dx4nw2adr8scv090bb28dd84pbe2:scrapling-fetcher","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-jeminay-scrapling-fetcher/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-jeminay-scrapling-fetcher/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-jeminay-scrapling-fetcher/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-jeminay-scrapling-fetcher/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-jeminay-scrapling-fetcher/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-jeminay-scrapling-fetcher/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-09T17:23:13.884Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-jeminay-scrapling-fetcher/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-jeminay-scrapling-fetcher/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-jeminay-scrapling-fetcher/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-jeminay-scrapling-fetcher/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-09T10:38:41.911Z","emptyReason":null},"readme":"Skill: Scrapling - Stealth Web Scraper\n\nOwner: jeminay\n\nSummary: Web scraping using Scrapling — a Python framework with anti-bot bypass (Cloudflare Turnstile, fingerprint spoofing), adaptive element tracking, stealth headl...\n\nTags: latest:1.0.3\n\nVersion history:\n\nv1.0.3 | 2026-02-25T07:47:52.031Z | user\n\nAdded license (MIT) and metadata.source/pypi fields to frontmatter so registry shows verified provenance instead of 'Source: unknown'.\n\nv1.0.2 | 2026-02-25T07:44:43.172Z | user\n\nClarified that patchright is a legit stealth-Playwright fork bundled by scrapling[all] (not a typo). Added install size note (~200 MB). No behavior changes.\n\nv1.0.1 | 2026-02-25T07:41:47.037Z | user\n\nAdded provenance links, install confirmation requirement, MCP server warning, auto_save note, ethical use disclaimer.\n\nv1.0.0 | 2026-02-25T07:39:27.783Z | user\n\nAnti-bot web scraping with Cloudflare bypass, adaptive element tracking, CSS/XPath extraction. Supports http/stealth/dynamic modes. Includes CLI script and patterns reference.\n\nArchive index:\n\nArchive v1.0.3: 5 files, 5882 bytes\n\nFiles: references/patterns.md (2932b), scripts/scrape.py (2724b), skill-card.md (2631b), SKILL.md (3302b), _meta.json (136b)\n\nFile v1.0.3:SKILL.md\n\n---\nname: scrapling\ndescription: \"Web scraping using Scrapling — a Python framework with anti-bot bypass (Cloudflare Turnstile, fingerprint spoofing), adaptive element tracking, stealth headless browser, and full CSS/XPath extraction. Use when web_fetch fails (Cloudflare, JS-rendered pages), or when extracting structured data from websites (prices, articles, lists). Supports HTTP, stealth, and full browser modes. Source: github.com/D4Vinci/Scrapling (PyPI: scrapling). Only use on sites you have permission to scrape.\"\nlicense: MIT\nmetadata:\n  source: https://github.com/D4Vinci/Scrapling\n  pypi: https://pypi.org/project/scrapling/\n---\n\n# Scrapling Skill\n\n**Source:** https://github.com/D4Vinci/Scrapling (open source, MIT-like license)\n**PyPI:** `scrapling` — install before first use (see below)\n\n> ⚠️ Only scrape sites you have permission to access. Respect `robots.txt` and Terms of Service. Do not use stealth modes to bypass paywalls or access restricted content without authorization.\n\n## Installation (one-time, confirm with user before running)\n\n```bash\npip install scrapling[all]\npatchright install chromium  # required for stealth/dynamic modes\n```\n\n- `scrapling[all]` installs `patchright` (a stealth fork of Playwright, bundled as a PyPI package — not a typo), `curl_cffi`, MCP server deps, and IPython shell.\n- `patchright install chromium` downloads Chromium (~100 MB) via patchright's own installer (same mechanism as `playwright install chromium`).\n- Confirm with user before running — installs ~200 MB of dependencies and browser binaries.\n\n## Script\n\n`scripts/scrape.py` — CLI wrapper for all three fetcher modes.\n\n```bash\n# Basic fetch (text output)\npython3 ~/skills/scrapling/scripts/scrape.py <url> -q\n\n# CSS selector extraction\npython3 ~/skills/scrapling/scripts/scrape.py <url> --selector \".class\" -q\n\n# Stealth mode (Cloudflare bypass) — only on sites you're authorized to access\npython3 ~/skills/scrapling/scripts/scrape.py <url> --mode stealth -q\n\n# JSON output\npython3 ~/skills/scrapling/scripts/scrape.py <url> --selector \"h2\" --json -q\n```\n\n## Fetcher Modes\n\n- **http** (default) — Fast HTTP with browser TLS fingerprint spoofing. Most sites.\n- **stealth** — Headless Chrome with anti-detect. For Cloudflare/anti-bot.\n- **dynamic** — Full Playwright browser. For heavy JS SPAs.\n\n## When to Use Each Mode\n\n- `web_fetch` returns 403/429/Cloudflare challenge → use `--mode stealth`\n- Page content requires JS execution → use `--mode dynamic`\n- Regular site, just need text/data → use `--mode http` (default)\n\n## Python Inline Usage\n\nFor custom logic beyond the CLI, write inline Python. See `references/patterns.md` for:\n- Adaptive scraping (`auto_save` / `adaptive` — saves element fingerprints locally)\n- Session/cookie handling\n- Async usage\n- XPath, find_similar, attribute extraction\n\n## Notes\n\n- **MCP server** (`scrapling mcp`): starts a local network service for AI-native scraping. Only start if explicitly needed and trusted — it exposes a local HTTP server.\n- **`auto_save=True`**: persists element fingerprints to disk for adaptive re-scraping. Creates local state in working directory.\n- Stealth/dynamic modes use Chromium headless — no `xvfb-run` needed.\n- For large-scale crawls, use the Spider API (see Scrapling docs).\n\nFile v1.0.3:_meta.json\n\n{\n  \"ownerId\": \"kn73589k2rv646tm107cp6d90s81g7xy\",\n  \"slug\": \"scrapling-fetcher\",\n  \"version\": \"1.0.3\",\n  \"publishedAt\": 1772005672031\n}\n\nFile v1.0.3:references/patterns.md\n\n# Scrapling Patterns Reference\n\n## Fetcher Selection Guide\n\n| Scenario | Fetcher | Notes |\n|---|---|---|\n| Regular sites, APIs | `Fetcher` | Fastest, HTTP-only |\n| Cloudflare, anti-bot | `StealthyFetcher` | Headless Chrome, fingerprint spoofing |\n| Heavy JS rendering | `DynamicFetcher` | Full Playwright browser |\n| Async pipeline | `AsyncFetcher` | Async equivalent of Fetcher |\n\n## Python Quick Patterns\n\n### Basic HTTP fetch\n```python\nfrom scrapling.fetchers import Fetcher\npage = Fetcher.get('https://example.com')\nprint(page.status)  # 200\ntext = page.get_all_text(ignore_tags=('script', 'style'))\n```\n\n### Stealth fetch (bypass Cloudflare)\n```python\nfrom scrapling.fetchers import StealthyFetcher\npage = StealthyFetcher.fetch('https://protected-site.com', headless=True, network_idle=True)\n```\n\n### Dynamic fetch (JS-rendered content)\n```python\nfrom scrapling.fetchers import DynamicFetcher\npage = DynamicFetcher.fetch('https://spa-site.com', headless=True, network_idle=True)\n```\n\n### CSS selector extraction\n```python\ntitles = page.css('h2.title')\nfor t in titles:\n    print(t.text)\n\n# Get attribute\nlinks = page.css('a.product-link')\nfor a in links:\n    print(a.attrib['href'])\n```\n\n### XPath extraction\n```python\nitems = page.xpath('//div[@class=\"item\"]/span/text()')\n```\n\n### Adaptive scraping (survives site redesigns)\n```python\n# First run: auto_save=True saves element fingerprints\nproducts = page.css('.product-card', auto_save=True)\n# Later runs: adaptive=True finds them even if CSS changed\nproducts = page.css('.product-card', adaptive=True)\n```\n\n### Find similar elements\n```python\nfirst = page.css('.price')[0]\nall_prices = first.find_similar()\n```\n\n### Session with cookies\n```python\nfrom scrapling.fetchers import FetcherSession\nsession = FetcherSession()\nsession.get('https://example.com/login', data={'user': 'x', 'pass': 'y'})\npage = session.get('https://example.com/dashboard')\n```\n\n### Async usage\n```python\nimport asyncio\nfrom scrapling.fetchers import AsyncFetcher\n\nasync def scrape():\n    page = await AsyncFetcher.get('https://example.com')\n    return page.css('h1')[0].text\n\nasyncio.run(scrape())\n```\n\n## CLI Usage\n\n```bash\n# Simple text extraction\npython3 scrape.py https://example.com\n\n# CSS selector extraction\npython3 scrape.py https://example.com --selector \"h2.title\"\n\n# Extract attribute value\npython3 scrape.py https://example.com --selector \"a.product\" --attr href\n\n# Stealth mode for protected sites\npython3 scrape.py https://cloudflare-site.com --mode stealth\n\n# JSON output\npython3 scrape.py https://example.com --selector \".price\" --json\n\n# Quiet mode (no INFO logs)\npython3 scrape.py https://example.com -q\n```\n\n## MCP Server Setup\n\n> ⚠️ The MCP server starts a local HTTP service. Only use in trusted environments.\n\n```bash\nscrapling mcp\n# or\npython3 -m scrapling.mcp\n```\n\nAdd to OpenClaw MCP config (mcporter) to get scraping as a native tool. Confirm with user before starting.\n\nFile v1.0.3:skill-card.md\n\n## Description:\n\nProvides a Scrapling-based web scraping workflow for fetching pages, extracting structured data with CSS or XPath selectors, and using HTTP, stealth, or dynamic browser modes on sites the user is authorized to scrape.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[jeminay](https://clawhub.ai/user/jeminay)\n\n### License/Terms of Use:\n\nMIT\n\n## Use Case:\n\nDevelopers and automation engineers use this skill to retrieve web page text or structured fields when ordinary fetch tools fail because of JavaScript rendering, rate limits, or anti-bot challenges. It is intended for authorized scraping workflows and selector-based extraction, not for bypassing restricted access.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill can send requests through HTTP, stealth, and browser-based modes to external sites.\n\nMitigation: Use it only for sites the user is authorized to scrape, respect applicable terms and robots.txt, and avoid passing credentials or sensitive URLs unless the destination is trusted.\n\nRisk: Installing full Scrapling dependencies and browser binaries expands the local runtime surface.\n\nMitigation: Install and run the skill in a virtual environment or disposable container without administrator privileges.\n\nRisk: The optional MCP server starts a local HTTP service.\n\nMitigation: Start the server only when explicitly needed and only in a trusted local environment.\n\nRisk: Adaptive scraping can persist element fingerprints or local state in the working directory.\n\nMitigation: Run adaptive workflows in a controlled directory and remove stored state when it is no longer needed.\n\n## Reference(s):\n\n- [Scrapling Patterns Reference](references/patterns.md)\n- [Scrapling on PyPI](https://pypi.org/project/scrapling/)\n- [ClawHub Skill Page](https://clawhub.ai/jeminay/skills/scrapling-fetcher)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, Code, Shell commands, Configuration, Guidance]\n\n**Output Format:** [Markdown guidance with inline shell commands, Python examples, plain text extraction, and optional JSON arrays]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Can emit selector matches as text lines or JSON; stealth and dynamic modes may use headless Chromium.]\n\n## Skill Version(s):\n\n1.0.3 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v1.0.2: 4 files, 4393 bytes\n\nFiles: references/patterns.md (2932b), scripts/scrape.py (2724b), SKILL.md (3188b), _meta.json (136b)\n\nFile v1.0.2:SKILL.md\n\n---\nname: scrapling\ndescription: \"Web scraping using Scrapling — a Python framework with anti-bot bypass (Cloudflare Turnstile, fingerprint spoofing), adaptive element tracking, stealth headless browser, and full CSS/XPath extraction. Use when web_fetch fails (Cloudflare, JS-rendered pages), or when extracting structured data from websites (prices, articles, lists). Supports HTTP, stealth, and full browser modes. Source: github.com/D4Vinci/Scrapling (PyPI: scrapling). Only use on sites you have permission to scrape.\"\n---\n\n# Scrapling Skill\n\n**Source:** https://github.com/D4Vinci/Scrapling (open source, MIT-like license)\n**PyPI:** `scrapling` — install before first use (see below)\n\n> ⚠️ Only scrape sites you have permission to access. Respect `robots.txt` and Terms of Service. Do not use stealth modes to bypass paywalls or access restricted content without authorization.\n\n## Installation (one-time, confirm with user before running)\n\n```bash\npip install scrapling[all]\npatchright install chromium  # required for stealth/dynamic modes\n```\n\n- `scrapling[all]` installs `patchright` (a stealth fork of Playwright, bundled as a PyPI package — not a typo), `curl_cffi`, MCP server deps, and IPython shell.\n- `patchright install chromium` downloads Chromium (~100 MB) via patchright's own installer (same mechanism as `playwright install chromium`).\n- Confirm with user before running — installs ~200 MB of dependencies and browser binaries.\n\n## Script\n\n`scripts/scrape.py` — CLI wrapper for all three fetcher modes.\n\n```bash\n# Basic fetch (text output)\npython3 ~/skills/scrapling/scripts/scrape.py <url> -q\n\n# CSS selector extraction\npython3 ~/skills/scrapling/scripts/scrape.py <url> --selector \".class\" -q\n\n# Stealth mode (Cloudflare bypass) — only on sites you're authorized to access\npython3 ~/skills/scrapling/scripts/scrape.py <url> --mode stealth -q\n\n# JSON output\npython3 ~/skills/scrapling/scripts/scrape.py <url> --selector \"h2\" --json -q\n```\n\n## Fetcher Modes\n\n- **http** (default) — Fast HTTP with browser TLS fingerprint spoofing. Most sites.\n- **stealth** — Headless Chrome with anti-detect. For Cloudflare/anti-bot.\n- **dynamic** — Full Playwright browser. For heavy JS SPAs.\n\n## When to Use Each Mode\n\n- `web_fetch` returns 403/429/Cloudflare challenge → use `--mode stealth`\n- Page content requires JS execution → use `--mode dynamic`\n- Regular site, just need text/data → use `--mode http` (default)\n\n## Python Inline Usage\n\nFor custom logic beyond the CLI, write inline Python. See `references/patterns.md` for:\n- Adaptive scraping (`auto_save` / `adaptive` — saves element fingerprints locally)\n- Session/cookie handling\n- Async usage\n- XPath, find_similar, attribute extraction\n\n## Notes\n\n- **MCP server** (`scrapling mcp`): starts a local network service for AI-native scraping. Only start if explicitly needed and trusted — it exposes a local HTTP server.\n- **`auto_save=True`**: persists element fingerprints to disk for adaptive re-scraping. Creates local state in working directory.\n- Stealth/dynamic modes use Chromium headless — no `xvfb-run` needed.\n- For large-scale crawls, use the Spider API (see Scrapling docs).\n\nFile v1.0.2:_meta.json\n\n{\n  \"ownerId\": \"kn73589k2rv646tm107cp6d90s81g7xy\",\n  \"slug\": \"scrapling-fetcher\",\n  \"version\": \"1.0.2\",\n  \"publishedAt\": 1772005483172\n}\n\nFile v1.0.2:references/patterns.md\n\n# Scrapling Patterns Reference\n\n## Fetcher Selection Guide\n\n| Scenario | Fetcher | Notes |\n|---|---|---|\n| Regular sites, APIs | `Fetcher` | Fastest, HTTP-only |\n| Cloudflare, anti-bot | `StealthyFetcher` | Headless Chrome, fingerprint spoofing |\n| Heavy JS rendering | `DynamicFetcher` | Full Playwright browser |\n| Async pipeline | `AsyncFetcher` | Async equivalent of Fetcher |\n\n## Python Quick Patterns\n\n### Basic HTTP fetch\n```python\nfrom scrapling.fetchers import Fetcher\npage = Fetcher.get('https://example.com')\nprint(page.status)  # 200\ntext = page.get_all_text(ignore_tags=('script', 'style'))\n```\n\n### Stealth fetch (bypass Cloudflare)\n```python\nfrom scrapling.fetchers import StealthyFetcher\npage = StealthyFetcher.fetch('https://protected-site.com', headless=True, network_idle=True)\n```\n\n### Dynamic fetch (JS-rendered content)\n```python\nfrom scrapling.fetchers import DynamicFetcher\npage = DynamicFetcher.fetch('https://spa-site.com', headless=True, network_idle=True)\n```\n\n### CSS selector extraction\n```python\ntitles = page.css('h2.title')\nfor t in titles:\n    print(t.text)\n\n# Get attribute\nlinks = page.css('a.product-link')\nfor a in links:\n    print(a.attrib['href'])\n```\n\n### XPath extraction\n```python\nitems = page.xpath('//div[@class=\"item\"]/span/text()')\n```\n\n### Adaptive scraping (survives site redesigns)\n```python\n# First run: auto_save=True saves element fingerprints\nproducts = page.css('.product-card', auto_save=True)\n# Later runs: adaptive=True finds them even if CSS changed\nproducts = page.css('.product-card', adaptive=True)\n```\n\n### Find similar elements\n```python\nfirst = page.css('.price')[0]\nall_prices = first.find_similar()\n```\n\n### Session with cookies\n```python\nfrom scrapling.fetchers import FetcherSession\nsession = FetcherSession()\nsession.get('https://example.com/login', data={'user': 'x', 'pass': 'y'})\npage = session.get('https://example.com/dashboard')\n```\n\n### Async usage\n```python\nimport asyncio\nfrom scrapling.fetchers import AsyncFetcher\n\nasync def scrape():\n    page = await AsyncFetcher.get('https://example.com')\n    return page.css('h1')[0].text\n\nasyncio.run(scrape())\n```\n\n## CLI Usage\n\n```bash\n# Simple text extraction\npython3 scrape.py https://example.com\n\n# CSS selector extraction\npython3 scrape.py https://example.com --selector \"h2.title\"\n\n# Extract attribute value\npython3 scrape.py https://example.com --selector \"a.product\" --attr href\n\n# Stealth mode for protected sites\npython3 scrape.py https://cloudflare-site.com --mode stealth\n\n# JSON output\npython3 scrape.py https://example.com --selector \".price\" --json\n\n# Quiet mode (no INFO logs)\npython3 scrape.py https://example.com -q\n```\n\n## MCP Server Setup\n\n> ⚠️ The MCP server starts a local HTTP service. Only use in trusted environments.\n\n```bash\nscrapling mcp\n# or\npython3 -m scrapling.mcp\n```\n\nAdd to OpenClaw MCP config (mcporter) to get scraping as a native tool. Confirm with user before starting.\n\nArchive v1.0.1: 4 files, 4275 bytes\n\nFiles: references/patterns.md (2932b), scripts/scrape.py (2724b), SKILL.md (2904b), _meta.json (136b)\n\nFile v1.0.1:SKILL.md\n\n---\nname: scrapling\ndescription: \"Web scraping using Scrapling — a Python framework with anti-bot bypass (Cloudflare Turnstile, fingerprint spoofing), adaptive element tracking, stealth headless browser, and full CSS/XPath extraction. Use when web_fetch fails (Cloudflare, JS-rendered pages), or when extracting structured data from websites (prices, articles, lists). Supports HTTP, stealth, and full browser modes. Source: github.com/D4Vinci/Scrapling (PyPI: scrapling). Only use on sites you have permission to scrape.\"\n---\n\n# Scrapling Skill\n\n**Source:** https://github.com/D4Vinci/Scrapling (open source, MIT-like license)\n**PyPI:** `scrapling` — install before first use (see below)\n\n> ⚠️ Only scrape sites you have permission to access. Respect `robots.txt` and Terms of Service. Do not use stealth modes to bypass paywalls or access restricted content without authorization.\n\n## Installation (one-time, confirm with user before running)\n\n```bash\npip install scrapling[all]\npatchright install chromium  # required for stealth/dynamic modes\n```\n\nInstallation pulls Playwright, Chromium (~100 MB), and curl_cffi. Confirm with user before running on their machine.\n\n## Script\n\n`scripts/scrape.py` — CLI wrapper for all three fetcher modes.\n\n```bash\n# Basic fetch (text output)\npython3 ~/skills/scrapling/scripts/scrape.py <url> -q\n\n# CSS selector extraction\npython3 ~/skills/scrapling/scripts/scrape.py <url> --selector \".class\" -q\n\n# Stealth mode (Cloudflare bypass) — only on sites you're authorized to access\npython3 ~/skills/scrapling/scripts/scrape.py <url> --mode stealth -q\n\n# JSON output\npython3 ~/skills/scrapling/scripts/scrape.py <url> --selector \"h2\" --json -q\n```\n\n## Fetcher Modes\n\n- **http** (default) — Fast HTTP with browser TLS fingerprint spoofing. Most sites.\n- **stealth** — Headless Chrome with anti-detect. For Cloudflare/anti-bot.\n- **dynamic** — Full Playwright browser. For heavy JS SPAs.\n\n## When to Use Each Mode\n\n- `web_fetch` returns 403/429/Cloudflare challenge → use `--mode stealth`\n- Page content requires JS execution → use `--mode dynamic`\n- Regular site, just need text/data → use `--mode http` (default)\n\n## Python Inline Usage\n\nFor custom logic beyond the CLI, write inline Python. See `references/patterns.md` for:\n- Adaptive scraping (`auto_save` / `adaptive` — saves element fingerprints locally)\n- Session/cookie handling\n- Async usage\n- XPath, find_similar, attribute extraction\n\n## Notes\n\n- **MCP server** (`scrapling mcp`): starts a local network service for AI-native scraping. Only start if explicitly needed and trusted — it exposes a local HTTP server.\n- **`auto_save=True`**: persists element fingerprints to disk for adaptive re-scraping. Creates local state in working directory.\n- Stealth/dynamic modes use Chromium headless — no `xvfb-run` needed.\n- For large-scale crawls, use the Spider API (see Scrapling docs).\n\nFile v1.0.1:_meta.json\n\n{\n  \"ownerId\": \"kn73589k2rv646tm107cp6d90s81g7xy\",\n  \"slug\": \"scrapling-fetcher\",\n  \"version\": \"1.0.1\",\n  \"publishedAt\": 1772005307037\n}\n\nFile v1.0.1:references/patterns.md\n\n# Scrapling Patterns Reference\n\n## Fetcher Selection Guide\n\n| Scenario | Fetcher | Notes |\n|---|---|---|\n| Regular sites, APIs | `Fetcher` | Fastest, HTTP-only |\n| Cloudflare, anti-bot | `StealthyFetcher` | Headless Chrome, fingerprint spoofing |\n| Heavy JS rendering | `DynamicFetcher` | Full Playwright browser |\n| Async pipeline | `AsyncFetcher` | Async equivalent of Fetcher |\n\n## Python Quick Patterns\n\n### Basic HTTP fetch\n```python\nfrom scrapling.fetchers import Fetcher\npage = Fetcher.get('https://example.com')\nprint(page.status)  # 200\ntext = page.get_all_text(ignore_tags=('script', 'style'))\n```\n\n### Stealth fetch (bypass Cloudflare)\n```python\nfrom scrapling.fetchers import StealthyFetcher\npage = StealthyFetcher.fetch('https://protected-site.com', headless=True, network_idle=True)\n```\n\n### Dynamic fetch (JS-rendered content)\n```python\nfrom scrapling.fetchers import DynamicFetcher\npage = DynamicFetcher.fetch('https://spa-site.com', headless=True, network_idle=True)\n```\n\n### CSS selector extraction\n```python\ntitles = page.css('h2.title')\nfor t in titles:\n    print(t.text)\n\n# Get attribute\nlinks = page.css('a.product-link')\nfor a in links:\n    print(a.attrib['href'])\n```\n\n### XPath extraction\n```python\nitems = page.xpath('//div[@class=\"item\"]/span/text()')\n```\n\n### Adaptive scraping (survives site redesigns)\n```python\n# First run: auto_save=True saves element fingerprints\nproducts = page.css('.product-card', auto_save=True)\n# Later runs: adaptive=True finds them even if CSS changed\nproducts = page.css('.product-card', adaptive=True)\n```\n\n### Find similar elements\n```python\nfirst = page.css('.price')[0]\nall_prices = first.find_similar()\n```\n\n### Session with cookies\n```python\nfrom scrapling.fetchers import FetcherSession\nsession = FetcherSession()\nsession.get('https://example.com/login', data={'user': 'x', 'pass': 'y'})\npage = session.get('https://example.com/dashboard')\n```\n\n### Async usage\n```python\nimport asyncio\nfrom scrapling.fetchers import AsyncFetcher\n\nasync def scrape():\n    page = await AsyncFetcher.get('https://example.com')\n    return page.css('h1')[0].text\n\nasyncio.run(scrape())\n```\n\n## CLI Usage\n\n```bash\n# Simple text extraction\npython3 scrape.py https://example.com\n\n# CSS selector extraction\npython3 scrape.py https://example.com --selector \"h2.title\"\n\n# Extract attribute value\npython3 scrape.py https://example.com --selector \"a.product\" --attr href\n\n# Stealth mode for protected sites\npython3 scrape.py https://cloudflare-site.com --mode stealth\n\n# JSON output\npython3 scrape.py https://example.com --selector \".price\" --json\n\n# Quiet mode (no INFO logs)\npython3 scrape.py https://example.com -q\n```\n\n## MCP Server Setup\n\n> ⚠️ The MCP server starts a local HTTP service. Only use in trusted environments.\n\n```bash\nscrapling mcp\n# or\npython3 -m scrapling.mcp\n```\n\nAdd to OpenClaw MCP config (mcporter) to get scraping as a native tool. Confirm with user before starting.\n\nArchive v1.0.0: 4 files, 3870 bytes\n\nFiles: references/patterns.md (2863b), scripts/scrape.py (2724b), SKILL.md (2161b), _meta.json (136b)\n\nFile v1.0.0:SKILL.md\n\n---\nname: scrapling\ndescription: Web scraping using Scrapling — a Python framework with anti-bot bypass (Cloudflare Turnstile, fingerprint spoofing), adaptive element tracking, stealth headless browser, and full CSS/XPath extraction. Use when web_fetch fails (Cloudflare, JS-rendered pages, paywalls), when extracting structured data from websites (prices, articles, lists), or when building scrapers that survive site redesigns. Supports HTTP, stealth, and full browser modes.\n---\n\n# Scrapling Skill\n\nScrapling is installed at system level (`pip install scrapling[all]`). Chromium is available for headless browser modes.\n\n## Script\n\n`scripts/scrape.py` — CLI wrapper for all three fetcher modes.\n\n```bash\n# Basic fetch (text output)\npython3 ~/skills/scrapling/scripts/scrape.py <url>\n\n# CSS selector\npython3 ~/skills/scrapling/scripts/scrape.py <url> --selector \".class\" \n\n# Stealth mode (Cloudflare bypass)\npython3 ~/skills/scrapling/scripts/scrape.py <url> --mode stealth\n\n# JSON output\npython3 ~/skills/scrapling/scripts/scrape.py <url> --selector \"h2\" --json\n\n# Quiet (suppress INFO logs)\npython3 ~/skills/scrapling/scripts/scrape.py <url> -q\n```\n\n## Fetcher Modes\n\n- **http** (default) — Fast HTTP with browser TLS fingerprint spoofing. Most sites.\n- **stealth** — Headless Chrome with anti-detect. For Cloudflare/anti-bot.\n- **dynamic** — Full Playwright browser. For heavy JS SPAs.\n\n## When to Use Each Mode\n\n- `web_fetch` returns 403/429/Cloudflare challenge → use `--mode stealth`\n- Page content requires JS execution → use `--mode dynamic`\n- Regular site, just need text/data → use `--mode http` (default)\n\n## Python Inline Usage\n\nFor custom logic beyond what the CLI covers, write inline Python. See `references/patterns.md` for:\n- Adaptive scraping (survives redesigns)\n- Session/cookie handling\n- Async usage\n- XPath, find_similar, attribute extraction\n- MCP server setup\n\n## Notes\n\n- Stealth/dynamic modes require display; they use Chromium headless (no `xvfb-run` needed).\n- `auto_save=True` on first scrape saves element fingerprints for future adaptive runs.\n- For large-scale crawls, use the Spider API (see Scrapling docs).\n\nFile v1.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn73589k2rv646tm107cp6d90s81g7xy\",\n  \"slug\": \"scrapling-fetcher\",\n  \"version\": \"1.0.0\",\n  \"publishedAt\": 1772005167783\n}\n\nFile v1.0.0:references/patterns.md\n\n# Scrapling Patterns Reference\n\n## Fetcher Selection Guide\n\n| Scenario | Fetcher | Notes |\n|---|---|---|\n| Regular sites, APIs | `Fetcher` | Fastest, HTTP-only |\n| Cloudflare, anti-bot | `StealthyFetcher` | Headless Chrome, fingerprint spoofing |\n| Heavy JS rendering | `DynamicFetcher` | Full Playwright browser |\n| Async pipeline | `AsyncFetcher` | Async equivalent of Fetcher |\n\n## Python Quick Patterns\n\n### Basic HTTP fetch\n```python\nfrom scrapling.fetchers import Fetcher\npage = Fetcher.get('https://example.com')\nprint(page.status)  # 200\ntext = page.get_all_text(ignore_tags=('script', 'style'))\n```\n\n### Stealth fetch (bypass Cloudflare)\n```python\nfrom scrapling.fetchers import StealthyFetcher\npage = StealthyFetcher.fetch('https://protected-site.com', headless=True, network_idle=True)\n```\n\n### Dynamic fetch (JS-rendered content)\n```python\nfrom scrapling.fetchers import DynamicFetcher\npage = DynamicFetcher.fetch('https://spa-site.com', headless=True, network_idle=True)\n```\n\n### CSS selector extraction\n```python\ntitles = page.css('h2.title')\nfor t in titles:\n    print(t.text)\n\n# Get attribute\nlinks = page.css('a.product-link')\nfor a in links:\n    print(a.attrib['href'])\n```\n\n### XPath extraction\n```python\nitems = page.xpath('//div[@class=\"item\"]/span/text()')\n```\n\n### Adaptive scraping (survives site redesigns)\n```python\n# First run: auto_save=True saves element fingerprints\nproducts = page.css('.product-card', auto_save=True)\n# Later runs: adaptive=True finds them even if CSS changed\nproducts = page.css('.product-card', adaptive=True)\n```\n\n### Find similar elements\n```python\nfirst = page.css('.price')[0]\nall_prices = first.find_similar()\n```\n\n### Session with cookies\n```python\nfrom scrapling.fetchers import FetcherSession\nsession = FetcherSession()\nsession.get('https://example.com/login', data={'user': 'x', 'pass': 'y'})\npage = session.get('https://example.com/dashboard')\n```\n\n### Async usage\n```python\nimport asyncio\nfrom scrapling.fetchers import AsyncFetcher\n\nasync def scrape():\n    page = await AsyncFetcher.get('https://example.com')\n    return page.css('h1')[0].text\n\nasyncio.run(scrape())\n```\n\n## CLI Usage\n\n```bash\n# Simple text extraction\npython3 scrape.py https://example.com\n\n# CSS selector extraction\npython3 scrape.py https://example.com --selector \"h2.title\"\n\n# Extract attribute value\npython3 scrape.py https://example.com --selector \"a.product\" --attr href\n\n# Stealth mode for protected sites\npython3 scrape.py https://cloudflare-site.com --mode stealth\n\n# JSON output\npython3 scrape.py https://example.com --selector \".price\" --json\n\n# Quiet mode (no INFO logs)\npython3 scrape.py https://example.com -q\n```\n\n## MCP Server Setup\n\nStart the built-in MCP server for AI-native scraping:\n```bash\nscrapling mcp\n# or\npython3 -m scrapling.mcp\n```\n\nAdd to OpenClaw MCP config (mcporter) to get scraping as a native tool.","readmeExcerpt":"Skill: Scrapling - Stealth Web Scraper Owner: jeminay Summary: Web scraping using Scrapling — a Python framework with anti-bot bypass (Cloudflare Turnstile, fingerprint spoofing), adaptive element tracking, stealth headl... Tags: latest:1.0.3 Version history: v1.0.3 | 2026-02-25T07:47:52.031Z | user Added license (MIT) and metadata.source/pypi fields to frontmatter so registry shows verified provenance instead of 'So","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"pip install scrapling[all]\npatchright install chromium  # required for stealth/dynamic modes"},{"language":"bash","snippet":"# Basic fetch (text output)\npython3 ~/skills/scrapling/scripts/scrape.py <url> -q\n\n# CSS selector extraction\npython3 ~/skills/scrapling/scripts/scrape.py <url> --selector \".class\" -q\n\n# Stealth mode (Cloudflare bypass) — only on sites you're authorized to access\npython3 ~/skills/scrapling/scripts/scrape.py <url> --mode stealth -q\n\n# JSON output\npython3 ~/skills/scrapling/scripts/scrape.py <url> --selector \"h2\" --json -q"},{"language":"python","snippet":"from scrapling.fetchers import Fetcher\npage = Fetcher.get('https://example.com')\nprint(page.status)  # 200\ntext = page.get_all_text(ignore_tags=('script', 'style'))"},{"language":"python","snippet":"from scrapling.fetchers import StealthyFetcher\npage = StealthyFetcher.fetch('https://protected-site.com', headless=True, network_idle=True)"},{"language":"python","snippet":"from scrapling.fetchers import DynamicFetcher\npage = DynamicFetcher.fetch('https://spa-site.com', headless=True, network_idle=True)"},{"language":"python","snippet":"titles = page.css('h2.title')\nfor t in titles:\n    print(t.text)\n\n# Get attribute\nlinks = page.css('a.product-link')\nfor a in links:\n    print(a.attrib['href'])"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: scrapling\ndescription: \"Web scraping using Scrapling — a Python framework with anti-bot bypass (Cloudflare Turnstile, fingerprint spoofing), adaptive element tracking, stealth headless browser, and full CSS/XPath extraction. Use when web_fetch fails (Cloudflare, JS-rendered pages), or when extracting structured data from websites (prices, articles, lists). Supports HTTP, stealth, and full browser modes. Source: github.com/D4Vinci/Scrapling (PyPI: scrapling). Only use on sites you have permission to scrape.\"\nlicense: MIT\nmetadata:\n  source: https://github.com/D4Vinci/Scrapling\n  pypi: https://pypi.org/project/scrapling/\n---\n\n# Scrapling Skill\n\n**Source:** https://github.com/D4Vinci/Scrapling (open source, MIT-like license)\n**PyPI:** `scrapling` — install before first use (see below)\n\n> ⚠️ Only scrape sites you have permission to access. Respect `robots.txt` and Terms of Service. Do not use stealth modes to bypass paywalls or access restricted content without authorization.\n\n## Installation (one-time, confirm with user before running)\n\n```bash\npip install scrapling[all]\npatchright install chromium  # required for stealth/dynamic modes\n```\n\n- `scrapling[all]` installs `patchright` (a stealth fork of Playwright, bundled as a PyPI package — not a typo), `curl_cffi`, MCP server deps, and IPython shell.\n- `patchright install chromium` downloads Chromium (~100 MB) via patchright's own installer (same mechanism as `playwright install chromium`).\n- Confirm with user before running — installs ~200 MB of dependencies and browser binaries.\n\n## Script\n\n`scripts/scrape.py` — CLI wrapper for all three fetcher modes.\n\n```bash\n# Basic fetch (text output)\npython3 ~/skills/scrapling/scripts/scrape.py <url> -q\n\n# CSS selector extraction\npython3 ~/skills/scrapling/scripts/scrape.py <url> --selector \".class\" -q\n\n# Stealth mode (Cloudflare bypass) — only on sites you're authorized to access\npython3 ~/skills/scrapling/scripts/scrape.py <url> --mode stealth -q\n\n# JSON output\npython3 ~/skills/scrapling/scripts/scrape.py <url> --selector \"h2\" --json -q\n```\n\n## Fetcher Modes\n\n- **http** (default) — Fast HTTP with browser TLS fingerprint spoofing. Most sites.\n- **stealth** — Headless Chrome with anti-detect. For Cloudflare/anti-bot.\n- **dynamic** — Full Playwright browser. For heavy JS SPAs.\n\n## When to Use Each Mode\n\n- `web_fetch` returns 403/429/Cloudflare challenge → use `--mode stealth`\n- Page content requires JS execution → use `--mode dynamic`\n- Regular site, just need text/data → use `--mode http` (default)\n\n## Python Inline Usage\n\nFor custom logic beyond the CLI, write inline Python. See `references/patterns.md` for:\n- Adaptive scraping (`auto_save` / `adaptive` — saves element fingerprints locally)\n- Session/cookie handling\n- Async usage\n- XPath, find_similar, attribute extraction\n\n## Notes\n\n- **MCP server** (`scrapling mcp`): starts a local network service for AI-native scraping. Only start if explicitly needed and trusted — it exposes a local HTTP server."},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn73589k2rv646tm107cp6d90s81g7xy\",\n  \"slug\": \"scrapling-fetcher\",\n  \"version\": \"1.0.3\",\n  \"publishedAt\": 1772005672031\n}"},{"path":"references/patterns.md","content":"# Scrapling Patterns Reference\n\n## Fetcher Selection Guide\n\n| Scenario | Fetcher | Notes |\n|---|---|---|\n| Regular sites, APIs | `Fetcher` | Fastest, HTTP-only |\n| Cloudflare, anti-bot | `StealthyFetcher` | Headless Chrome, fingerprint spoofing |\n| Heavy JS rendering | `DynamicFetcher` | Full Playwright browser |\n| Async pipeline | `AsyncFetcher` | Async equivalent of Fetcher |\n\n## Python Quick Patterns\n\n### Basic HTTP fetch\n```python\nfrom scrapling.fetchers import Fetcher\npage = Fetcher.get('https://example.com')\nprint(page.status)  # 200\ntext = page.get_all_text(ignore_tags=('script', 'style'))\n```\n\n### Stealth fetch (bypass Cloudflare)\n```python\nfrom scrapling.fetchers import StealthyFetcher\npage = StealthyFetcher.fetch('https://protected-site.com', headless=True, network_idle=True)\n```\n\n### Dynamic fetch (JS-rendered content)\n```python\nfrom scrapling.fetchers import DynamicFetcher\npage = DynamicFetcher.fetch('https://spa-site.com', headless=True, network_idle=True)\n```\n\n### CSS selector extraction\n```python\ntitles = page.css('h2.title')\nfor t in titles:\n    print(t.text)\n\n# Get attribute\nlinks = page.css('a.product-link')\nfor a in links:\n    print(a.attrib['href'])\n```\n\n### XPath extraction\n```python\nitems = page.xpath('//div[@class=\"item\"]/span/text()')\n```\n\n### Adaptive scraping (survives site redesigns)\n```python\n# First run: auto_save=True saves element fingerprints\nproducts = page.css('.product-card', auto_save=True)\n# Later runs: adaptive=True finds them even if CSS changed\nproducts = page.css('.product-card', adaptive=True)\n```\n\n### Find similar elements\n```python\nfirst = page.css('.price')[0]\nall_prices = first.find_similar()\n```\n\n### Session with cookies\n```python\nfrom scrapling.fetchers import FetcherSession\nsession = FetcherSession()\nsession.get('https://example.com/login', data={'user': 'x', 'pass': 'y'})\npage = session.get('https://example.com/dashboard')\n```\n\n### Async usage\n```python\nimport asyncio\nfrom scrapling.fetchers import AsyncFetcher\n\nasync def scrape():\n    page = await AsyncFetcher.get('https://example.com')\n    return page.css('h1')[0].text\n\nasyncio.run(scrape())\n```\n\n## CLI Usage\n\n```bash\n# Simple text extraction\npython3 scrape.py https://example.com\n\n# CSS selector extraction\npython3 scrape.py https://example.com --selector \"h2.title\"\n\n# Extract attribute value\npython3 scrape.py https://example.com --selector \"a.product\" --attr href\n\n# Stealth mode for protected sites\npython3 scrape.py https://cloudflare-site.com --mode stealth\n\n# JSON output\npython3 scrape.py https://example.com --selector \".price\" --json\n\n# Quiet mode (no INFO logs)\npython3 scrape.py https://example.com -q\n```\n\n## MCP Server Setup\n\n> ⚠️ The MCP server starts a local HTTP service. Only use in trusted environments.\n\n```bash\nscrapling mcp\n# or\npython3 -m scrapling.mcp\n```\n\nAdd to OpenClaw MCP config (mcporter) to get scraping as a native tool. Confirm with user before starting."},{"path":"skill-card.md","content":"## Description:\n\nProvides a Scrapling-based web scraping workflow for fetching pages, extracting structured data with CSS or XPath selectors, and using HTTP, stealth, or dynamic browser modes on sites the user is authorized to scrape.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[jeminay](https://clawhub.ai/user/jeminay)\n\n### License/Terms of Use:\n\nMIT\n\n## Use Case:\n\nDevelopers and automation engineers use this skill to retrieve web page text or structured fields when ordinary fetch tools fail because of JavaScript rendering, rate limits, or anti-bot challenges. It is intended for authorized scraping workflows and selector-based extraction, not for bypassing restricted access.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill can send requests through HTTP, stealth, and browser-based modes to external sites.\n\nMitigation: Use it only for sites the user is authorized to scrape, respect applicable terms and robots.txt, and avoid passing credentials or sensitive URLs unless the destination is trusted.\n\nRisk: Installing full Scrapling dependencies and browser binaries expands the local runtime surface.\n\nMitigation: Install and run the skill in a virtual environment or disposable container without administrator privileges.\n\nRisk: The optional MCP server starts a local HTTP service.\n\nMitigation: Start the server only when explicitly needed and only in a trusted local environment.\n\nRisk: Adaptive scraping can persist element fingerprints or local state in the working directory.\n\nMitigation: Run adaptive workflows in a controlled directory and remove stored state when it is no longer needed.\n\n## Reference(s):\n\n- [Scrapling Patterns Reference](references/patterns.md)\n- [Scrapling on PyPI](https://pypi.org/project/scrapling/)\n- [ClawHub Skill Page](https://clawhub.ai/jeminay/skills/scrapling-fetcher)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, Code, Shell commands, Configuration, Guidance]\n\n**Output Format:** [Markdown guidance with inline shell commands, Python examples, plain text extraction, and optional JSON arrays]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Can emit selector matches as text lines or JSON; stealth and dynamic modes may use headless Chromium.]\n\n## Skill Version(s):\n\n1.0.3 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Web scraping using Scrapling — a Python framework with anti-bot bypass (Cloudflare Turnstile, fingerprint spoofing), adaptive element tracking, stealth headl... Skill: Scrapling - Stealth Web Scraper Owner: jeminay Summary: Web scraping using Scrapling — a Python framework with anti-bot bypass (Cloudflare Turnstile, fingerprint spoofing), adaptive element tracking, stealth headl... Tags: latest:1.0.3 Version history: v1.0.3 | 2026-02-25T07:47:52.031Z | user Added license (MIT) and metadata.source/pypi fields to frontmatter so registry shows verified provenance instead of 'So","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1265,"uniquenessScore":47,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T10:38:41.911Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T10:38:41.911Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T17:23:13.885Z","emptyReason":null},"items":[{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-10T18:48:31.762Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}