{"id":"7345dc87-10a3-43ab-88ed-b82f045e1f7f","entityType":"agent","slug":"clawhub-cn-big-cabbage-cn-scrapling","name":"scrapling","canonicalUrl":"https://www.xpersona.co/agent/clawhub-cn-big-cabbage-cn-scrapling","canonicalPath":"/agent/clawhub-cn-big-cabbage-cn-scrapling","generatedAt":"2026-10-11T20:57:37.875Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T18:08:57.473Z","emptyReason":null},"description":"高性能自适应 Python 网页抓取框架，内置反爬虫绕过（Cloudflare Turnstile）、智能元素重定位、完整爬虫框架和 MCP 服务器，适合 AI 辅助数据提取和大规模爬取任务","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s17dbd0j0kazk5ay31x9kv1z3n84f04v:cn-scrapling","sourceUrl":"https://clawhub.ai/cn-big-cabbage/cn-scrapling","homepage":"https://clawhub.ai/cn-big-cabbage/skills/cn-scrapling","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/cn-big-cabbage/cn-scrapling","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/cn-big-cabbage/skills/cn-scrapling","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":60,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"scrapling technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T18:08:57.473Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T18:08:57.473Z","emptyReason":null},"stars":null,"forks":null,"downloads":1012,"packageName":null,"latestVersion":"0.1.0","tractionLabel":"1K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T18:08:57.466Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T18:08:57.473Z","lastCrawledAt":"2026-10-11T18:08:57.466Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T18:08:57.466Z","lastVerifiedAt":null,"highlights":[{"version":"0.1.0","createdAt":"2026-04-22T00:43:46.652Z","changelog":"Initial release of Scrapling, a high-performance adaptive Python web scraping framework. - Supports automated anti-bot bypass (Cloudflare Turnstile), adaptive element re-location, and complete spider framework. - Includes MCP server for AI-assisted data extraction to reduce token consumption. - Offers three Fetcher classes (Fetcher, StealthyFetcher, DynamicFetcher) for HTTP, stealth/anti-bot, and browser automation tasks. - Features proxy rotation, session management, pause/resume for spiders, and CLI utilities. - Provides comprehensive documentation and integration guides for rapid use and extension.","fileCount":7,"zipByteSize":14601}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17dbd0j0kazk5ay31x9kv1z3n84f04v:cn-scrapling","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s17dbd0j0kazk5ay31x9kv1z3n84f04v:cn-scrapling` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/cn-big-cabbage/cn-scrapling before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-cn-big-cabbage-cn-scrapling/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-cn-big-cabbage-cn-scrapling/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-cn-big-cabbage-cn-scrapling/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-cn-big-cabbage-cn-scrapling/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-cn-big-cabbage-cn-scrapling/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-cn-big-cabbage-cn-scrapling/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T20:57:37.875Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-cn-big-cabbage-cn-scrapling/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-cn-big-cabbage-cn-scrapling/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-cn-big-cabbage-cn-scrapling/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-cn-big-cabbage-cn-scrapling/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T18:08:57.473Z","emptyReason":null},"readme":"Skill: scrapling\n\nOwner: cn-big-cabbage\n\nSummary: 高性能自适应 Python 网页抓取框架，内置反爬虫绕过（Cloudflare Turnstile）、智能元素重定位、完整爬虫框架和 MCP 服务器，适合 AI 辅助数据提取和大规模爬取任务\n\nTags: latest:0.1.0\n\nVersion history:\n\nv0.1.0 | 2026-04-22T00:43:46.652Z | auto\n\nInitial release of Scrapling, a high-performance adaptive Python web scraping framework.\n\n- Supports automated anti-bot bypass (Cloudflare Turnstile), adaptive element re-location, and complete spider framework.\n- Includes MCP server for AI-assisted data extraction to reduce token consumption.\n- Offers three Fetcher classes (Fetcher, StealthyFetcher, DynamicFetcher) for HTTP, stealth/anti-bot, and browser automation tasks.\n- Features proxy rotation, session management, pause/resume for spiders, and CLI utilities.\n- Provides comprehensive documentation and integration guides for rapid use and extension.\n\nArchive index:\n\nArchive v0.1.0: 7 files, 14601 bytes\n\nFiles: guides/01-installation.md (3181b), guides/02-quickstart.md (5503b), guides/03-advanced-usage.md (6682b), skill-card.md (2625b), SKILL.md (5538b), troubleshooting.md (7649b), _meta.json (131b)\n\nFile v0.1.0:SKILL.md\n\n---\nname: scrapling\ndescription: 高性能自适应 Python 网页抓取框架，内置反爬虫绕过（Cloudflare Turnstile）、智能元素重定位、完整爬虫框架和 MCP 服务器，适合 AI 辅助数据提取和大规模爬取任务\nversion: 0.1.0\nmetadata:\n  openclaw_requires: \">=1.0.0\"\n  emoji: 🕷️\n  homepage: https://scrapling.readthedocs.io\n---\n\n# Scrapling — 自适应网页抓取框架\n\nScrapling 是 Google Chrome DevTools 生态之外最强大的 Python 网页抓取框架之一，能够处理从单次 HTTP 请求到大规模并发爬取的所有场景。它的自适应解析引擎在网页改版后自动重新定位元素，内置 Cloudflare Turnstile 绕过能力，Spider 框架支持暂停/恢复，并提供 MCP 服务器让 AI 直接辅助数据提取，从源头减少 Token 消耗。\n\n## 核心使用场景\n\n- **反爬虫网站抓取**：`StealthyFetcher` 内置 Cloudflare Turnstile 绕过，支持 TLS 指纹伪装和浏览器自动化\n- **自适应数据采集**：网页改版后，`auto_save=True` 保存元素快照，`adaptive=True` 自动重新定位变化元素\n- **大规模并发爬取**：Spider 框架支持多 Session、代理轮换、暂停恢复，像 Scrapy 一样定义爬虫\n- **AI 辅助提取**：内置 MCP 服务器，Claude/Cursor 等 AI 工具可直接调用 Scrapling 提取目标内容\n- **动态页面处理**：`DynamicFetcher` 基于 Playwright，支持完整浏览器自动化和网络空闲等待\n\n## AI 辅助使用流程\n\n1. **安装依赖** — AI 执行 `pip install scrapling` 并按需安装浏览器驱动\n2. **选择 Fetcher** — AI 根据目标网站类型推荐 `Fetcher`/`StealthyFetcher`/`DynamicFetcher`\n3. **编写抓取逻辑** — AI 生成 CSS/XPath 选择器代码，配置 `auto_save` 实现自适应\n4. **调试与优化** — AI 分析响应结果，调整选择器或切换 Fetcher 策略\n5. **扩展为 Spider** — AI 将单页抓取扩展为完整 Spider 类，配置并发和代理\n6. **MCP 模式** — 启动 Scrapling MCP Server，让 AI 直接操控浏览器提取数据\n\n## 关键章节导航\n\n- [安装指南](guides/01-installation.md) — pip 安装、浏览器驱动、Docker 镜像\n- [快速开始](guides/02-quickstart.md) — Fetcher 选型、CSS/XPath 选择器、自适应抓取\n- [高级用法](guides/03-advanced-usage.md) — Spider 框架、代理轮换、MCP 服务器、CLI 工具\n- [故障排查](troubleshooting.md) — 反爬虫、浏览器驱动、超时、代理问题\n\n## AI 助手能力\n\n使用本技能时，AI 可以：\n\n- ✅ 安装 Scrapling 并配置浏览器驱动（`scrapling install playwright` / `scrapling install camoufox`）\n- ✅ 根据目标网站自动选择最合适的 Fetcher 类\n- ✅ 编写 CSS/XPath 选择器提取目标数据\n- ✅ 配置 `auto_save=True` 和 `adaptive=True` 实现自适应抓取\n- ✅ 构建完整的 Spider 类实现并发爬取，配置暂停/恢复\n- ✅ 设置代理轮换和防 DNS 泄露（DoH 模式）\n- ✅ 启动和配置 Scrapling MCP 服务器\n- ✅ 使用 CLI 工具快速测试 URL 抓取效果\n\n## 核心功能\n\n- ✅ **三种 Fetcher** — `Fetcher`（快速 HTTP）、`StealthyFetcher`（反爬绕过）、`DynamicFetcher`（浏览器自动化）\n- ✅ **自适应解析** — 网页改版后自动重定位元素，降低维护成本\n- ✅ **Cloudflare 绕过** — 内置 Turnstile/Interstitial 解决方案，免额外服务\n- ✅ **Spider 框架** — Scrapy 风格 API，支持并发、多 Session、暂停恢复\n- ✅ **流式输出** — `spider.stream()` 实时推送抓取结果，适合大规模任务\n- ✅ **MCP 服务器** — AI 工具直接调用 Scrapling 提取数据，减少 Token 消耗\n- ✅ **代理轮换** — 内置 `ProxyRotator`，支持循环或自定义策略\n- ✅ **会话管理** — `FetcherSession`/`StealthySession`/`DynamicSession` 跨请求保持状态\n- ✅ **开发模式** — 首次运行缓存响应，后续离线回放，快速迭代解析逻辑\n- ✅ **CLI 工具** — 无需写代码直接从终端抓取页面\n- ✅ **IPython Shell** — 交互式调试，内置 curl 转换工具\n- ✅ **Docker 镜像** — 预置所有浏览器的生产就绪镜像\n\n## 快速示例\n\n```python\nfrom scrapling.fetchers import Fetcher, StealthyFetcher, DynamicFetcher\n\n# 普通 HTTP 抓取（最快）\npage = Fetcher.get('https://quotes.toscrape.com/')\nquotes = page.css('.quote .text::text').getall()\n\n# 隐身模式绕过 Cloudflare\npage = StealthyFetcher.fetch('https://protected-site.com', headless=True)\ndata = page.css('.content::text').get()\n\n# 自适应抓取（网站改版后自动重定位）\npage = Fetcher.get('https://example.com/products')\nproducts = page.css('.product', auto_save=True)   # 首次保存元素快照\n# 网站改版后：\nproducts = page.css('.product', adaptive=True)    # 自动重新定位\n```\n\n```bash\n# CLI 快速测试（无需写代码）\nscrapling fetch https://quotes.toscrape.com/ --css \".quote .text\"\n\n# 启动 MCP 服务器\nscrapling mcp\n```\n\n## 安装要求\n\n| 依赖 | 版本要求 |\n|------|---------|\n| Python | >= 3.9 |\n| pip | 任意版本 |\n| Playwright | 可选（DynamicFetcher 使用） |\n| Camoufox | 可选（StealthyFetcher 使用） |\n| Docker | 可选（使用官方镜像） |\n\n## 项目链接\n\n- GitHub：https://github.com/D4Vinci/Scrapling\n- 文档：https://scrapling.readthedocs.io/en/latest/\n- PyPI：https://pypi.org/project/scrapling/\n- MCP 文档：https://scrapling.readthedocs.io/en/latest/ai/mcp-server.html\n- Discord：https://discord.gg/EMgGbDceNQ\n\nFile v0.1.0:_meta.json\n\n{\n  \"ownerId\": \"kn7c5ry95ff7x22mt3fx6j0q3d81p8t3\",\n  \"slug\": \"cn-scrapling\",\n  \"version\": \"0.1.0\",\n  \"publishedAt\": 1776818626652\n}\n\nFile v0.1.0:guides/01-installation.md\n\n# 安装指南\n\n## 适用场景\n\n- 安装 Scrapling 核心库并配置浏览器驱动\n- 在 Docker 环境中使用预置镜像\n- 为不同 Fetcher 安装对应依赖\n\n---\n\n## 基础安装\n\n> **AI 可自动执行**\n\n```bash\npip install scrapling\n```\n\n验证安装：\n```bash\npython -c \"import scrapling; print(scrapling.__version__)\"\n```\n\n---\n\n## 按需安装浏览器驱动\n\nScrapling 有三种 Fetcher，各需不同依赖：\n\n### Fetcher（纯 HTTP，无需额外驱动）\n\n```bash\npip install scrapling\n# 无需额外安装，开箱即用\n```\n\n### StealthyFetcher（隐身模式，需要 Camoufox）\n\n```bash\npip install scrapling\nscrapling install camoufox   # 安装修改版 Firefox 驱动\n```\n\n或手动安装：\n```bash\npip install camoufox[geoip]\npython -m camoufox fetch\n```\n\n### DynamicFetcher（完整浏览器自动化，需要 Playwright）\n\n```bash\npip install scrapling\nscrapling install playwright   # 安装 Playwright 和 Chromium\n```\n\n或手动安装：\n```bash\npip install playwright\nplaywright install chromium\n```\n\n---\n\n## 一次性安装全部依赖\n\n```bash\npip install scrapling\nscrapling install all\n```\n\n---\n\n## Docker 安装（推荐生产环境）\n\n使用官方预置镜像（含所有浏览器驱动）：\n\n```bash\n# 拉取最新镜像\ndocker pull d4vinci/scrapling:latest\n\n# 运行容器\ndocker run -it d4vinci/scrapling:latest python3\n\n# 在容器内直接使用\ndocker run --rm d4vinci/scrapling:latest python3 -c \"\nfrom scrapling.fetchers import StealthyFetcher\npage = StealthyFetcher.fetch('https://example.com', headless=True)\nprint(page.css('title::text').get())\n\"\n```\n\n---\n\n## 虚拟环境（推荐）\n\n```bash\npython -m venv scrapling-env\nsource scrapling-env/bin/activate   # Windows: scrapling-env\\Scripts\\activate\npip install scrapling\nscrapling install playwright\n```\n\n---\n\n## MCP 服务器安装（AI 集成）\n\nScrapling 内置 MCP 服务器，让 Claude/Cursor 等 AI 直接调用：\n\n### Claude Code\n\n```bash\nclaude mcp add scrapling --scope user npx -y scrapling-mcp\n```\n\n或手动安装后配置：\n```json\n{\n  \"mcpServers\": {\n    \"scrapling\": {\n      \"command\": \"python\",\n      \"args\": [\"-m\", \"scrapling.mcp\"]\n    }\n  }\n}\n```\n\n### 直接启动 MCP 服务器\n\n```bash\nscrapling mcp\n```\n\n---\n\n## 验证安装\n\n```python\n# 验证核心安装\nfrom scrapling.fetchers import Fetcher\npage = Fetcher.get('https://httpbin.org/get')\nprint(page.status)  # 期望：200\n\n# 验证 Playwright（DynamicFetcher）\nfrom scrapling.fetchers import DynamicFetcher\npage = DynamicFetcher.fetch('https://example.com', headless=True)\nprint(page.css('title::text').get())\n\n# 验证 Camoufox（StealthyFetcher）\nfrom scrapling.fetchers import StealthyFetcher\npage = StealthyFetcher.fetch('https://example.com', headless=True)\nprint(page.css('title::text').get())\n```\n\n---\n\n## 完成确认检查清单\n\n- [ ] `pip install scrapling` 执行成功\n- [ ] `python -c \"import scrapling\"` 无报错\n- [ ] 按需安装了 Playwright 或 Camoufox（视场景而定）\n- [ ] `Fetcher.get('https://httpbin.org/get').status == 200` 验证通过\n\n---\n\n## 下一步\n\n- [快速开始](02-quickstart.md) — Fetcher 选型指南、CSS/XPath 选择器、自适应抓取\n\nFile v0.1.0:guides/02-quickstart.md\n\n# 快速开始\n\n## 适用场景\n\n- 从静态或动态页面提取数据\n- 处理需要绕过反爬保护的网站\n- 使用 CSS/XPath 选择器精准提取内容\n- 实现网页改版后的自适应抓取\n\n---\n\n## 选择正确的 Fetcher\n\n| 场景 | Fetcher | 速度 |\n|------|---------|------|\n| 普通 HTTP 请求 | `Fetcher` | 最快 |\n| 需要 TLS 指纹伪装 | `Fetcher(impersonate='chrome')` | 快 |\n| Cloudflare / 反爬 | `StealthyFetcher` | 中 |\n| 需要 JS 渲染 | `DynamicFetcher` | 慢 |\n\n---\n\n## 基础 HTTP 抓取\n\n```python\nfrom scrapling.fetchers import Fetcher\n\n# 单次请求\npage = Fetcher.get('https://quotes.toscrape.com/')\n\n# 提取数据\nquotes = page.css('.quote .text::text').getall()\nauthors = page.css('.quote .author::text').getall()\nprint(quotes[:3])\n\n# XPath 方式\ntitles = page.xpath('//span[@class=\"text\"]/text()').getall()\n```\n\n---\n\n## Session 复用（跨请求保持状态）\n\n```python\nfrom scrapling.fetchers import FetcherSession\n\nwith FetcherSession(impersonate='chrome') as session:\n    # 使用最新版 Chrome TLS 指纹\n    page1 = session.get('https://example.com/', stealthy_headers=True)\n    page2 = session.get('https://example.com/products')  # 复用 cookie\n    \n    products = page2.css('.product h2::text').getall()\n```\n\n---\n\n## 绕过 Cloudflare 保护\n\n```python\nfrom scrapling.fetchers import StealthyFetcher, StealthySession\n\n# 单次请求（每次打开/关闭浏览器）\npage = StealthyFetcher.fetch(\n    'https://protected-site.com',\n    headless=True,\n    solve_cloudflare=True   # 自动处理 Cloudflare Turnstile\n)\ndata = page.css('.content').get()\n\n# Session 模式（保持浏览器，效率更高）\nwith StealthySession(headless=True) as session:\n    page = session.fetch('https://protected-site.com', google_search=False)\n    links = page.css('a.product-link::attr(href)').getall()\n```\n\n---\n\n## 动态页面（JS 渲染）\n\n```python\nfrom scrapling.fetchers import DynamicFetcher, DynamicSession\n\n# 等待网络空闲后抓取（确保异步数据加载完成）\npage = DynamicFetcher.fetch(\n    'https://spa-app.com',\n    headless=True,\n    network_idle=True\n)\nitems = page.css('.item-list .item::text').getall()\n\n# Session 模式（连续操作多个页面）\nwith DynamicSession(headless=True) as session:\n    page = session.fetch('https://example.com/login')\n    # 可以执行页面交互（通过 Playwright）\n    page2 = session.fetch('https://example.com/dashboard')\n    data = page2.css('.metric::text').getall()\n```\n\n---\n\n## CSS 和 XPath 选择器\n\n```python\npage = Fetcher.get('https://quotes.toscrape.com/')\n\n# CSS 选择器\ntitle = page.css('h1::text').get()                    # 第一个元素\nall_texts = page.css('.quote .text::text').getall()   # 全部\n\n# XPath 选择器\nauthor = page.xpath('//small[@class=\"author\"]/text()').get()\nhrefs = page.xpath('//a/@href').getall()\n\n# 属性提取\nlink = page.css('a::attr(href)').get()\nsrc = page.css('img::attr(src)').get()\n\n# 正则提取\nprice = page.css('.price::text').re_first(r'\\$[\\d.]+')\nnumbers = page.css('p::text').re(r'\\d+')\n```\n\n---\n\n## 自适应抓取（网页改版后自动重定位）\n\n这是 Scrapling 最独特的功能：\n\n```python\nfrom scrapling.fetchers import Fetcher\n\n# 第一次抓取：保存元素快照（存入本地数据库）\npage = Fetcher.get('https://example.com/products')\nproducts = page.css('.product-card', auto_save=True)\nprint(f\"找到 {len(products)} 个商品\")\nfor p in products:\n    print(p.css('h2::text').get(), p.css('.price::text').get())\n\n# 网站改版后（.product-card 变成了 .item-wrapper）\n# 传入 adaptive=True，自动通过历史快照重新定位\npage = Fetcher.get('https://example.com/products')\nproducts = page.css('.product-card', adaptive=True)   # 自动找到\n```\n\n仅查找相似元素：\n```python\n# 找到一个已知元素，然后查找页面上所有相似的元素\npage = Fetcher.get('https://example.com')\nfirst_item = page.css('.item').get()\nsimilar = page.find_similar(first_item)  # 自动定位同类元素\n```\n\n---\n\n## 查找相似元素\n\n```python\npage = Fetcher.get('https://news-site.com')\n\n# 找到第一篇文章\nfirst_article = page.css('article.news-item').get()\n\n# 自动找到所有相似的文章（即使 HTML 结构不完全一样）\narticles = page.find_similar(first_article)\nfor article in articles:\n    print(article.css('h2::text').get())\n```\n\n---\n\n## CLI 快速测试\n\n```bash\n# 抓取页面内容\nscrapling fetch https://quotes.toscrape.com/ --css \".quote .text\"\n\n# 指定 Fetcher 类型\nscrapling fetch https://protected-site.com --fetcher stealthy --headless\n\n# 输出完整 HTML\nscrapling fetch https://example.com --html\n```\n\n---\n\n## 交互式调试 Shell\n\n```bash\n# 启动 IPython Shell（含 Scrapling 集成）\nscrapling shell https://quotes.toscrape.com/\n\n# 在 Shell 中：\n>>> quotes = page.css('.quote .text::text').getall()\n>>> authors = page.css('.author::text').getall()\n>>> # 将 curl 命令转换为 Scrapling 代码：\n>>> curl_to_scrapling('curl -H \"User-Agent: Mozilla\" https://example.com')\n```\n\n---\n\n## 完成确认检查清单\n\n- [ ] `Fetcher.get()` 成功返回 200 状态码\n- [ ] CSS 选择器 `.getall()` 返回非空列表\n- [ ] 选择适当的 Fetcher（普通/隐身/动态）\n- [ ] `auto_save=True` 测试通过（数据保存到本地）\n- [ ] `adaptive=True` 在元素改变后仍能定位（可选测试）\n\n---\n\n## 下一步\n\n- [高级用法](03-advanced-usage.md) — Spider 框架、代理轮换、MCP 服务器、流式输出\n\nFile v0.1.0:guides/03-advanced-usage.md\n\n# 高级用法\n\n## Spider 框架（大规模并发爬取）\n\nSpider 框架提供 Scrapy 风格的 API，适合需要跨多页面爬取的任务：\n\n```python\nfrom scrapling.spiders import Spider, Request, Response\n\nclass QuotesSpider(Spider):\n    name = \"quotes\"\n    start_urls = [\"https://quotes.toscrape.com/\"]\n    concurrent_requests = 10       # 最大并发数\n    download_delay = 0.5           # 请求间隔（秒）\n    robots_txt_obey = True         # 遵守 robots.txt\n\n    async def parse(self, response: Response):\n        for quote in response.css('.quote'):\n            yield {\n                \"text\": quote.css('.text::text').get(),\n                \"author\": quote.css('.author::text').get(),\n                \"tags\": quote.css('.tag::text').getall(),\n            }\n        # 翻页\n        next_page = response.css('li.next a::attr(href)').get()\n        if next_page:\n            yield Request(response.urljoin(next_page))\n\n# 运行爬虫\nresult = QuotesSpider().start()\nresult.items.to_json('quotes.json')       # 导出 JSON\nresult.items.to_jsonl('quotes.jsonl')     # 导出 JSONL\n```\n\n---\n\n## 暂停与恢复爬虫\n\n```python\nfrom scrapling.spiders import Spider\n\nclass LargeSpider(Spider):\n    name = \"large\"\n    start_urls = [\"https://large-site.com/\"]\n    # 开启检查点持久化\n    checkpoint = True\n\n    async def parse(self, response):\n        # ...爬取逻辑...\n        pass\n\n# 启动时按 Ctrl+C 优雅停止\nresult = LargeSpider().start()\n\n# 下次运行时自动从上次停止的地方继续\nresult = LargeSpider().start()   # 无需额外配置\n```\n\n---\n\n## 流式输出（实时处理）\n\n```python\nfrom scrapling.spiders import Spider\nimport asyncio\n\nclass StreamingSpider(Spider):\n    name = \"stream\"\n    start_urls = [\"https://news-site.com/\"]\n\n    async def parse(self, response):\n        for article in response.css('article'):\n            yield {\"title\": article.css('h2::text').get()}\n\nasync def main():\n    spider = StreamingSpider()\n    async for item in spider.stream():\n        # 实时处理每个抓取到的 item\n        print(item)\n        # 实时写入数据库、发送到队列等\n\nasyncio.run(main())\n```\n\n---\n\n## 多 Session 路由（混用 HTTP 和浏览器）\n\n```python\nfrom scrapling.spiders import Spider, Request, Response\n\nclass HybridSpider(Spider):\n    name = \"hybrid\"\n    start_urls = [\"https://example.com/catalog\"]\n\n    async def parse(self, response: Response):\n        for link in response.css('.product-link::attr(href)').getall():\n            # 普通页面用 HTTP，详情页面用浏览器\n            if '/protected/' in link:\n                yield Request(link, session_id='stealthy')   # 使用 StealthyFetcher\n            else:\n                yield Request(link)                           # 使用默认 HTTP\n\n    async def parse_product(self, response: Response):\n        yield {\"title\": response.css('h1::text').get()}\n```\n\n---\n\n## 代理轮换\n\n```python\nfrom scrapling.fetchers import Fetcher, FetcherSession, StealthyFetcher\nfrom scrapling.proxy_rotator import ProxyRotator\n\n# 设置代理列表\nproxies = [\n    \"http://user:pass@proxy1.example.com:8080\",\n    \"http://user:pass@proxy2.example.com:8080\",\n    \"socks5://user:pass@proxy3.example.com:1080\",\n]\n\n# 循环轮换\nrotator = ProxyRotator(proxies, mode='cyclic')\n\nwith FetcherSession(proxy=rotator) as session:\n    page = session.get('https://example.com')\n    print(page.css('title::text').get())\n\n# Spider 中使用代理\nclass ProxySpider(Spider):\n    name = \"proxy\"\n    start_urls = [\"https://example.com/\"]\n    proxies = proxies   # 直接指定代理列表（自动轮换）\n\n    async def parse(self, response):\n        yield {\"url\": response.url}\n```\n\n---\n\n## 防 DNS 泄露（DoH 模式）\n\n使用代理时，防止真实 IP 通过 DNS 查询泄露：\n\n```python\nfrom scrapling.fetchers import Fetcher\n\npage = Fetcher.get(\n    'https://example.com',\n    doh=True   # 通过 Cloudflare DoH 路由 DNS 查询\n)\n```\n\n---\n\n## 广告拦截和域名过滤\n\n```python\nfrom scrapling.fetchers import DynamicFetcher, DynamicSession\n\n# 开启内置广告拦截（约 3500 个广告/追踪域名）\nwith DynamicSession(headless=True, block_ads=True) as session:\n    page = session.fetch('https://news-site.com/')\n\n# 自定义域名黑名单（拦截指定域名及其所有子域名）\nwith DynamicSession(\n    headless=True,\n    blocked_domains=['ads.example.com', 'tracker.net']\n) as session:\n    page = session.fetch('https://example.com/')\n```\n\n---\n\n## MCP 服务器（AI 集成）\n\nScrapling 内置 MCP 服务器，让 Claude/Cursor 直接控制浏览器提取数据，在传给 AI 之前完成精准提取，大幅降低 Token 消耗：\n\n### 启动 MCP 服务器\n\n```bash\n# 直接启动（用于手动测试）\nscrapling mcp\n\n# 在 Claude Code 中配置\nclaude mcp add scrapling --scope user python -m scrapling.mcp\n```\n\n### Claude Code 配置\n\n```json\n{\n  \"mcpServers\": {\n    \"scrapling\": {\n      \"command\": \"python\",\n      \"args\": [\"-m\", \"scrapling.mcp\"],\n      \"env\": {}\n    }\n  }\n}\n```\n\n### 在 AI 对话中使用\n\n```\n请用 Scrapling 抓取 https://news.ycombinator.com/ 的所有文章标题和链接\n\n请用隐身模式抓取 https://example.com/products 的商品列表，这个网站有 Cloudflare 保护\n```\n\n---\n\n## 开发模式（离线调试）\n\n```python\nfrom scrapling.fetchers import Fetcher\n\n# 首次运行：缓存响应到磁盘\npage = Fetcher.get('https://example.com', cache=True)\nproducts = page.css('.product', auto_save=True)\n\n# 后续运行：从缓存读取，无需网络请求（快速迭代解析逻辑）\npage = Fetcher.get('https://example.com', cache=True, dev_mode=True)\nproducts = page.css('.product', adaptive=True)\n```\n\n---\n\n## 异步并发抓取\n\n```python\nimport asyncio\nfrom scrapling.fetchers import AsyncFetcher\n\nasync def fetch_all(urls):\n    async with AsyncFetcher() as fetcher:\n        tasks = [fetcher.get(url) for url in urls]\n        pages = await asyncio.gather(*tasks)\n        return [page.css('title::text').get() for page in pages]\n\nurls = [f\"https://example.com/page/{i}\" for i in range(1, 20)]\ntitles = asyncio.run(fetch_all(urls))\n```\n\n---\n\n## 完成确认检查清单\n\n- [ ] Spider 类成功爬取多页数据并导出 JSON\n- [ ] 代理轮换配置正确（可选）\n- [ ] MCP 服务器启动成功（如需 AI 集成）\n- [ ] 暂停/恢复功能测试通过（长任务场景）\n\n---\n\n## 相关链接\n\n- [故障排查](../troubleshooting.md)\n- [完整文档](https://scrapling.readthedocs.io/en/latest/)\n- [Spider API 参考](https://scrapling.readthedocs.io/en/latest/spiders/architecture.html)\n- [MCP 文档](https://scrapling.readthedocs.io/en/latest/ai/mcp-server.html)\n\nFile v0.1.0:skill-card.md\n\n## Description:\n\nScrapling helps agents install and use the Scrapling Python web scraping framework for adaptive extraction, browser automation, spider workflows, and MCP-assisted data collection.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[cn-big-cabbage](https://clawhub.ai/user/cn-big-cabbage)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and external users use this skill to guide agents through Scrapling installation, fetcher selection, scraping code generation, browser-driver setup, spider configuration, proxy usage, and troubleshooting for web data extraction tasks.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill can guide agents toward anti-bot bypass, browser automation, proxy use, and high-volume scraping that may conflict with site terms, authorization, privacy, or rate limits.\n\nMitigation: Use only for authorized targets; check site terms, robots and rate-limit requirements; reduce concurrency where needed; and keep scraped sensitive or access-restricted data out of AI and MCP workflows.\n\nRisk: Persistent MCP registration or agent-run shell commands can broaden access beyond the immediate scraping task.\n\nMitigation: Prefer project-scoped MCP registration, pin package and Docker versions, and review installation, browser-driver, proxy, and MCP commands before execution.\n\n## Reference(s):\n\n- [Scrapling documentation](https://scrapling.readthedocs.io)\n- [Scrapling latest documentation](https://scrapling.readthedocs.io/en/latest/)\n- [Scrapling PyPI package](https://pypi.org/project/scrapling/)\n- [Scrapling MCP server documentation](https://scrapling.readthedocs.io/en/latest/ai/mcp-server.html)\n- [Scrapling GitHub project](https://github.com/D4Vinci/Scrapling)\n- [Scrapling Spider architecture](https://scrapling.readthedocs.io/en/latest/spiders/architecture.html)\n\n## Skill Output:\n\n**Output Type(s):** [guidance, markdown, code, shell commands, configuration]\n\n**Output Format:** [Markdown guidance with Python, shell, and JSON configuration snippets]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May recommend browser automation, MCP registration, proxy configuration, and local file or database state for adaptive scraping workflows.]\n\n## Skill Version(s):\n\n0.1.0 (source: frontmatter and release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v0.1.0:troubleshooting.md\n\n# 故障排查\n\n## 安装问题\n\n---\n\n### 问题 1：`camoufox` 安装失败或 `scrapling install camoufox` 报错\n\n**难度：** 低\n\n**症状：** `ERROR: Could not build wheels for camoufox` 或 `ModuleNotFoundError: No module named 'camoufox'`\n\n**排查步骤：**\n```bash\npip show camoufox\npython --version  # 需要 >= 3.9\n```\n\n**解决方案：**\n```bash\n# 方式一：通过 scrapling 安装\npip install scrapling\nscrapling install camoufox\n\n# 方式二：直接安装（含 GeoIP 数据）\npip install \"camoufox[geoip]\"\npython -m camoufox fetch\n\n# 如果遇到编译错误，先升级 pip\npip install --upgrade pip setuptools wheel\npip install \"camoufox[geoip]\"\n```\n\n---\n\n### 问题 2：Playwright 安装后浏览器找不到\n\n**难度：** 低\n\n**症状：** `BrowserNotFoundError: Executable doesn't exist at ...` 或 `Playwright requires browser binaries`\n\n**解决方案：**\n```bash\n# 必须在安装 playwright 包后，额外下载浏览器\npip install playwright\nplaywright install chromium   # 只装 Chromium（体积小）\n# 或\nplaywright install            # 安装所有浏览器\n\n# 验证\nplaywright --version\npython -c \"from playwright.sync_api import sync_playwright; sync_playwright().start()\"\n```\n\n---\n\n### 问题 3：Linux 服务器缺少系统依赖\n\n**难度：** 中\n\n**症状：** `error while loading shared libraries: libatk-1.0.so.0` 或类似 `libXXX.so` 缺失错误\n\n**解决方案：**\n```bash\n# Ubuntu/Debian\nplaywright install-deps chromium\n\n# 或手动安装\nsudo apt-get install -y libatk1.0-0 libatk-bridge2.0-0 libcups2 \\\n  libdrm2 libxkbcommon0 libxcomposite1 libxdamage1 libxrandr2 \\\n  libgbm1 libpango-1.0-0 libcairo2 libasound2\n\n# 推荐：直接使用 Docker 镜像避免依赖问题\ndocker pull d4vinci/scrapling:latest\n```\n\n---\n\n## 使用问题\n\n---\n\n### 问题 4：Cloudflare 或反爬虫未被绕过\n\n**难度：** 中\n\n**症状：** 返回 403 状态码、Cloudflare 验证页面 HTML、或空白内容\n\n**常见原因：**\n- 使用了普通 `Fetcher` 而不是 `StealthyFetcher`（概率 50%）\n- `headless=True` 模式下被检测（概率 30%）\n- 需要 `solve_cloudflare=True` 参数（概率 20%）\n\n**解决方案：**\n```python\nfrom scrapling.fetchers import StealthyFetcher\n\n# 基础反爬绕过\npage = StealthyFetcher.fetch(\n    'https://protected-site.com',\n    headless=True,\n    network_idle=True\n)\nprint(page.status)  # 期望 200\n\n# Cloudflare Turnstile 专用参数\npage = StealthyFetcher.fetch(\n    'https://cloudflare-site.com',\n    headless=True,\n    solve_cloudflare=True,    # 自动处理 Turnstile\n    google_search=False        # 不模拟来自 Google 搜索\n)\n\n# 检查是否被拦截\nif 'cloudflare' in page.html.lower() or page.status == 403:\n    print(\"仍被拦截，尝试 headless=False（有头模式）\")\n```\n\n---\n\n### 问题 5：`adaptive=True` 找不到元素（返回空列表）\n\n**难度：** 中\n\n**症状：** `products = page.css('.product', adaptive=True)` 返回 `[]`\n\n**常见原因：**\n- 从未使用 `auto_save=True` 保存过该元素的快照（概率 60%）\n- 快照数据库路径不一致（概率 25%）\n- 网页结构变化太大，相似度算法无法匹配（概率 15%）\n\n**解决方案：**\n```python\n# 步骤 1：确认先用 auto_save 保存过快照\npage = Fetcher.get('https://example.com/products')\nproducts = page.css('.product', auto_save=True)  # 必须先保存\nprint(f\"已保存 {len(products)} 个元素快照\")\n\n# 步骤 2：检查快照数据库存在\nimport os\ndb_path = os.path.expanduser('~/.scrapling/storage.db')\nprint(f\"数据库存在: {os.path.exists(db_path)}\")\n\n# 步骤 3：验证 adaptive 模式\npage = Fetcher.get('https://example.com/products')\nproducts = page.css('.product', adaptive=True)\nprint(f\"自适应找到 {len(products)} 个元素\")\n```\n\n---\n\n### 问题 6：Spider 爬取速度过慢\n\n**难度：** 中\n\n**症状：** 爬虫运行时 CPU 占用低，但进度很慢\n\n**解决方案：**\n```python\nfrom scrapling.spiders import Spider\n\nclass FastSpider(Spider):\n    name = \"fast\"\n    start_urls = [\"https://example.com/\"]\n    \n    # 增加并发（根据目标网站承受能力调整）\n    concurrent_requests = 20\n    \n    # 减少延迟（礼貌爬取建议 >= 0.5）\n    download_delay = 0.2\n    \n    # 禁用 robots.txt 检查（如确认不需要）\n    robots_txt_obey = False\n    \n    async def parse(self, response):\n        pass\n```\n\n也可以改用异步 Fetcher 批量并发：\n```python\nimport asyncio\nfrom scrapling.fetchers import AsyncFetcher\n\nasync def main():\n    urls = [\"https://example.com/page/{}\".format(i) for i in range(100)]\n    async with AsyncFetcher() as f:\n        results = await asyncio.gather(*[f.get(u) for u in urls])\n```\n\n---\n\n### 问题 7：CSS 选择器或 XPath 返回空结果\n\n**难度：** 低\n\n**症状：** `.get()` 返回 `None`，`.getall()` 返回 `[]`\n\n**排查步骤：**\n```python\npage = Fetcher.get('https://example.com')\n\n# 1. 确认页面已加载\nprint(page.status, len(page.html))\n\n# 2. 检查页面是否需要 JS 渲染\nif '<noscript>' in page.html or 'Please enable JavaScript' in page.html:\n    print(\"需要切换到 DynamicFetcher\")\n\n# 3. 打印页面结构\nprint(page.css('body').get()[:500])\n\n# 4. 用交互式 Shell 调试\n# scrapling shell https://example.com\n```\n\n**解决方案：**\n```python\n# 如果页面需要 JS 渲染，切换 Fetcher\nfrom scrapling.fetchers import DynamicFetcher\n\npage = DynamicFetcher.fetch('https://example.com', headless=True, network_idle=True)\ndata = page.css('.dynamic-content::text').getall()\n```\n\n---\n\n## 网络/环境问题\n\n---\n\n### 问题 8：代理连接失败\n\n**难度：** 中\n\n**症状：** `ProxyError: Cannot connect to proxy` 或 `407 Proxy Authentication Required`\n\n**排查步骤：**\n```bash\n# 先验证代理是否可用\ncurl -x \"http://user:pass@proxy.example.com:8080\" https://httpbin.org/ip\n```\n\n**解决方案：**\n```python\nfrom scrapling.fetchers import Fetcher\n\n# 确认代理格式正确（包含协议头）\nproxy = \"http://username:password@proxy.example.com:8080\"\npage = Fetcher.get('https://httpbin.org/ip', proxy=proxy)\nprint(page.json()['origin'])  # 应显示代理 IP\n\n# SOCKS5 代理\nproxy_socks = \"socks5://user:pass@proxy.example.com:1080\"\npage = Fetcher.get('https://httpbin.org/ip', proxy=proxy_socks)\n```\n\n---\n\n### 问题 9：并发爬取触发目标网站限流（429）\n\n**难度：** 中\n\n**症状：** 大量请求返回 `429 Too Many Requests` 或被封禁\n\n**解决方案：**\n```python\nfrom scrapling.spiders import Spider\n\nclass PoliteSipder(Spider):\n    name = \"polite\"\n    start_urls = [\"https://example.com/\"]\n    \n    # 降低并发和增加延迟\n    concurrent_requests = 3\n    download_delay = 2.0            # 2 秒间隔\n    randomize_download_delay = True  # 随机化延迟（更自然）\n    \n    # 配合代理轮换\n    proxies = [\"http://proxy1:8080\", \"http://proxy2:8080\"]\n\n    async def parse(self, response):\n        if response.status == 429:\n            # 触发重试（Spider 会自动重试）\n            return\n```\n\n---\n\n## 通用诊断\n\n```python\n# 快速诊断页面获取情况\nfrom scrapling.fetchers import Fetcher\n\npage = Fetcher.get('https://example.com')\nprint({\n    \"status\": page.status,\n    \"html_length\": len(page.html),\n    \"title\": page.css('title::text').get(),\n    \"url\": page.url,\n})\n```\n\n```bash\n# CLI 快速诊断\nscrapling fetch https://example.com --html | head -50\n\n# 交互式调试\nscrapling shell https://example.com\n```\n\n**文档：** https://scrapling.readthedocs.io/en/latest/\n\n**GitHub Issues：** https://github.com/D4Vinci/Scrapling/issues\n\n**Discord 社区：** https://discord.gg/EMgGbDceNQ","readmeExcerpt":"Skill: scrapling Owner: cn-big-cabbage Summary: 高性能自适应 Python 网页抓取框架，内置反爬虫绕过（Cloudflare Turnstile）、智能元素重定位、完整爬虫框架和 MCP 服务器，适合 AI 辅助数据提取和大规模爬取任务 Tags: latest:0.1.0 Version history: v0.1.0 | 2026-04-22T00:43:46.652Z | auto Initial release of Scrapling, a high-performance adaptive Python web scraping framework. - Supports automated anti-bot bypass (Cloudflare Turnstile), adaptive element re-location, and complete spider","codeSnippets":[],"executableExamples":[{"language":"python","snippet":"from scrapling.fetchers import Fetcher, StealthyFetcher, DynamicFetcher\n\n# 普通 HTTP 抓取（最快）\npage = Fetcher.get('https://quotes.toscrape.com/')\nquotes = page.css('.quote .text::text').getall()\n\n# 隐身模式绕过 Cloudflare\npage = StealthyFetcher.fetch('https://protected-site.com', headless=True)\ndata = page.css('.content::text').get()\n\n# 自适应抓取（网站改版后自动重定位）\npage = Fetcher.get('https://example.com/products')\nproducts = page.css('.product', auto_save=True)   # 首次保存元素快照\n# 网站改版后：\nproducts = page.css('.product', adaptive=True)    # 自动重新定位"},{"language":"bash","snippet":"# CLI 快速测试（无需写代码）\nscrapling fetch https://quotes.toscrape.com/ --css \".quote .text\"\n\n# 启动 MCP 服务器\nscrapling mcp"},{"language":"bash","snippet":"pip install scrapling"},{"language":"bash","snippet":"python -c \"import scrapling; print(scrapling.__version__)\""},{"language":"bash","snippet":"pip install scrapling\n# 无需额外安装，开箱即用"},{"language":"bash","snippet":"pip install scrapling\nscrapling install camoufox   # 安装修改版 Firefox 驱动"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: scrapling\ndescription: 高性能自适应 Python 网页抓取框架，内置反爬虫绕过（Cloudflare Turnstile）、智能元素重定位、完整爬虫框架和 MCP 服务器，适合 AI 辅助数据提取和大规模爬取任务\nversion: 0.1.0\nmetadata:\n  openclaw_requires: \">=1.0.0\"\n  emoji: 🕷️\n  homepage: https://scrapling.readthedocs.io\n---\n\n# Scrapling — 自适应网页抓取框架\n\nScrapling 是 Google Chrome DevTools 生态之外最强大的 Python 网页抓取框架之一，能够处理从单次 HTTP 请求到大规模并发爬取的所有场景。它的自适应解析引擎在网页改版后自动重新定位元素，内置 Cloudflare Turnstile 绕过能力，Spider 框架支持暂停/恢复，并提供 MCP 服务器让 AI 直接辅助数据提取，从源头减少 Token 消耗。\n\n## 核心使用场景\n\n- **反爬虫网站抓取**：`StealthyFetcher` 内置 Cloudflare Turnstile 绕过，支持 TLS 指纹伪装和浏览器自动化\n- **自适应数据采集**：网页改版后，`auto_save=True` 保存元素快照，`adaptive=True` 自动重新定位变化元素\n- **大规模并发爬取**：Spider 框架支持多 Session、代理轮换、暂停恢复，像 Scrapy 一样定义爬虫\n- **AI 辅助提取**：内置 MCP 服务器，Claude/Cursor 等 AI 工具可直接调用 Scrapling 提取目标内容\n- **动态页面处理**：`DynamicFetcher` 基于 Playwright，支持完整浏览器自动化和网络空闲等待\n\n## AI 辅助使用流程\n\n1. **安装依赖** — AI 执行 `pip install scrapling` 并按需安装浏览器驱动\n2. **选择 Fetcher** — AI 根据目标网站类型推荐 `Fetcher`/`StealthyFetcher`/`DynamicFetcher`\n3. **编写抓取逻辑** — AI 生成 CSS/XPath 选择器代码，配置 `auto_save` 实现自适应\n4. **调试与优化** — AI 分析响应结果，调整选择器或切换 Fetcher 策略\n5. **扩展为 Spider** — AI 将单页抓取扩展为完整 Spider 类，配置并发和代理\n6. **MCP 模式** — 启动 Scrapling MCP Server，让 AI 直接操控浏览器提取数据\n\n## 关键章节导航\n\n- [安装指南](guides/01-installation.md) — pip 安装、浏览器驱动、Docker 镜像\n- [快速开始](guides/02-quickstart.md) — Fetcher 选型、CSS/XPath 选择器、自适应抓取\n- [高级用法](guides/03-advanced-usage.md) — Spider 框架、代理轮换、MCP 服务器、CLI 工具\n- [故障排查](troubleshooting.md) — 反爬虫、浏览器驱动、超时、代理问题\n\n## AI 助手能力\n\n使用本技能时，AI 可以：\n\n- ✅ 安装 Scrapling 并配置浏览器驱动（`scrapling install playwright` / `scrapling install camoufox`）\n- ✅ 根据目标网站自动选择最合适的 Fetcher 类\n- ✅ 编写 CSS/XPath 选择器提取目标数据\n- ✅ 配置 `auto_save=True` 和 `adaptive=True` 实现自适应抓取\n- ✅ 构建完整的 Spider 类实现并发爬取，配置暂停/恢复\n- ✅ 设置代理轮换和防 DNS 泄露（DoH 模式）\n- ✅ 启动和配置 Scrapling MCP 服务器\n- ✅ 使用 CLI 工具快速测试 URL 抓取效果\n\n## 核心功能\n\n- ✅ **三种 Fetcher** — `Fetcher`（快速 HTTP）、`StealthyFetcher`（反爬绕过）、`DynamicFetcher`（浏览器自动化）\n- ✅ **自适应解析** — 网页改版后自动重定位元素，降低维护成本\n- ✅ **Cloudflare 绕过** — 内置 Turnstile/Interstitial 解决方案，免额外服务\n- ✅ **Spider 框架** — Scrapy 风格 API，支持并发、多 Session、暂停恢复\n- ✅ **流式输出** — `spider.stream()` 实时推送抓取结果，适合大规模任务\n- ✅ **MCP 服务器** — AI 工具直接调用 Scrapling 提取数据，减少 Token 消耗\n- ✅ **代理轮换** — 内置 `ProxyRotator`，支持循环或自定义策略\n- ✅ **会话管理** — `FetcherSession`/`StealthySession`/`DynamicSession` 跨请求保持状态\n- ✅ **开发模式** — 首次运行缓存响应，后续离线回放，快速迭代解析逻辑\n- ✅ **CLI 工具** — 无需写代码直接从终端抓取页面\n- ✅ **IPython Shell** — 交互式调试，内置 curl 转换工具\n- ✅ **Docker 镜像** — 预置所有浏览器的生产就绪镜像\n\n## 快速示例\n\n```python\nfrom scrapling.fetchers import Fetcher, StealthyFetcher, DynamicFetcher\n\n# 普通 HTTP 抓取（最快）\npage = Fetcher.get('https://quotes.toscrape.com/')\nquotes = page.css('.quote .text::text').getall()\n\n# 隐身模式绕过 Cloudflare\npage = StealthyFetcher.fetch('https://protected-site.com', headless=True)\ndata = page.css('.content::text').get()\n\n# 自适应抓取（网站改版后自动重定位）\npage = Fetcher.get('https://example.com/products')\nproducts = page.css('.product', auto_save=True)   # 首次保存元素快照\n# 网站改版后：\nproducts = page.css('.product', adaptive=True)    # 自动重新定位\n```\n\n```bash\n# CLI 快速测试（无需写代码）\nscrapling fetc"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7c5ry95ff7x22mt3fx6j0q3d81p8t3\",\n  \"slug\": \"cn-scrapling\",\n  \"version\": \"0.1.0\",\n  \"publishedAt\": 1776818626652\n}"},{"path":"guides/01-installation.md","content":"# 安装指南\n\n## 适用场景\n\n- 安装 Scrapling 核心库并配置浏览器驱动\n- 在 Docker 环境中使用预置镜像\n- 为不同 Fetcher 安装对应依赖\n\n---\n\n## 基础安装\n\n> **AI 可自动执行**\n\n```bash\npip install scrapling\n```\n\n验证安装：\n```bash\npython -c \"import scrapling; print(scrapling.__version__)\"\n```\n\n---\n\n## 按需安装浏览器驱动\n\nScrapling 有三种 Fetcher，各需不同依赖：\n\n### Fetcher（纯 HTTP，无需额外驱动）\n\n```bash\npip install scrapling\n# 无需额外安装，开箱即用\n```\n\n### StealthyFetcher（隐身模式，需要 Camoufox）\n\n```bash\npip install scrapling\nscrapling install camoufox   # 安装修改版 Firefox 驱动\n```\n\n或手动安装：\n```bash\npip install camoufox[geoip]\npython -m camoufox fetch\n```\n\n### DynamicFetcher（完整浏览器自动化，需要 Playwright）\n\n```bash\npip install scrapling\nscrapling install playwright   # 安装 Playwright 和 Chromium\n```\n\n或手动安装：\n```bash\npip install playwright\nplaywright install chromium\n```\n\n---\n\n## 一次性安装全部依赖\n\n```bash\npip install scrapling\nscrapling install all\n```\n\n---\n\n## Docker 安装（推荐生产环境）\n\n使用官方预置镜像（含所有浏览器驱动）：\n\n```bash\n# 拉取最新镜像\ndocker pull d4vinci/scrapling:latest\n\n# 运行容器\ndocker run -it d4vinci/scrapling:latest python3\n\n# 在容器内直接使用\ndocker run --rm d4vinci/scrapling:latest python3 -c \"\nfrom scrapling.fetchers import StealthyFetcher\npage = StealthyFetcher.fetch('https://example.com', headless=True)\nprint(page.css('title::text').get())\n\"\n```\n\n---\n\n## 虚拟环境（推荐）\n\n```bash\npython -m venv scrapling-env\nsource scrapling-env/bin/activate   # Windows: scrapling-env\\Scripts\\activate\npip install scrapling\nscrapling install playwright\n```\n\n---\n\n## MCP 服务器安装（AI 集成）\n\nScrapling 内置 MCP 服务器，让 Claude/Cursor 等 AI 直接调用：\n\n### Claude Code\n\n```bash\nclaude mcp add scrapling --scope user npx -y scrapling-mcp\n```\n\n或手动安装后配置：\n```json\n{\n  \"mcpServers\": {\n    \"scrapling\": {\n      \"command\": \"python\",\n      \"args\": [\"-m\", \"scrapling.mcp\"]\n    }\n  }\n}\n```\n\n### 直接启动 MCP 服务器\n\n```bash\nscrapling mcp\n```\n\n---\n\n## 验证安装\n\n```python\n# 验证核心安装\nfrom scrapling.fetchers import Fetcher\npage = Fetcher.get('https://httpbin.org/get')\nprint(page.status)  # 期望：200\n\n# 验证 Playwright（DynamicFetcher）\nfrom scrapling.fetchers import DynamicFetcher\npage = DynamicFetcher.fetch('https://example.com', headless=True)\nprint(page.css('title::text').get())\n\n# 验证 Camoufox（StealthyFetcher）\nfrom scrapling.fetchers import StealthyFetcher\npage = StealthyFetcher.fetch('https://example.com', headless=True)\nprint(page.css('title::text').get())\n```\n\n---\n\n## 完成确认检查清单\n\n- [ ] `pip install scrapling` 执行成功\n- [ ] `python -c \"import scrapling\"` 无报错\n- [ ] 按需安装了 Playwright 或 Camoufox（视场景而定）\n- [ ] `Fetcher.get('https://httpbin.org/get').status == 200` 验证通过\n\n---\n\n## 下一步\n\n- [快速开始](02-quickstart.md) — Fetcher 选型指南、CSS/XPath 选择器、自适应抓取"},{"path":"guides/02-quickstart.md","content":"# 快速开始\n\n## 适用场景\n\n- 从静态或动态页面提取数据\n- 处理需要绕过反爬保护的网站\n- 使用 CSS/XPath 选择器精准提取内容\n- 实现网页改版后的自适应抓取\n\n---\n\n## 选择正确的 Fetcher\n\n| 场景 | Fetcher | 速度 |\n|------|---------|------|\n| 普通 HTTP 请求 | `Fetcher` | 最快 |\n| 需要 TLS 指纹伪装 | `Fetcher(impersonate='chrome')` | 快 |\n| Cloudflare / 反爬 | `StealthyFetcher` | 中 |\n| 需要 JS 渲染 | `DynamicFetcher` | 慢 |\n\n---\n\n## 基础 HTTP 抓取\n\n```python\nfrom scrapling.fetchers import Fetcher\n\n# 单次请求\npage = Fetcher.get('https://quotes.toscrape.com/')\n\n# 提取数据\nquotes = page.css('.quote .text::text').getall()\nauthors = page.css('.quote .author::text').getall()\nprint(quotes[:3])\n\n# XPath 方式\ntitles = page.xpath('//span[@class=\"text\"]/text()').getall()\n```\n\n---\n\n## Session 复用（跨请求保持状态）\n\n```python\nfrom scrapling.fetchers import FetcherSession\n\nwith FetcherSession(impersonate='chrome') as session:\n    # 使用最新版 Chrome TLS 指纹\n    page1 = session.get('https://example.com/', stealthy_headers=True)\n    page2 = session.get('https://example.com/products')  # 复用 cookie\n    \n    products = page2.css('.product h2::text').getall()\n```\n\n---\n\n## 绕过 Cloudflare 保护\n\n```python\nfrom scrapling.fetchers import StealthyFetcher, StealthySession\n\n# 单次请求（每次打开/关闭浏览器）\npage = StealthyFetcher.fetch(\n    'https://protected-site.com',\n    headless=True,\n    solve_cloudflare=True   # 自动处理 Cloudflare Turnstile\n)\ndata = page.css('.content').get()\n\n# Session 模式（保持浏览器，效率更高）\nwith StealthySession(headless=True) as session:\n    page = session.fetch('https://protected-site.com', google_search=False)\n    links = page.css('a.product-link::attr(href)').getall()\n```\n\n---\n\n## 动态页面（JS 渲染）\n\n```python\nfrom scrapling.fetchers import DynamicFetcher, DynamicSession\n\n# 等待网络空闲后抓取（确保异步数据加载完成）\npage = DynamicFetcher.fetch(\n    'https://spa-app.com',\n    headless=True,\n    network_idle=True\n)\nitems = page.css('.item-list .item::text').getall()\n\n# Session 模式（连续操作多个页面）\nwith DynamicSession(headless=True) as session:\n    page = session.fetch('https://example.com/login')\n    # 可以执行页面交互（通过 Playwright）\n    page2 = session.fetch('https://example.com/dashboard')\n    data = page2.css('.metric::text').getall()\n```\n\n---\n\n## CSS 和 XPath 选择器\n\n```python\npage = Fetcher.get('https://quotes.toscrape.com/')\n\n# CSS 选择器\ntitle = page.css('h1::text').get()                    # 第一个元素\nall_texts = page.css('.quote .text::text').getall()   # 全部\n\n# XPath 选择器\nauthor = page.xpath('//small[@class=\"author\"]/text()').get()\nhrefs = page.xpath('//a/@href').getall()\n\n# 属性提取\nlink = page.css('a::attr(href)').get()\nsrc = page.css('img::attr(src)').get()\n\n# 正则提取\nprice = page.css('.price::text').re_first(r'\\$[\\d.]+')\nnumbers = page.css('p::text').re(r'\\d+')\n```\n\n---\n\n## 自适应抓取（网页改版后自动重定位）\n\n这是 Scrapling 最独特的功能：\n\n```python\nfrom scrapling.fetchers import Fetcher\n\n# 第一次抓取：保存元素快照（存入本地数据库）\npage = Fetcher.get('https://example.com/products')\nproducts = page.css('.product-card', auto_save=True)\nprint(f\"找到 {len(products)} 个商品\")\nfor p in products:\n    print(p.css('h2::text').get(), p.css('.price::text').get())\n\n# 网站改版后（.product-card 变成了 .item-wrapper）\n# 传入 adapt"},{"path":"guides/03-advanced-usage.md","content":"# 高级用法\n\n## Spider 框架（大规模并发爬取）\n\nSpider 框架提供 Scrapy 风格的 API，适合需要跨多页面爬取的任务：\n\n```python\nfrom scrapling.spiders import Spider, Request, Response\n\nclass QuotesSpider(Spider):\n    name = \"quotes\"\n    start_urls = [\"https://quotes.toscrape.com/\"]\n    concurrent_requests = 10       # 最大并发数\n    download_delay = 0.5           # 请求间隔（秒）\n    robots_txt_obey = True         # 遵守 robots.txt\n\n    async def parse(self, response: Response):\n        for quote in response.css('.quote'):\n            yield {\n                \"text\": quote.css('.text::text').get(),\n                \"author\": quote.css('.author::text').get(),\n                \"tags\": quote.css('.tag::text').getall(),\n            }\n        # 翻页\n        next_page = response.css('li.next a::attr(href)').get()\n        if next_page:\n            yield Request(response.urljoin(next_page))\n\n# 运行爬虫\nresult = QuotesSpider().start()\nresult.items.to_json('quotes.json')       # 导出 JSON\nresult.items.to_jsonl('quotes.jsonl')     # 导出 JSONL\n```\n\n---\n\n## 暂停与恢复爬虫\n\n```python\nfrom scrapling.spiders import Spider\n\nclass LargeSpider(Spider):\n    name = \"large\"\n    start_urls = [\"https://large-site.com/\"]\n    # 开启检查点持久化\n    checkpoint = True\n\n    async def parse(self, response):\n        # ...爬取逻辑...\n        pass\n\n# 启动时按 Ctrl+C 优雅停止\nresult = LargeSpider().start()\n\n# 下次运行时自动从上次停止的地方继续\nresult = LargeSpider().start()   # 无需额外配置\n```\n\n---\n\n## 流式输出（实时处理）\n\n```python\nfrom scrapling.spiders import Spider\nimport asyncio\n\nclass StreamingSpider(Spider):\n    name = \"stream\"\n    start_urls = [\"https://news-site.com/\"]\n\n    async def parse(self, response):\n        for article in response.css('article'):\n            yield {\"title\": article.css('h2::text').get()}\n\nasync def main():\n    spider = StreamingSpider()\n    async for item in spider.stream():\n        # 实时处理每个抓取到的 item\n        print(item)\n        # 实时写入数据库、发送到队列等\n\nasyncio.run(main())\n```\n\n---\n\n## 多 Session 路由（混用 HTTP 和浏览器）\n\n```python\nfrom scrapling.spiders import Spider, Request, Response\n\nclass HybridSpider(Spider):\n    name = \"hybrid\"\n    start_urls = [\"https://example.com/catalog\"]\n\n    async def parse(self, response: Response):\n        for link in response.css('.product-link::attr(href)').getall():\n            # 普通页面用 HTTP，详情页面用浏览器\n            if '/protected/' in link:\n                yield Request(link, session_id='stealthy')   # 使用 StealthyFetcher\n            else:\n                yield Request(link)                           # 使用默认 HTTP\n\n    async def parse_product(self, response: Response):\n        yield {\"title\": response.css('h1::text').get()}\n```\n\n---\n\n## 代理轮换\n\n```python\nfrom scrapling.fetchers import Fetcher, FetcherSession, StealthyFetcher\nfrom scrapling.proxy_rotator import ProxyRotator\n\n# 设置代理列表\nproxies = [\n    \"http://user:pass@proxy1.example.com:8080\",\n    \"http://user:pass@proxy2.example.com:8080\",\n    \"socks5://user:pass@proxy3.example.com:1080\",\n]\n\n# 循环轮换\nrotator = ProxyRotator(proxies, mode='cyclic')\n\nwith FetcherSession(proxy=rotator) as session:\n    page = session.get("}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":1094,"uniquenessScore":42,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T18:08:57.473Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T18:08:57.473Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T20:57:37.875Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}