{"id":"587293aa-955b-4f4a-b13f-25d97f0c5065","entityType":"agent","slug":"clawhub-jiafar-amazon-scraper","name":"amazon-scraper","canonicalUrl":"https://www.xpersona.co/agent/clawhub-jiafar-amazon-scraper","canonicalPath":"/agent/clawhub-jiafar-amazon-scraper","generatedAt":"2026-10-09T12:55:56.038Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T05:12:46.315Z","emptyReason":null},"description":"Containerized Amazon.com scraper (Docker + Playwright) for BSR/new-releases/movers rankings, keyword search results, and product detail pages. Requires a paid ISP/residential proxy. Use when the request is specifically about Amazon product or marketplace data: 亚马逊/Amazon, ASIN, BSR, Best Sellers, 畅销榜, 新品榜, 飙升榜, 选品, 竞品分析, 类目分析, listing 分析, 月销量 (bought in past month), 评分分布, 评论分析, Amazon 关键词搜索结果, Amazon 产品详情. Do NOT use for general-purpose web scraping just because the user said 爬取/抓取/采集/scrape/crawl. Every run spends metered proxy bandwidth, and the generic mode exists only as a fallback for pages related to an Amazon task. For unrelated sites prefer web_search/web_fetch, or ask first.","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 4.6K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s171dacsa9hkrd446pra6ma8sd83q3hc:amazon-scraper","sourceUrl":"https://clawhub.ai/jiafar/amazon-scraper","homepage":"https://clawhub.ai/jiafar/skills/amazon-scraper","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/jiafar/amazon-scraper","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/jiafar/skills/amazon-scraper","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":65,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"amazon-scraper technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T05:12:46.315Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T05:12:46.315Z","emptyReason":null},"stars":null,"forks":null,"downloads":4558,"packageName":null,"latestVersion":"4.0.1","tractionLabel":"4.6K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T05:12:46.315Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T05:12:46.315Z","lastCrawledAt":"2026-10-09T05:12:46.315Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T05:12:46.315Z","lastVerifiedAt":null,"highlights":[{"version":"4.0.1","createdAt":"2026-10-02T07:28:50.735Z","changelog":"OpenClaw skill format. Docker is required via SKILL.md metadata. Credentials stay out of the bundle. Scheduled monitors use openclaw automations.","fileCount":32,"zipByteSize":91906},{"version":"4.0.0","createdAt":"2026-10-02T07:25:24.089Z","changelog":"Security hardening. Proxy credentials are no longer bundled; supply them at run time. Rotate any password that shipped in 3.4.1 config/proxies.json.","fileCount":32,"zipByteSize":91509},{"version":"3.4.0","createdAt":"2026-04-22T09:55:06.599Z","changelog":"- Added: proxy support (AMAZON_PROXY / AMAZON_PROXIES env vars)\\n- Added: --output flag to save results to JSON file\\n- Added: /data volume mount for persistent output\\n- Changed: Docker image name from clawd-crawlee to amazon-scraper\\n- Changed: test script URL updated to /gp/bestsellers/electronics\\n- Updated: setup.sh with proxy usage examples and output instructions","fileCount":7,"zipByteSize":12998},{"version":"3.3.4","createdAt":"2026-04-22T05:07:56.425Z","changelog":"- Fixed: stealth plugin corrected to puppeteer-extra-plugin-stealth@^2.11.2\\n- playwright-extra-plugin-stealth only has 0.0.1 on npm; correct package is puppeteer-extra-plugin-stealth\\n- Updated require() in amazon_handler.js accordingly","fileCount":7,"zipByteSize":9492},{"version":"3.3.3","createdAt":"2026-04-22T05:04:22.551Z","changelog":"- Fixed: removed crawlee from dependencies (no longer used)\\n- Fixed: playwright-extra-plugin-stealth version corrected to ^2.11.2 (was ^0.0.1)","fileCount":7,"zipByteSize":9479},{"version":"3.3.2","createdAt":"2026-04-22T05:03:32.432Z","changelog":"- Fixed: SKILL.md 反爬能力章节更新为 stealth 插件能力描述\\n- Fixed: 移除 crawlee 残留参数 maxRetries，冷启动时间更新为 15 秒","fileCount":7,"zipByteSize":9480},{"version":"3.3.1","createdAt":"2026-04-22T05:02:32.331Z","changelog":"- Fixed: SKILL.md description updated from Crawlee to playwright-extra + Stealth\\n- Fixed: package.json description updated to reflect stealth capability\\n- No code changes from 3.3.0","fileCount":7,"zipByteSize":9384},{"version":"3.3.0","createdAt":"2026-04-22T05:01:33.519Z","changelog":"- Changed: replaced crawlee PlaywrightCrawler with playwright-extra + stealth plugin\\n- Added: stealth mode to bypass Amazon headless detection\\n- Added: viewport 1920x1080 and Chrome 120 userAgent\\n- Unchanged: all data extraction logic (bestsellers/search/product-detail/generic)\\n- Unchanged: output format and ASIN deduplication\\n- Updated: package.json dependencies (playwright-extra + playwright-extra-plugin-stealth)\\n- Updated: Dockerfile CMD now points to amazon_handler.js","fileCount":7,"zipByteSize":9310}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s171dacsa9hkrd446pra6ma8sd83q3hc:amazon-scraper","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s171dacsa9hkrd446pra6ma8sd83q3hc:amazon-scraper` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/jiafar/amazon-scraper before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-jiafar-amazon-scraper/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-jiafar-amazon-scraper/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-jiafar-amazon-scraper/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-jiafar-amazon-scraper/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-jiafar-amazon-scraper/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-jiafar-amazon-scraper/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-09T12:55:56.035Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-jiafar-amazon-scraper/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-jiafar-amazon-scraper/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-jiafar-amazon-scraper/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-jiafar-amazon-scraper/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T05:12:46.315Z","emptyReason":null},"readme":"Skill: amazon-scraper\n\nOwner: jiafar\n\nSummary: Containerized Amazon.com scraper (Docker + Playwright) for BSR/new-releases/movers rankings, keyword search results, and product detail pages. Requires a paid ISP/residential proxy. Use when the request is specifically about Amazon product or marketplace data: 亚马逊/Amazon, ASIN, BSR, Best Sellers, 畅销榜, 新品榜, 飙升榜, 选品, 竞品分析, 类目分析, listing 分析, 月销量 (bought in past month), 评分分布, 评论分析, Amazon 关键词搜索结果, Amazon 产品详情. Do NOT use for general-purpose web scraping just because the user said 爬取/抓取/采集/scrape/crawl. Every run spends metered proxy bandwidth, and the generic mode exists only as a fallback for pages related to an Amazon task. For unrelated sites prefer web_search/web_fetch, or ask first.\n\nTags: latest:4.0.1\n\nVersion history:\n\nv4.0.1 | 2026-10-02T07:28:50.735Z | user\n\nOpenClaw skill format. Docker is required via SKILL.md metadata. Credentials stay out of the bundle. Scheduled monitors use openclaw automations.\n\nv4.0.0 | 2026-10-02T07:25:24.089Z | user\n\nSecurity hardening. Proxy credentials are no longer bundled; supply them at run time. Rotate any password that shipped in 3.4.1 config/proxies.json.\n\nv3.4.0 | 2026-04-22T09:55:06.599Z | user\n\n- Added: proxy support (AMAZON_PROXY / AMAZON_PROXIES env vars)\\n- Added: --output flag to save results to JSON file\\n- Added: /data volume mount for persistent output\\n- Changed: Docker image name from clawd-crawlee to amazon-scraper\\n- Changed: test script URL updated to /gp/bestsellers/electronics\\n- Updated: setup.sh with proxy usage examples and output instructions\n\nv3.3.4 | 2026-04-22T05:07:56.425Z | user\n\n- Fixed: stealth plugin corrected to puppeteer-extra-plugin-stealth@^2.11.2\\n- playwright-extra-plugin-stealth only has 0.0.1 on npm; correct package is puppeteer-extra-plugin-stealth\\n- Updated require() in amazon_handler.js accordingly\n\nv3.3.3 | 2026-04-22T05:04:22.551Z | user\n\n- Fixed: removed crawlee from dependencies (no longer used)\\n- Fixed: playwright-extra-plugin-stealth version corrected to ^2.11.2 (was ^0.0.1)\n\nv3.3.2 | 2026-04-22T05:03:32.432Z | user\n\n- Fixed: SKILL.md 反爬能力章节更新为 stealth 插件能力描述\\n- Fixed: 移除 crawlee 残留参数 maxRetries，冷启动时间更新为 15 秒\n\nv3.3.1 | 2026-04-22T05:02:32.331Z | user\n\n- Fixed: SKILL.md description updated from Crawlee to playwright-extra + Stealth\\n- Fixed: package.json description updated to reflect stealth capability\\n- No code changes from 3.3.0\n\nv3.3.0 | 2026-04-22T05:01:33.519Z | user\n\n- Changed: replaced crawlee PlaywrightCrawler with playwright-extra + stealth plugin\\n- Added: stealth mode to bypass Amazon headless detection\\n- Added: viewport 1920x1080 and Chrome 120 userAgent\\n- Unchanged: all data extraction logic (bestsellers/search/product-detail/generic)\\n- Unchanged: output format and ASIN deduplication\\n- Updated: package.json dependencies (playwright-extra + playwright-extra-plugin-stealth)\\n- Updated: Dockerfile CMD now points to amazon_handler.js\n\nv3.2.0 | 2026-04-22T04:20:37.963Z | user\n\n- Fixed: Dockerfile renamed to Dockerfile.sh so clawhub server bundles it correctly\\n- clawhub server filters out files without text extensions; .sh extension bypasses this\\n- setup.sh now copies Dockerfile.sh -> Dockerfile before docker build\\n- Users can now run setup.sh after install and docker build will work\n\nv3.1.9 | 2026-04-22T03:43:01.638Z | user\n\n- Correct release (replaces 3.1.6/3.1.7/3.1.8 which had wrong Dockerfile)\\n- Dockerfile: playwright:v1.40.0-jammy, full chromium install\\n- setup.sh: simple one-click docker build\\n- All files match the verified reference bundle\n\nv3.1.8 | 2026-04-22T03:42:00.990Z | user\n\n- Synced Dockerfile and setup.sh to reference version\\n- Dockerfile: playwright:v1.40.0-jammy, full install with chromium\\n- setup.sh: simplified one-click docker build\n\nv3.1.7 | 2026-04-22T03:40:45.023Z | user\n\n- Fixed: setup.sh now auto-generates Dockerfile if missing (clawhub cannot bundle files without extensions)\\n- setup.sh is now fully self-contained: no dependency on Dockerfile being present in the zip\n\nv3.1.6 | 2026-04-22T03:38:57.583Z | user\n\n- Stable release: Dockerfile confirmed included in release bundle\\n- All previous fixes from v3.1.x consolidated\\n- Ready for production use\n\nv3.1.5 | 2026-04-22T03:01:12.927Z | user\n\n- Fixed: Added .clawhubignore to ensure Dockerfile is included in release bundle\\n- Fixed: .dockerignore cleaned up (removed SKILL.md/scripts exclusions that don't belong there)\\n- setup.sh one-click build now works correctly after download\n\nv3.1.4 | 2026-04-22T02:47:13.202Z | user\n\n- Fixed: package.json main entry changed from main_handler.js to amazon_handler.js\n\nv3.1.3 | 2026-04-22T02:45:43.678Z | user\n\n- Removed: youtube_handler.js (deleted from assets)\\n- Fixed: package.json description and keywords no longer mention YouTube\\n- Fixed: SKILL.md subtitle and generic mode trigger no longer mention YouTube/social media\\n- Skill is now 100% Amazon-focused with generic fallback only\n\nv3.1.2 | 2026-04-22T02:44:40.076Z | user\n\n- Removed: YouTube mode from main_handler.js (out of scope)\\n- Removed: YouTube/TikTok/Twitter/X trigger keywords from SKILL.md\\n- Fixed: main_handler.js comment header renamed from 'Deep-Scraper' to 'Amazon-Scraper'\\n- Fixed: Dockerfile CMD now points to amazon_handler.js as primary entry\\n- Fixed: SKILL.md frontmatter description now leads with Amazon use case\\n- Fixed: Decision tree cleaned up (no more YouTube branch)\n\nv3.1.1 | 2026-04-22T02:33:21.987Z | user\n\n- Fixed: package.json name corrected from 'deep-scraper' to 'amazon-scraper'\\n- Fixed: description updated to highlight Amazon as primary use case\\n- Fixed: added requiredBinaries docker to registry metadata\\n- Added: scripts/setup.sh for one-click Docker image build\\n- Added: .dockerignore to reduce build context\\n- Added: System requirements section in SKILL.md\\n- Updated: SKILL.md installation steps to use setup.sh\\n- Bumped version to 3.1.1\n\nv3.1.0 | 2026-04-14T11:56:31.396Z | auto\n\namazon-scraper 3.1.0\n\n- Added batch scraping support with the new `scripts/batch-scrape.sh` script, enabling efficient multi-ASIN processing and file-based result output.\n- Expanded documentation: clarified batch and file output workflows, added best practices for large-scale/batch analysis, and described browser-based manual extraction methods.\n- Updated output management section: strongly recommends direct-to-file output to avoid stdout mixing in multi-process scenarios.\n- Enhanced guidance on DOM selectors and inline JS scripts for browser-assisted extraction.\n- Minor optimizations and clarifications throughout documentation for a streamlined user experience.\n\nv3.0.6 | 2026-04-11T07:39:09.675Z | user\n\nAdd recommended volume mount pattern for writing output to local files\n\nv3.0.5 | 2026-04-11T07:36:55.421Z | user\n\nFix: upgrade Playwright base image from v1.52.0 to v1.59.1 to fix Chromium executable error\n\nv3.0.4 | 2026-04-11T07:00:21.318Z | user\n\nAdd local deployment guide for Mac/Linux/Windows\n\nv3.0.3 | 2026-04-11T06:59:53.436Z | user\n\nAdd VPS installation guide\n\nv3.0.2 | 2026-04-11T06:58:36.610Z | user\n\nRemove platform-specific paths\n\nv3.0.1 | 2026-04-11T06:51:33.248Z | user\n\nBug fixes\n\nv3.0.0 | 2026-03-03T00:29:37.180Z | user\n\nAmazon structured data scraper. Extracts ASIN, price, rating, reviews, boughtPastMonth from Best Sellers, search results, and product detail pages. Also supports YouTube transcripts and generic page scraping. Docker + Playwright.\n\nv1.0.0 | 2026-02-28T10:41:12.790Z | user\n\nInitial release: Playwright-based Amazon product scraper with anti-bot evasion, natural pagination, and stealth browsing\n\nArchive index:\n\nArchive v4.0.1: 32 files, 91906 bytes\n\nFiles: assets/amazon_handler.js (28708b), assets/fingerprint.js (6399b), assets/main_handler.js (4284b), assets/proxy.js (3264b), CHANGELOG.md (7398b), config/proxies.example.json (307b), Dockerfile.sh (1682b), package.json (634b), references/batch-parallel-scraping.md (3831b), references/bsr-top100-strategy.md (3378b), references/cdp-fallback-strategy.md (6436b), references/cron-price-monitor-pattern.md (3372b), references/full-extract-mode.md (3767b), references/new-layout-fixes-2026-07.md (4077b), references/oxylabs-ddc-format.md (4308b), references/oxylabs-product-matrix.md (2987b), references/oxylabs-proxy-format.md (3326b), references/product-form-fusion.md (3323b), references/reviews-strategy.md (13443b), references/search-page-title-fix.md (3234b), references/sharing-checklist.md (3084b), scripts/asin_monitor.py (22264b), scripts/asin_price_monitor.py (3896b), scripts/cdp_fallback_scrape.py (9804b), scripts/full_extract.js (17497b), scripts/pickleball_price_monitor.py (6824b), scripts/scrape_reviews.py (9436b), scripts/scraper_client.py (2291b), scripts/setup.sh (1894b), skill-card.md (2322b), SKILL.md (19955b), _meta.json (133b)\n\nFile v4.0.1:SKILL.md\n\n---\nname: amazon-scraper\ndescription: >\n  Containerized Amazon.com scraper (Docker + Playwright) for BSR/new-releases/movers rankings,\n  keyword search results, and product detail pages. Requires a paid ISP/residential proxy.\n\n  Use when the request is specifically about Amazon product or marketplace data:\n  亚马逊/Amazon, ASIN, BSR, Best Sellers, 畅销榜, 新品榜, 飙升榜,\n  选品, 竞品分析, 类目分析, listing 分析, 月销量 (bought in past month),\n  评分分布, 评论分析, Amazon 关键词搜索结果, Amazon 产品详情.\n\n  Do NOT use for general-purpose web scraping just because the user said\n  爬取/抓取/采集/scrape/crawl. Every run spends metered proxy bandwidth, and the\n  generic mode exists only as a fallback for pages related to an Amazon task.\n  For unrelated sites prefer web_search/web_fetch, or ask first.\nmetadata:\n  openclaw:\n    requires:\n      bins:\n        - docker\n---\n\n# Amazon Scraper\n\nDocker 容器化爬虫，Playwright Chromium。不使用 stealth 插件。支持亚马逊榜单/搜索/详情及通用动态页。\n\n## 第 0 步（开跑前必须先做，不过就停）\n\n先确认出口，再碰亚马逊：\n\n```bash\nexport AMAZON_PROXIES=\"http://USER:PASS@HOST:PORT\"   # 从密码管理器取，不要贴进对话\ncurl -s -x \"$AMAZON_PROXIES\" http://api.ipify.org\n```\n\n返回的纯文本必须等于你配置的代理 host。不是这个 IP、超时、407、403，都停，不要开爬，也不要重建后直接跑。\n\n- **凭证不进仓库、不进镜像、不进命令回显。** 用 `-e AMAZON_PROXIES`（值从环境取，不要写在命令行里）或挂载 `config/proxies.json`。该文件已在 `.gitignore` / `.dockerignore` 里。\n- 这一步只证明代理层通。裸 curl 打亚马逊拿到 500/202，不算第 0 步失败。\n- 第 0 步过了，才允许 `docker run amazon-scraper`。\n\n## 系统要求\n\n- **Docker Engine 20.10+**（必须已安装并运行）\n- **磁盘空间**：~2GB（镜像 + Playwright 浏览器二进制文件）\n- **内存**：建议 2GB+（Playwright 运行时需要）\n\n## 快速开始\n\n首次使用：在 skill 目录下执行一键构建脚本：\n\n```bash\nbash scripts/setup.sh\n```\n\n脚本会自动完成：构建 `amazon-scraper` 镜像 + 创建 `~/scrapes` 输出目录。\n\n\n## 模式选择规则\n\n### 1. Amazon模式 (`amazon_handler.js`)\n**自动触发条件:** URL包含 `amazon.com`，或用户提到亚马逊/Amazon/ASIN/BSR/选品/竞品/畅销榜/类目分析等关键词\n\n根据URL自动识别页面类型：\n\n| URL特征 | 页面类型 | 可获取字段 |\n|---|---|---|\n| `/gp/bestsellers/` | 畅销榜 | rank, title, asin, price, rating, reviews, image, url |\n| `/zg/new-releases/` | 新品榜 | 同上 |\n| `/zg/movers-and-shakers/` | 飙升榜 | 同上 |\n| `/s?k=` 或 `/s/` | 搜索结果 | title, asin, price, rating, reviews, image, url, **boughtPastMonth**, sponsored |\n| `/dp/` 或 `/gp/product/` | 产品详情 | title, asin, price, rating, reviews, brand, bsr, **boughtPastMonth**, **seller**, dateFirstAvailable, category, bullets, details |\n\n**⚠️ 重要规则:**\n- **Best Sellers 页面没有月销量(boughtPastMonth)数据** — 亚马逊不在榜单页显示此信息\n- **要获取月销量，必须用搜索页(`/s?k=关键词`)或产品详情页(`/dp/ASIN`)**\n- 如果用户同时需要排名+月销量，建议：先爬 Best Sellers 拿排名，再用搜索页补月销\n- **BSR URL 必须使用 `/gp/bestsellers/`**，`/zgbs/` 会返回 Page Not Found\n- **BSR 单 URL 只能拿 60 个产品**（2 页限制）。拿 Top 100 用搜索页 `--pages 5` 或多子类目 BSR 合并。详见 `references/bsr-top100-strategy.md`\n- **评论（rating/title/body/date/helpful/verified）不在这张表里**：需要登录态 + CDP，走 `scripts/scrape_reviews.py`（`/portal/customer-reviews/ASIN` 瀑布流）。先读 `references/reviews-strategy.md` 顶部的账号/合规代价说明\n- 视觉化选品 fallback：把 `image` URL 喂给 `vision_analyze`，让它\"看图识物\"。参考 `references/product-form-fusion.md`\n\n```bash\n# 畅销榜（有排名，无月销，最多 60 个）\ndocker run --rm amazon-scraper node assets/amazon_handler.js \"https://www.amazon.com/gp/bestsellers/electronics\"\n\n# 搜索结果（有月销，多页可达 100+）\ndocker run --rm amazon-scraper node assets/amazon_handler.js \"https://www.amazon.com/s?k=feather+duster\"\n\n# 产品详情（最全字段：BSR、品牌、卖点、月销）\ndocker run --rm amazon-scraper node assets/amazon_handler.js \"https://www.amazon.com/dp/B001TQ6IHS\"\n\n# 多页爬取（搜索页建议 5 页拿 Top 100）\ndocker run --rm amazon-scraper node assets/amazon_handler.js \"URL\" --pages 2\n\n# 保存结果到文件（必须挂载 /data 才能在主机读到）\ndocker run --rm -v ~/scrapes:/data amazon-scraper node assets/amazon_handler.js \"URL\" --output result.json\n\n# 用自己的代理覆盖内置配置\ndocker run --rm -e AMAZON_PROXIES=\"http://user:***@host:8001,...\" amazon-scraper node assets/amazon_handler.js \"URL\"\n```\n\n**输出状态（必须检查，不要只看 products 长度）**\n\n| `status` | 退出码 | 含义 | 该怎么办 |\n|---|---|---|---|\n| `SUCCESS` | 0 | 所有页都抓到了 | 正常使用数据 |\n| `PARTIAL` | 3 | 部分页被拦/超时，`failures[]` 列出来了 | 可以用已拿到的，但**不要**把缺的那部分当成\"不存在\" |\n| `ERROR` | 2 / 1 | 全部失败 | 丢弃结果。先查第 0 步出口，再看是不是被软拦截 |\n\n软拦截的典型表现是 HTTP 200 + 整页没渲染，于是所有字段都是 null。现在详情页遇到这种情况会算失败并换代理重试，不会再以 `SUCCESS` 混进结果。**写入数据库/生成报表前先判 `status`** —— 把软拦截当成\"价格没变\"比没有数据更糟。\n\n**输出格式:** JSON\n```json\n{\n  \"status\": \"SUCCESS\",\n  \"failedJobs\": [],\n  \"failures\": [],\n  \"type\": \"bestsellers|search|product-detail\",\n  \"category\": \"品类名\",\n  \"totalProducts\": 30,\n  \"scrapedAt\": \"ISO时间\",\n  \"products\": [\n    {\n      \"rank\": 1,\n      \"title\": \"产品名\",\n      \"asin\": \"B001TQ6IHS\",\n      \"price\": 9.94,\n      \"priceStr\": \"$9.94\",\n      \"rating\": 4.6,\n      \"reviews\": 20547,\n      \"boughtPastMonth\": \"1K+\",\n      \"image\": \"https://...\",\n      \"url\": \"https://...\",\n      \"sponsored\": false\n    }\n  ]\n}\n```\n\n### 2. 通用模式 (`main_handler.js`)\n**触发条件:** 非Amazon的URL，或用户提到爬取/抓取任意网页内容\n\n- 和 Amazon 模式同一套 Playwright（无 stealth）\n- 内置代理已预配置，无需额外设置\n- 支持 `--output` 文件保存\n- 可通过环境变量覆盖内置代理\n- Playwright打开页面，等待JS加载完成\n- 提取 `document.body.innerText`（纯文本，去广告噪音）\n- 输出上限10000字符\n- 输出: `{status:\"SUCCESS\", type:\"GENERIC\", title, data}`\n\n```bash\n# 通用爬取（代理已内置）\ndocker run --rm amazon-scraper node assets/main_handler.js \"https://任意网址\"\n\n# 保存文件\ndocker run --rm -v ~/scrapes:/data \\\n  amazon-scraper node assets/main_handler.js \"https://任意网址\" --output page.json\n```\n\n## Agent调用决策树\n\n```\n用户给了URL?\n├─ 包含 amazon.com → 用 amazon_handler.js\n│   ├─ 需要月销量? → 建议用搜索URL(/s?k=) 或详情页(/dp/)\n│   ├─ 需要排名? → 用畅销榜URL(/gp/bestsellers/)\n│   └─ 需要 Top 100? → 搜索页 --pages 5 (或看 references/bsr-top100-strategy.md)\n└─ 其他网站 → 用 main_handler.js (通用模式)\n\n用户没给URL，只说了需求?\n├─ \"爬亚马逊XX品类Top\" / \"XX类目排行\" / \"XX畅销榜\" → 构造 https://www.amazon.com/gp/bestsellers/品类\n├─ \"搜亚马逊XX\" / \"XX关键词搜索\" / \"找XX产品\" → 构造 https://www.amazon.com/s?k=关键词\n├─ \"分析某个ASIN\" / \"看看这个产品\" / \"XX的详情\" → 构造 https://www.amazon.com/dp/ASIN\n├─ \"XX的月销量\" / \"XX卖了多少\" / \"XX销量怎么样\" → 用搜索页或详情页（有boughtPastMonth）\n├─ \"竞品分析\" / \"竞品调研\" / \"对手在卖什么\" → 先搜索再逐个爬详情\n├─ \"选品\" / \"什么好卖\" / \"品类机会\" / \"市场调研\" → Best Sellers + 搜索结合\n├─ \"Top 100\" / \"Top 200\" / \"拿 100 个产品\" → /s?k=关键词 --pages 5 (一次拿 100+)\n├─ \"抓评论\" / \"获取评价\" / \"reviews\" → 需登录态 Chrome + CDP，用 scripts/scrape_reviews.py（瀑布流模式）\n└─ 其他网页 → 先web_search找到URL，再用通用模式爬\n```\n\n## 常见用户意图 → 操作映射\n\n| 用户说 | 操作 |\n|---|---|\n| \"帮我看看亚马逊XX品类\" | 爬 /gp/bestsellers/品类 畅销榜 |\n| \"XX在亚马逊卖得怎么样\" | 搜索 /s?k=XX 看月销 |\n| \"分析一下这个ASIN: BXXXXXXXXX\" | 爬 /dp/ASIN 详情页 |\n| \"XX品类有什么机会\" | 畅销榜 + 搜索 综合分析 |\n| \"帮我爬这个链接\" | 判断URL类型，选对应handler |\n| \"帮我抓XX网站的内容\" | 通用模式 |\n| \"搜一下XX的竞品\" | 搜索页爬取 + 分析 |\n| \"XX月销多少\" / \"XX一个月卖多少\" | 搜索页或详情页 |\n| \"帮我看看top 100\" / \"热门产品\" | **/s?k=关键词 --pages 5**（BSR 单 URL 只到 60） |\n| \"新品有哪些\" / \"最近上了什么新品\" | /zg/new-releases/ |\n| \"什么产品涨得快\" / \"飙升榜\" | /zg/movers-and-shakers/ |\n| \"抓评论\" / \"获取评价\" / \"XX的评论\" | **scripts/scrape_reviews.py**（需已登录 Chrome + CDP，瀑布流模式） |\n\n## 代理配置\n\n代理配置存放于 `config/proxies.json`，格式为 JSON 数组：\n\n```json\n{\n  \"proxies\": [\n    \"http://user-XXX:password@ENTRY_POINT:PORT\",\n    \"http://user-XXX:password@ENTRY_POINT:PORT\"\n  ]\n}\n```\n\n优先级（`assets/proxy.js`，环境变量优先）：\n1. `AMAZON_PROXIES`（逗号分隔多条）\n2. `AMAZON_PROXY`（单条）\n3. `AMAZON_PROXY_FILE` 指向的文件，否则 `config/proxies.json`\n\n一条都没配就**直接报错退出**，不会悄悄用宿主机 IP 去爬（真要这么做得显式 `AMAZON_ALLOW_DIRECT=1`）。日志只打 `host:port`，不打凭证。\n\n凭证**不再烘进镜像**（`Dockerfile` 不 `COPY config/`）。运行时二选一：\n\n```bash\ndocker run --rm -e AMAZON_PROXIES amazon-scraper node assets/amazon_handler.js \"URL\"\ndocker run --rm -v \"$PWD/config/proxies.json:/app/config/proxies.json:ro\" amazon-scraper node assets/amazon_handler.js \"URL\"\n```\n\n改凭证不需要重 build，只有改 `assets/` 才需要。协议只接受 `http://` / `https://`，`socks5://` 会在启动时报错。\n\n并发被代理条数卡死：`concurrency = min(请求值, 任务数, proxies.length)`。只有 1 条时 `--concurrency 5` 实际是 1，并且现在会打一行 WARNING 说明被降到了几。不要开 5 个容器打同一个 IP。\n\n### 入口域名速查（Oxylabs 各产品）\n\n| 产品 | 入口域名 | Amazon 适用? |\n|---|---|---|\n| Datacenter / DDC | `ddc.oxylabs.io` | ❌（被 Amazon ASN 黑名单） |\n| **ISP Proxies** ⭐ | `isn.oxylabs.io` 或 `pr.oxylabs.io` | ✅ **推荐** |\n| Residential | `pr.oxylabs.io` | ✅ |\n| Mobile | `mob.oxylabs.io` | ✅（最严目标） |\n\n> 📖 **完整产品矩阵 / 凭证格式 / 怎么从 dashboard 拿入口域名**：[references/oxylabs-product-matrix.md](references/oxylabs-product-matrix.md)\n> 📖 **DDC 故障排查 / response code 速查**：[references/oxylabs-ddc-format.md](references/oxylabs-ddc-format.md)\n> 📖 **Oxylabs Response Code + 怎么判断\"代理层通 vs 目标站拦\"**：[references/oxylabs-proxy-format.md](references/oxylabs-proxy-format.md)\n\n### ⚠️ 代理配置陷阱（来自真实踩坑）\n\n1. **不要把 dashboard 上的 \"Assigned IP\" 当入口**：那是出口 IP，不是入口。直接打 `45.x.x.x:8001` 在 VPS 上会 `No route to host`。**必须用 `ddc.oxylabs.io` 域名 + 端口**。详见 `references/oxylabs-ddc-format.md`。\n\n2. **DDC / Datacenter 代理对 Amazon 不可用**：Amazon 把整个 datacenter ASN 段都标为高风险，配置 100% 正确也会拿到 \"Sorry! Something went wrong!\" 软拦截页。**Amazon 爬取用 ISP Proxies**，不要用 DDC。详见 `references/oxylabs-product-matrix.md`。\n\n3. **凭证不区分协议前缀**：`proxies.json` 里 `http://` 和 `https://` 都能用（爬虫内部走 HTTP CONNECT）。别误用 `socks5://`。\n\n4. **凭证过期/被封 = 静默 0 results**：如果 Amazon 返回 `totalProducts: 0` 且无明显错误，第一反应是代理被 Amazon 软拦截（datacenter）或凭证挂了。先单独跑 `curl -x ddc.oxylabs.io:PORT -U user:pass http://ip.oxylabs.io/location` 验证代理层通不通。\n\n5. **改 `proxies.json` 不再需要重 build**：`Dockerfile` 已经不 `COPY config/`，凭证走 `-e AMAZON_PROXIES` 或挂载 `/app/config/proxies.json`。改 `assets/` 下的代码仍然要 `docker build -t amazon-scraper <skill 目录>`。\n6. **`--output` 是容器内路径**：必须 `-v /host/path:/data` 挂载，否则文件在容器内拿不到。\n7. **`Accept-Encoding: identity` 头导致详情页加载失败**：原 `createContext` 的 `extraHTTPHeaders` 和 `setExtraHTTPHeaders` 里设了 `Accept-Encoding: identity`，会导致 Amazon 新版详情页 JS 不渲染（`#productTitle` 等 selector 全部 timeout）。已移除该头和整个 `setExtraHTTPHeaders` 调用，详情页恢复正常。\n8. **新版 Amazon 详情页 DOM 结构变化**（2026年7月实测）：\n   - BSR 不在 `body.innerText` 正则里了，在 `#prodDetails tr` 的 `th/td` 行里（key=\"Best Sellers Rank\", val=\"#84 in Electronics...\"）\n   - brand 从 `#bylineInfo` 拿到的是 \"Brand: XXX\" 或 \"Visit the XXX Store\"，需清理前缀和后缀\n   - seller 从 `#sellerProfileTriggerId` 提取（如 \"BeataTap-SKALON\"），但部分 ASIN 的 `#sellerProfileTriggerId` 会抓到 Amazon 比价组件的垃圾文本（\"different sellers.\\nShow details...\"），需在分析层过滤\n   - `dateFirstAvailable` 在新版页面 **100% 缺失**（Amazon 不再展示此字段），详情页正则和 `#prodDetails tr` 都拿不到\n   - 详情页需要 `waitForSelector('#productTitle, #dp', { timeout: 15000 })` + `waitForTimeout(3000)` 才能拿到完整数据\n9. **datacenter 代理对 Amazon 详情页 100% 软拦截**（返回空页面），ISP 代理正常。用 `isp.oxylabs.io` 这类 ISP/住宅入口，不要用 `ddc.*` / `disp.*`。\n10. **搜索页 `--pages 5` 实际只返回 ~26 个唯一 ASIN**（不是 100）：Amazon 搜索结果含大量广告/重复，拿真正 Top 100 需要更多页或关键词变体。\n11. **搜索页翻页参数是 `&page=N` 不是 `&pg=N`**（2026-07-04 修复）：原代码用 `&pg=N` 翻页，Amazon 不认这个参数，每页都返回第一页数据（全是重复），所以 `--pages 10` 只拿到 29 个唯一 ASIN。改成 `&page=N` 后 10 页拿到 234 个唯一 ASIN。如果发现多页爬取结果去重后远少于预期，先检查翻页参数。\n12. **批量只能按当前代理条数来**：并发被代码卡成 `min(请求值, 任务数, 代理条数)`，只有 1 条出口时开 5 个容器就是 5 个浏览器打同一个 IP。超过约 10 分钟的批量必须后台跑（前台 600s 会杀掉容器）。要并行，先加不同出口 IP，再给每个容器 `-e AMAZON_PROXIES` 指定各自的出口。合并按 `_source`，不按返回的 `asin`。\n13. **CDP Fallback 补漏方案（2026-07-04 验证）**：即使安全加速模式仍有 ~25% ASIN 被软拦截返回 null。**解法**：用本地 Chrome CDP（port 9222）串行补爬失败的 ASIN，成功率 100%。脚本 `scripts/cdp_fallback_scrape.py`，详见 `references/cdp-fallback-strategy.md`。完整流程：Docker 批量（快但有失败）→ 提取失败 ASIN → CDP 串行补漏（慢但 100% 成功）→ 合并结果。\n\n## 反爬能力\n- **出口**：按配置的代理条数轮询；失败换下一条；页间隔 1.5s\n- **UA**：Linux Chrome 119，对齐 Playwright 1.40 内核（不再写 Mac/Win）\n- **端口绑死**：分辨率 / 时区 / 核数；cookie 落 `/data/fingerprint-state/{namespace}/{port}.json`（需 `-v ~/scrapes:/data`）。通用模式按目标域名分 namespace，不和亚马逊共用一个 cookie 罐\n- ⚠️ `BY_PORT` 的指纹表只覆盖端口 8001–8005。代理端口不在这个范围时会走 host 哈希的 fallback，所谓\"同端口同指纹\"实际没生效 —— 换代理商后要同步更新这张表\n- **先开首页再跳目标**，同一 page 带 referer\n- **不授权 geolocation**\n- **无 stealth 插件**（特征库公开，已关掉做 A/B）\n- **WebRTC**：`--force-webrtc-ip-handling-policy=disable_non_proxied_udp`，API 在，不走宿主机 UDP\n- 轻量 `mouse.move`，不是拟人鼠标库\n- Docker 用完即毁；`--no-sandbox` 仍在（容器必需）\n\n这不是红手指分控箱：没有一人一号、没有真实字体/显卡。\n\n## 局限\n- 通用模式输出上限10000字符\n- Amazon BSR 单 URL 最多 60 个产品（2 页限制），拿 Top 100 用搜索页\n- Amazon 单页搜索结果最多约 30-50 个产品\n- 不支持需要登录的页面\n- Docker 容器启动有冷启动时间（Playwright Chromium）\n- Amazon 对 datacenter 代理段主动风控，**Amazon 爬取必须用 ISP / Residential / Mobile 代理**\n- **评论抓取不在 Docker 内**：需已登录的 Chrome + CDP（`scripts/scrape_reviews.py`），非匿名页面，详见 `references/reviews-strategy.md`\n- **ASIN 详情页返回所有字段为 null**：`status: SUCCESS` + `products[0]` 存在但 `title`/`price`/`rating` 全为 null。两种原因：(1) ASIN 已下架/不可用（换一个有效 ASIN 验证即可区分）；(2) 代理被亚马逊软拦截（503/captcha 页）。用 `curl -x <proxy> -s -o /dev/null -w \"%{http_code}\" \"https://www.amazon.com/dp/<ASIN>\"` 检查，正常应返回 200，503 = 代理被拦\n\n## 相关 references\n- `references/new-layout-fixes-2026-07.md` — Amazon 新版页面 DOM 变化 + 修复详情（seller 字段、BSR 提取、brand 清理、proxy 切换）\n- `references/oxylabs-product-matrix.md` — 哪个 Oxylabs 产品对应哪个入口域名，Amazon 该买哪个\n- `references/oxylabs-ddc-format.md` — DDC 故障排查、response code 速查、\"代理通 vs 目标拦\"判别\n- `references/oxylabs-proxy-format.md` — 通用 Oxylabs 代理格式 + 验证命令\n- `references/bsr-top100-strategy.md` — BSR 60 个上限 + 拿 Top 100 的三种策略\n- `references/search-page-title-fix.md` — 搜索页 `title: null` 根因 + 已合入的修复 + 重 build 步骤\n- `references/product-form-fusion.md` — 视觉化选品 / 产品形态融合创意（vision_analyze fallback）\n- `references/sharing-checklist.md` — 打包分享前的凭证安全检查\n- `references/reviews-strategy.md` — 评论页抓取策略（瀑布流模式 + CDP page WS + 选择器参考）\n- `references/batch-parallel-scraping.md` — 批量详情页并行爬取模式（5 容器 × 独立代理端口 × 低并发）\n- `scripts/scrape_reviews.py` — 可直接运行的评论抓取脚本（CDP + 瀑布流 + SQLite 存储）\n- `scripts/asin_price_monitor.py` — 每日竞品 ASIN 价格监控脚本，用 `openclaw automations` 的 command 定时跑，不另起一轮模型。stdout 可推到 Discord/Telegram。含价格变化检测。修改 ASINS 列表增删监控目标\n- `scripts/cdp_fallback_scrape.py` — CDP 补漏脚本：当 Docker 代理批量爬取被软拦截时，用本地 Chrome CDP 串行补爬失败 ASIN，成功率 100%。详见 `references/cdp-fallback-strategy.md`\n- `scripts/pickleball_price_monitor.py` — 多 ASIN 价格监控模板（Docker 爬虫 + SQLite + 飞书多维表格 + Discord 推送），用 OpenClaw automation 定时跑\n- `references/cron-price-monitor-pattern.md` — 价格监控定时任务：OpenClaw automations + 飞书多维表格 + Discord，含时区和 lark-base 字段类型注意事项\n- `references/cdp-fallback-strategy.md` — CDP Fallback 完整策略：前置条件、使用方法、Docker+CDP 两阶段工作流\n\nFile v4.0.1:_meta.json\n\n{\n  \"ownerId\": \"kn7ewmmms6dthpk632rzrc05v981ah4f\",\n  \"slug\": \"amazon-scraper\",\n  \"version\": \"4.0.1\",\n  \"publishedAt\": 1790926130735\n}\n\nFile v4.0.1:references/batch-parallel-scraping.md\n\n# Batch Parallel Scraping Pattern\n\n**当前不适用。** 活代理只有 1 条（出口 IP 和端口见你本地的 `config/proxies.json`，不要写进文档）。代码把并发卡成 `min(请求值, 任务数, proxies.length)`，所以 `--concurrency 5` 仍是 1。`-e AMAZON_PROXIES` 在 `proxies.json` 非空时不生效。下面的 5 容器 × Oxylabs 8001–8005 是旧方案，现在执行会让 5 个浏览器打同一个 IP。\n\n大批量在只有 1 条 IP 时：一个容器，`--asins`，`--output` 必须配 `-v 宿主机目录:/data`，超过约 10 分钟改后台。先用 `http://api.ipify.org` 确认出口 IP 等于你配置的代理 host，再跑亚马逊。裸 curl 拿到 500/202 不算爬虫失败，以 handler 的 JSON 为准。\n\n旧方案（要有 5 个不同出口才能用，且必须挂载覆盖 `/app/config/proxies.json`，不能靠环境变量）：\n\nSafe high-throughput detail page scraping using multiple Docker containers with dedicated proxy ports.\n\n## Problem\nSingle container `--asins \"100_asins\" --concurrency 5` is slow (~20 min for 100 ASINs). But naive parallelism (5 containers × concurrency 5 = 25 total concurrent requests) causes ISP proxy soft-blocks — 76% of ASINs return all-null data.\n\n## Safe Pattern: 5 Containers × 2-3 Concurrency\n\n```bash\nPROXY_USER=\"user-XXX\"\nPROXY_PASS=\"XXX\"\n\n# Split 100 ASINs into 5 batches of 20\n# Each batch → its own Docker container with a dedicated proxy port\n\nfor i in 0 1 2 3 4; do\n  PORT=$((8001 + i))\n  BATCH=\"<20 ASINs comma-separated>\"\n  docker run --rm -v ~/scrapes:/data \\\n    -e AMAZON_PROXIES=\"http://${PROXY_USER}:${PROXY_PASS}@isp.oxylabs.io:${PORT}\" \\\n    amazon-scraper node assets/amazon_handler.js \\\n    --asins \"$BATCH\" --concurrency 2 \\\n    --output \"batch_${i}.json\" \\\n    > \"batch_${i}.log\" 2>&1 &\ndone\nwait\n```\n\n## Key Parameters\n\n| Parameter | Safe Value | Risk Value | Notes |\n|---|---|---|---|\n| Containers | 5 | >8 | One per proxy port (8001-8005) |\n| Concurrency per container | 2-3 | ≥5 | Each proxy port handles 2-3 concurrent connections |\n| Total concurrent | 10-15 | ≥25 | >15 triggers Amazon soft-blocks on ISP proxies |\n| ASINs per batch | 20 | >30 | Bigger batches = longer single-container runtime |\n\n## Performance\n\n| Mode | 100 ASINs | Success Rate |\n|---|---|---|\n| Single container, concurrency 5 | ~20 min | ~90% |\n| 5 containers × concurrency 3 | ~5 min | ~25% (too aggressive) |\n| 5 containers × concurrency 2 | ~5-7 min | ~25% (still aggressive on retry) |\n| 5 containers × concurrency 2, then retry failed with concurrency 1 | ~10 min | ~50% cumulative |\n\n## Retry Strategy for Failed ASINs\n\nAfter the first pass, collect ASINs that returned all-null (proxy soft-blocked), then retry with lower concurrency:\n\n```python\n# Collect failed ASINs\nfor p in all_products:\n    if not (p.get('title') or p.get('brand')):\n        failed_asins.append(p['asin'])\n```\n\nRetry with `--concurrency 1` or `--concurrency 2` on the same 5-container pattern. Typical improvement: +10-15% success on retry.\n\n## Merging Results\n\nAfter all batches + retries complete, merge by ASIN — prefer rows with actual data (title/brand not null):\n\n```python\ndetail_map = {}\nfor prefix in ['batch_', 'retry_']:\n    for i in range(5):\n        d = json.load(open(f'{prefix}{i}.json'))\n        for p in d['products']:\n            if p.get('title') or p.get('brand'):  # has real data\n                detail_map[p['asin']] = p\n```\n\n## When This Pattern is Needed\n- 100+ ASIN detail pages needed (monopoly analysis, competitive research)\n- Time constraint (user wants results in <10 min, not 20+)\n- Search page data already collected (just need brand/seller/bsr/bullets/details)\n\n## When to Use Single Container Instead\n- <30 ASINs (single container concurrency 3-5 is fine)\n- No time pressure\n- Want maximum success rate per ASIN\n\nFile v4.0.1:references/bsr-top100-strategy.md\n\n# Amazon BSR Top 100 抓取策略\n\n## 核心限制（必读）\n\n**`/gp/bestsellers/` URL 只能拿到 2 页 = 60 个产品。** Amazon 官方限制。`?pg=3` 会返回 \"Page Not Found\"。\n\n这意味着：\n- 拿 Top 30 ✅ 直接 `/gp/bestsellers/{category}`\n- 拿 Top 50/60 ✅ `/gp/bestsellers/{category} --pages 2`\n- **拿 Top 100 ❌ 单个 BSR URL 不行**\n\n## 三种 Top 100 策略\n\n### 策略 A：搜索页替代（最快、推荐）⭐\n\n```bash\ndocker run --rm -v /tmp/top100:/data amazon-scraper \\\n  node assets/amazon_handler.js \\\n  \"https://www.amazon.com/s?k=cable+management\" \\\n  --pages 5 \\\n  --output /data/cm.json\n```\n\n- 一次调用，~3 分钟\n- 5 页 = 100+ 个产品（去重后约 60-80 个独立 ASIN）\n- **数据是搜索算法排序的，不是 BSR 严格排名**（混了广告位）\n- 对\"市场分析\"够用\n\n**适合**：快速拿数据做品类分析、价格带分析、视觉调研。\n\n### 策略 B：多子类目 BSR 合并\n\nBSR 父类目下钻到 5 个子类目，每个拿 Top 30 = 150 个产品：\n\n```python\n# 主类目 electronics 没有子节点\n# 子类目节点 ID（在 URL 里能看到）\nsubcats = [\n    \"electronics/172541\",      # Audio & Video\n    \"electronics/281407\",      # Computers & Accessories\n    \"electronics/2407745011\",  # Wearable Technology\n    \"electronics/13896617011\", # Computer & Accessories\n    \"electronics/3024167031\",  # Cell Phones\n]\n# 拼 URL: https://www.amazon.com/gp/bestsellers/{subcat}\n```\n\n每个子 BSR 限 2 页 = 30 个。**5 × 30 = 150，去重 ~120 个**。\n\n**适合**：要做严格的\"畅销榜\"分析（不被广告位污染）。\n\n### 策略 C：单 session 串行（最稳但最慢）\n\n把 BSR Top 30 拿到 ASIN，逐个爬详情，3 个代理并发：\n\n```bash\n# /tmp/top100.sh\nwhile read asin; do\n  docker run --rm amazon-scraper node assets/amazon_handler.js \\\n    \"https://www.amazon.com/dp/${asin}\" \\\n    --output \"/tmp/details/${asin}.json\" 2>/dev/null\n  sleep 3\ndone < /tmp/asins.txt\n```\n\n**耗时**：30 个 ASIN × 15-25 秒 = 7-12 分钟（单容器，1 代理）\n\n**适合**：要拿详情做品牌/BSR/详情页分析。\n\n## 提速方案\n\n| 方案 | 速度 | 限制 |\n|---|---|---|\n| 单容器 `--pages 5` | 1x | skill 本身 |\n| 多 Docker 容器并发 | Nx（N=容器数） | 需 N 个不同代理 |\n| 改 handler.js 加并发 | 内部可控 | 需重新 build 镜像 |\n\n**多容器并发的现实约束**：你的代理数 = 最大并发数。\n- 3 个 DDC IP → 最多 3 个并发 → 3x 提速\n- 5 个 ISP 端口 → 最多 5 个并发 → 5x 提速\n\nskill 自带轮询：handler 内部已支持 1 个容器内多代理轮询 + 故障切换，无需额外配置。\n\n## 速度与限流\n\n- 单个 session：~30 秒/页（含 15 秒冷启动 + 3-5 秒 waitFor）\n- Amazon 限流：30-60 请求/小时/同一 IP 是安全线，超了会触发 503\n- 跑 Top 100 一次消耗约 5-10 个\"请求单位\"\n\n## 输出文件路径\n\n⚠️ **坑**：`--output` 是**容器内路径**。要保存到主机必须挂载卷：\n\n```bash\n# 错：文件在容器里\ndocker run --rm amazon-scraper node .../amazon_handler.js \"URL\" --output /tmp/x.json\n\n# 对：挂载 /data\ndocker run --rm -v /tmp/results:/data amazon-scraper node .../amazon_handler.js \\\n  \"URL\" --output /data/x.json\n# 文件在主机的 /tmp/results/x.json\n```\n\n代码里 `--output` 的路径会拼到 `/data/` 前缀下。\n\nFile v4.0.1:references/cdp-fallback-strategy.md\n\n# CDP Fallback 爬取策略\n\n## 问题场景\n\n批量爬取 Amazon 详情页时（`--asins` 模式），Docker 代理方案在以下情况会被 Amazon 软拦截：\n- 总并发 ≥ 15（5 容器 × concurrency 3）\n- 单代理端口并发 ≥ 3\n- 短时间内同 IP 大量请求\n\n软拦截表现：`status: SUCCESS` 但 `title/brand/seller/bsr` 全部 null，页面未渲染。\n\n## 解决方案：Chrome CDP 直连\n\n**核心思路**：绕过 Docker 代理，用 VPS 本地 Chrome 浏览器（CDP 协议）逐个串行爬取，配合 2-4 秒随机延迟。\n\n**实测结果（2026-07-04）**：\n- Docker 代理方案：75 个失败 ASIN，成功率 0%（全部被拦）\n- CDP 直连方案：75 个失败 ASIN，**成功率 100%**（0 失败）\n- 耗时：75 个 ASIN × ~5 秒/个 = 约 6 分钟\n\n## 前置条件\n\n1. Chrome 已启动并监听 CDP：\n\n```bash\nexport DISPLAY=:99\nXvfb :99 -screen 0 1920x1080x24 &>/dev/null &\n/opt/google/chrome/chrome --disable-gpu --no-first-run \\\n  --no-default-browser-check \\\n  --remote-debugging-port=9222 \\\n  --remote-debugging-address=127.0.0.1 \\\n  --user-data-dir=\"$HOME/.cache/amazon-scraper-chrome\" \\\n  \"https://www.amazon.com\"\n```\n\n> ### ⚠️ 不要加 `--remote-allow-origins=*`\n>\n> CDP **没有任何认证机制**：谁能连上 9222，谁就完全控制这个浏览器 —— 读 cookie、以登录用户身份下单、改收货地址、导出 session。本方案的前提恰恰是这个 Chrome 带着真实 Amazon 登录态，所以它的调试口就等于账号凭证。\n>\n> - `--remote-allow-origins=*` 关掉了 WebSocket 的 Origin 校验，于是**你在这个浏览器里打开的任意网页**都能连上 9222 接管它。只在确实遇到 Origin 报错时，针对具体来源写白名单，不要用 `*`。\n> - 必须显式 `--remote-debugging-address=127.0.0.1`。绑到 `0.0.0.0` 的 VPS 等于开了一个公网无密码浏览器后门；即使绑回环，也要确认没有端口转发或 docker 规则把它暴露出去（`ss -lntp | grep 9222` 自查）。\n> - 不要用 `--no-sandbox`。VPS 上常以 root 跑 Chrome，关掉沙箱意味着一个渲染器漏洞就能拿到 root。需要在容器里跑就改用非 root 用户 + `--user-ns` 之类的方案。\n> - profile 不要放 `/tmp`（全局可写，其他本地用户可读你的 cookie）。放 `$HOME/.cache/...` 并保持 `chmod 700`。\n> - 用完把这个 Chrome 关掉，别长期挂着一个带登录态的调试口。\n\n本目录的两个 CDP 脚本都会**自己新开一个标签页**并在结束时关掉，不会劫持你正在用的标签页。\n\n2. Python 依赖：\n```bash\npip install websocket-client\n```\n\n## 使用方法\n\n```bash\n# 方式1：直接传 ASIN 列表\npython3 ~/.openclaw/skills/amazon-scraper/scripts/cdp_fallback_scrape.py \\\n  --asins \"B07XXX,B08YYY,B09ZZZ\" \\\n  --output cdp_results.json\n\n# 方式2：从 JSON 文件读 ASIN 列表\npython3 ~/.openclaw/skills/amazon-scraper/scripts/cdp_fallback_scrape.py \\\n  --asin-file failed_asins.json \\\n  --output cdp_results.json\n```\n\n## 完整工作流：Docker 批量 + CDP 补漏\n\n```bash\n# Step 1: Docker 批量爬取（快但有失败）\n# 每个容器一个独立出口端口。没有 N 个不同出口就不要开 N 个容器。\n# 凭证从环境变量取，不要写进命令行（会进 shell history 和 ps 输出）。\nread -rsp 'proxy user: ' PROXY_USER; echo\nread -rsp 'proxy pass: ' PROXY_PASS; echo\n\nfor i in 0 1 2 3 4; do\n  PORT=$((8001 + i))\n  AMAZON_PROXIES=\"http://${PROXY_USER}:${PROXY_PASS}@isp.oxylabs.io:${PORT}\" \\\n  docker run --rm -v ~/scrapes:/data \\\n    -e AMAZON_PROXIES \\\n    amazon-scraper node assets/amazon_handler.js \\\n    --asins \"$BATCH_$i\" --concurrency 2 \\\n    --output \"batch_${i}.json\" &\ndone\nwait\n\n# Step 2: 找出失败的 ASIN（title/brand 全 null）\npython3 -c \"\nimport json\nfailed = []\nfor i in range(5):\n    d = json.load(open(f'batch_{i}.json'))\n    for p in d.get('products', []):\n        if not (p.get('title') or p.get('brand')):\n            failed.append(p['asin'])\njson.dump(failed, open('failed_asins.json', 'w'))\nprint(f'{len(failed)} failed ASINs')\n\"\n\n# Step 3: CDP 补漏（串行，100% 成功率）\npython3 ~/.openclaw/skills/amazon-scraper/scripts/cdp_fallback_scrape.py \\\n  --asin-file failed_asins.json \\\n  --output cdp_results.json\n\n# Step 4: 合并所有结果\npython3 -c \"\nimport json\nall_details = {}\nfor i in range(5):\n    d = json.load(open(f'batch_{i}.json'))\n    for p in d.get('products', []):\n        if p.get('title') or p.get('brand'):\n            all_details[p['asin']] = p\nfor p in json.load(open('cdp_results.json')):\n    if p.get('title') or p.get('brand'):\n        all_details[p['asin']] = p\njson.dump(list(all_details.values()), open('merged_details.json', 'w'), ensure_ascii=False, indent=2)\nprint(f'Merged: {len(all_details)} ASINs with full data')\n\"\n```\n\n## 为什么 CDP 方案不会被拦截\n\n1. **真实浏览器指纹**：Chrome 150 的真实 User-Agent、Canvas、WebGL 指纹，比 playwright-extra stealth 更难检测\n2. **VPS 本地 IP**：不走代理，用 VPS 的数据中心 IP 直连。虽然 Amazon 对 datacenter IP 有风控，但**单 IP 低频串行请求（每 3-5 秒一个）不会触发**\n3. **已登录态**：Chrome profile 可能有 Amazon 登录 cookie（如果之前登录过），进一步降低风控。\n\n   ⚠️ 这条是双刃剑，而且代价不对称：带登录态抓取会把抓取行为直接绑到一个真实账号上。被判定为自动化访问时，封的是这个买家账号（连带历史订单、Prime、礼品卡余额），如果它和卖家账号有关联信息，还可能牵连卖家账号。详情页（`/dp/`）匿名可见，**不需要登录态就能抓，就不要用登录态抓**。只有评论页这种必须登录的场景才值得权衡，并且应该用一个专用的、与主业务无关的账号。\n4. **CDP 协议**：通过 Chrome DevTools Protocol 控制真实浏览器，不是 headless 模式\n\n## 注意事项\n\n- **串行慢但稳**：每个 ASIN 约 5 秒（3 秒页面加载 + 2-4 秒随机延迟），100 个 ASIN 约 8-10 分钟\n- **不要并发**：CDP 方案的核心优势就是串行低频，并发会破坏这个优势\n- **Chrome 需保持运行**：脚本运行期间不要关闭 Chrome 或断开 Xvfb\n- **进度保存**：每 10 个 ASIN 自动保存一次结果到 output 文件\n- **适用范围**：仅适用于详情页（`/dp/ASIN`），搜索页/BSR 页仍用 Docker 方案\n\nFile v4.0.1:references/cron-price-monitor-pattern.md\n\n# 价格监控定时任务\n\n> OpenClaw automations + amazon-scraper + 飞书多维表格 + Discord 推送\n\n## 架构\n\n```\nopenclaw automations（每 6 小时，command，不另起一轮模型）\n  └─ pickleball_price_monitor.py\n      ├─ Docker amazon-scraper 抓取 ASIN 价格\n      ├─ SQLite 存储历史价格（~/.openclaw/amazon-scraper/）\n      ├─ 飞书多维表格写入每次检查记录 (lark-cli base +record-batch-create)\n      └─ 价格变化时 → stdout → 按 automation 的 delivery 推送\n          无变化时 → stdout 为空\n```\n\n## 创建飞书多维表格\n\n```bash\nlark-cli base +base-create --name \"监控表名\" --table-name \"价格记录\" \\\n  --fields '[\n    {\"name\":\"ASIN\",\"type\":\"text\"},\n    {\"name\":\"商品标题\",\"type\":\"text\"},\n    {\"name\":\"当前价格\",\"type\":\"number\"},\n    {\"name\":\"上次价格\",\"type\":\"number\"},\n    {\"name\":\"价格变化\",\"type\":\"number\"},\n    {\"name\":\"变化幅度\",\"type\":\"text\"},\n    {\"name\":\"检查时间\",\"type\":\"datetime\"},\n    {\"name\":\"状态\",\"type\":\"text\"}\n  ]' --as user\n```\n\n返回 `base_token` 和 `table.id`，填入脚本的 `BASE_TOKEN` 和 `TABLE_ID`。\n\n## 飞书写入格式 (lark-cli base +record-batch-create)\n\n`--json` 必须是 `{\"fields\":[...],\"rows\":[[...]]}` 格式，不是对象数组：\n\n```python\nbatch_json = json.dumps({\n    \"fields\": [\"ASIN\", \"商品标题\", \"当前价格\", \"上次价格\", \"价格变化\", \"变化幅度\", \"检查时间\", \"状态\"],\n    \"rows\": [\n        [\"B0F6XSV7XB\", \"商品标题\", 42.99, 45.99, -3.00, \"↓ -6.5%\", 1720000000000, \"降价\"],\n        # ...\n    ]\n}, ensure_ascii=False)\n```\n\n- `检查时间` datetime 字段需**毫秒时间戳**（秒级 × 1000）\n- `ensure_ascii=False` 保留中文\n- 单批最多 200 行\n\n## 创建定时任务\n\n`--command` 在 Gateway 主机上直接跑脚本，不另起一轮模型。`--every 6h` 与时区无关。\n\n```bash\nopenclaw automations add \\\n  --every 6h \\\n  --name \"Pickleball价格监控\" \\\n  --command \"python3 ~/.openclaw/skills/amazon-scraper/scripts/pickleball_price_monitor.py\" \\\n  --session isolated \\\n  --announce\n```\n\n每天北京时间 9:00 用 cron 表达式，并显式指定时区，不要靠把小时减 8：\n\n```bash\nopenclaw automations create \"0 9 * * *\" \\\n  --name \"每天9点价格监控\" \\\n  --tz Asia/Shanghai \\\n  --command \"python3 ~/.openclaw/skills/amazon-scraper/scripts/asin_price_monitor.py\" \\\n  --session isolated \\\n  --announce\n```\n\n## 时区\n\n不带 `--tz` 时，cron 表达式按 **Gateway 主机时区**计算，不是固定 UTC。要固定北京时间就加 `--tz Asia/Shanghai`，表达式里直接写北京时间的小时。`--every 6h` 这种间隔不受时区影响。\n\n## command 任务的输出\n\n- `--command` 只跑进程，不调用模型\n- 脚本在有价格变化时把 Markdown 打到 stdout；无变化时 stdout 为空\n- `--announce` 把这次运行的输出按任务的 delivery 送出（例如 Discord）\n- 脚本 stderr 是日志，不要把报告打到 stderr\n\n## 修改监控 ASIN 列表\n\n编辑脚本顶部的 `ASINS` 列表：\n```python\nASINS = [\"B0F6XSV7XB\", \"B0FTQWG86Q\", \"B0G6CTNVQT\"]\n```\n\n修改后无需重启 cron，下次执行自动生效（脚本每次运行时读取）。\n\n## 依赖\n\n- `amazon-scraper` Docker 镜像（已构建）\n- `lark-cli` 已安装并认证（`lark-cli auth login`）\n- Python 3.11+（sqlite3 内置）\n\nFile v4.0.1:references/full-extract-mode.md\n\n# Full Extract Mode — 43字段全量抓取\n\n> `scripts/full_extract.js` — 独立于 `amazon_handler.js` 的全量提取脚本\n\n## 什么时候用\n\n- 用户说\"爬取页面里所有内容\"、\"最大化字段\"、\"所有可见字段\"、\"完整抓取\"\n- 需要原来 15 个字段之外的数据：划线价、评分分布、全部图片、视频、A+内容、评论摘要、变体、优惠券、交付时间等\n- 需要分析竞品 Listing 全貌（不只看价格/评分）\n\n## 用法\n\n```bash\ndocker run --rm \\\n  -v ~/.openclaw/skills/amazon-scraper/scripts/full_extract.js:/app/full_extract.js \\\n  amazon-scraper node /app/full_extract.js \"https://www.amazon.com/dp/ASIN/\"\n```\n\n## 抓取的 43 个字段\n\n| 分类 | 字段 | 说明 |\n|---|---|---|\n| **基本信息** | title, asin, brand, brandUrl | 标题/ASIN/品牌/品牌店铺链接 |\n| **价格** | prices.main, prices.listPrice, prices.dealPrice, prices.buybox, prices.allPriceElements | 主价/划线价/优惠价/buybox/所有价格元素 |\n| **评分评论** | rating, ratingText, reviews, reviewsText, ratingHistogram | 评分/评分数/评分分布直方图 |\n| **图片** | images[main/thumbnail/hires/aplus] | 全部图片分类标记 |\n| **视频** | videos[video] | 含poster帧 |\n| **五点描述** | bullets, bulletsCount | 完整五点 |\n| **A+内容** | aplusContent, aplusLength | A+文本和长度 |\n| **详情表** | details (20+ keys), detailsCount | 技术规格键值对 |\n| **BSR** | bsr, bsrFullText | BSR数字+含分类排名全文 |\n| **分类** | category, categoryFull | 面包屑/完整路径 |\n| **销量** | boughtPastMonth | 月销量 |\n| **日期** | dateFirstAvailable | 上架日期（常为null） |\n| **卖家** | seller, shipsFrom, fulfilledBy | 卖家/发货方/FBA |\n| **库存** | availability, isPrime, delivery | 库存状态/Prime标记/预计送达 |\n| **促销** | coupon, subscribeAndSave | 优惠券/订阅优惠 |\n| **变体** | variations (allOptions, allAsins) | 颜色/尺寸选项+变体ASIN |\n| **推荐** | frequentlyBoughtTogether, carousels | FBT/关联推荐 |\n| **Q&A** | customerQA | 客户问答 |\n| **评论** | topReviews[8] | 顶部评论（作者/评分/标题/日期/正文/verified/有用数） |\n| **安全** | safetyWarning, newerVersion | 安全警告/新版提示 |\n| **元数据** | metaTags, canonicalUrl, pageUrl | 页面meta/规范URL |\n| **结构化** | structuredData (JSON-LD) | 结构化数据 |\n| **全文** | fullPageText | 整页纯文本（截断2万字） |\n\n## 与 amazon_handler.js 对比\n\n| 维度 | amazon_handler.js | full_extract.js |\n|---|---|---|\n| 字段数 | 15 | 43 |\n| 价格 | 仅 price/priceStr | 划线价/deal价/buybox/全部价格元素 |\n| 图片 | 仅1张主图 | 全部图片（main/thumb/hires/aplus分类） |\n| 视频 | 无 | 有（含poster） |\n| 评论 | 仅 rating + reviews 数 | 8条顶部评论（含作者/日期/正文/verified） |\n| A+内容 | 无 | 有 |\n| 变体 | 无 | 有（选项+变体ASIN） |\n| 交付/库存 | 无 | availability/isPrime/delivery |\n| 优惠券 | 无 | 有 |\n| 输出大小 | ~3KB | ~85KB |\n\n## 注意事项\n\n1. **proxy日志混入stdout首行**：`full_extract.js` 的 stdout 第一行可能是 `Using proxy: ...`，解析JSON时从第一个 `{` 开始截取\n2. **容器内代理路径**：硬编码为 `/app/config/proxies.json`，不是 `__dirname/../config/`\n3. **懒加载**：需要8次滚动 + 1秒等待才能加载全部图片和A+内容\n4. **偶发超时**：`waitForSelector('#productTitle, #dp', { timeout: 15000 })` 偶尔超时，重试即可（代理轮换）\n5. **dateFirstAvailable 常为 null**：Amazon 新版页面不再展示此字段\n6. **canonicalUrl 可能指向变体ASIN**：如 `B0F6XSV7XB` 的 canonical 指向 `B0GY92641V`（父ASIN）\n\nFile v4.0.1:references/new-layout-fixes-2026-07.md\n\n# Amazon New Layout Fixes (2026-07)\n\nSession-specific details of the DOM structure changes and fixes applied to `amazon_handler.js`.\n\n## Problem: Detail pages returning all-null fields\n\n### Root Cause\n1. `Accept-Encoding: identity` in `extraHTTPHeaders` + `setExtraHTTPHeaders` caused Amazon's JS to not render the product page content. Removing the header and the entire `setExtraHTTPHeaders` call fixed it.\n2. Old `waitForTimeout(3000)` was too short for new JS-heavy pages. Added `waitForSelector('#productTitle, #dp', { timeout: 15000 })` for product-detail pages.\n\n### New Amazon Detail Page DOM Structure (July 2026)\n\n| Field | Old Selector/Regex | New Selector | Notes |\n|---|---|---|---|\n| title | `#productTitle` | `#productTitle` | Unchanged |\n| brand | `#bylineInfo` | `#bylineInfo` | Returns \"Brand: XXX\" or \"Visit the XXX Store\" — must strip prefixes |\n| seller | N/A | `#sellerProfileTriggerId` | New field, e.g. \"BeataTap-SKALON\" |\n| bsr | `body.innerText` regex | `#prodDetails tr` th/td scan | Regex `Best Sellers Rank.*?#([\\d,]+)` no longer matches; must scan `#prodDetails tr` rows |\n| dateFirstAvailable | `body.innerText` regex | N/A — **100% missing** | Amazon no longer shows this on most product pages |\n| details | `#productDetails_techSpec_section_1 tr` | `#prodDetails tr` | New layout uses `#prodDetails` instead of `#productDetails_techSpec_section_1` |\n\n### Brand Cleanup Regex\n```js\nbrand = brand.replace(/^Brand:\\s*/i, '').replace(/^Visit the\\s+/i, '').replace(/\\s+Store$/i, '').trim();\n```\n\n### BSR Extraction (New Layout)\n```js\ndocument.querySelectorAll('#prodDetails tr').forEach(row => {\n    const cells = row.querySelectorAll('th, td');\n    if (cells.length >= 2) {\n        const key = cells[0].textContent.trim().replace(/[:\\s]+$/, '');\n        const val = cells[1].textContent.trim();\n        if (key.match(/Best Sellers Rank/i)) {\n            const m = val.match(/#([\\d,]+)/);\n            if (m) bsr = parseInt(m[1].replace(/,/g, ''));\n        }\n    }\n});\n```\n\n### Seller Extraction (3 methods, in order)\n1. `#sellerProfileTriggerId` — most reliable (e.g. \"BeataTap-SKALON\")\n2. `body.innerText.match(/Sold by\\s+([\\w\\s\\-\\.]+)/i)` — fallback\n3. `#prodDetails tr` row with \"Ships from\" or \"Sold by\" key — last resort\n\n### Seller Data Garbage Filter (in analyze.py)\nSome ASINs' `#sellerProfileTriggerId` captures Amazon comparison widget text:\n```\n\"different sellers.\\nShow details\\nExplore more from across the store\\nPage 1 of 7\\n...\"\n```\nFilter rule: if seller contains `\\n`, \"Show details\", \"Explore more\", or \"Page 1 of\" → set to null.\n\n## Proxy: Datacenter → ISP\n\n| Proxy Type | Entry Domain | Amazon Works? |\n|---|---|---|\n| Datacenter (DDC) | `disp.oxylabs.io` | ❌ 100% soft-blocked (empty page) |\n| ISP | `isp.oxylabs.io` | ✅ Working |\n\nChanged `config/proxies.json` from `disp.oxylabs.io` to `isp.oxylabs.io`. Must `docker build` after changing proxies.json since it's COPY'd into the image.\n\n## Search Page Limitation\n\n`--pages 5` returns ~26 unique ASINs, not 100. Amazon search results contain many sponsored/duplicate listings. To get closer to Top 100:\n- Use more pages (--pages 10) → returns ~234 unique ASINs (enough for Top 100)\n- Or use keyword variations and merge results\n\n## Search Page Pagination Fix (2026-07-04)\n\n**Bug**: `amazon_handler.js` used `&pg=${pg}` for search page pagination. Amazon does NOT recognize `&pg=N` — it silently returns page 1 data for every page. This caused `--pages 10` to return only 29 unique ASINs (all from page 1, just duplicated 10 times).\n\n**Fix**: Changed `&pg=N` to `&page=N` in the URL construction:\n```js\n// Before (broken):\nif (pg > 1) url = job.url.includes('?') ? `${job.url}&pg=${pg}` : `${job.url}?pg=${pg}`;\n// After (fixed):\nif (pg > 1) url = job.url.includes('?') ? `${job.url}&page=${pg}` : `${job.url}?page=${pg}`;\n```\n\n**Verification**: `--pages 10` with `&page=N` returns 234 unique ASINs (vs 29 with `&pg=N`).\n\n**Diagnosis tip**: If multi-page search scraping returns far fewer unique ASINs than expected, check the pagination parameter first.\n\nFile v4.0.1:references/oxylabs-ddc-format.md\n\n# Oxylabs DDC — 工作格式与故障排查\n\n## 结论（2026-07-02 实测 + 官方文档）\n\n**Oxylabs DDC 配爬虫，必须用入口域名 `ddc.oxylabs.io` + 端口，不要直接打 \"Assigned IP\"。**\n\n| 形式 | 结果 |\n|---|---|\n| `http://user-XXX:pass@ddc.oxylabs.io:8001` | ✅ 官方推荐格式，代理层通（前提：目标站没被 Oxylabs 标 restricted） |\n| `http://user-XXX:pass@45.73.183.199:8001` | ❌ 直连 IP。VPS 上 `socket.connect()` 直接 `No route to host`（45.x.x.x 段路由不通） |\n\n每个 user 关联的 \"Assigned IP\" 在 Oxylabs dashboard 的 `My Products → Dedicated Datacenter Proxies → Proxy list` 里能看到。**那个 IP 是 Oxylabs 帮你分配的出口 IP，不是入口**。文档原文：\n\n> \"Please note that you will use ports for making requests with your IPs, meaning you will not directly access your IPs.\"\n\n> \"Entry point: The gateway to connect to your proxies — this value never changes and is always `ddc.oxylabs.io`.\"\n\n## ⚠️ 关于\"用 DDC 爬 Amazon\"本身\n\n**DDC（datacenter 代理）对 Amazon 大规模爬取基本不可用。** 不管用域名还是直连 IP，Amazon 都会：\n\n- 整段 datacenter ASN 范围（AS14061 / AS204957 / AS8100 / AS6079 等）标为高风险\n- 返回 \"Sorry! Something went wrong!\" 软拦截页面（totalProducts: 0，HTTP 200，伪装成正常）\n- 偶尔能跑通 1-2 个请求（stealth 模式 + 时机），**不可重复**\n\n`proxies.json` 配置 100% 正确 ≠ 拿到数据。**剩下的不是配置问题，是 Amazon 反爬针对 datacenter 段的问题**。\n\n**对 Amazon 真正可用的方案**（按成本升序）：\n\n1. **Oxylabs ISP Proxies** — 入口 `isn.oxylabs.io` 或 `pr.oxylabs.io:8001`，住宅 ISP 段 IP，Amazon 不拦。**2026-07-02 实测跑通 cable management Top 100**。\n2. **Oxylabs Web Scraper API** — 不是代理，是 HTTP POST API，Oxylabs 自己爬好你拿 JSON。`https://realtime.oxylabs.io/v1/queries` + `source: \"amazon_search\"`。无爬虫代码，$1.50/1K 请求。\n3. **其他 ISP 池** — Smartproxy / IPIDEA / Bright Data ISP 套餐。\n\n## 故障排查命令\n\n单独验证代理还活着（**先确认代理本身通**，再考虑目标站拦不拦）：\n\n```bash\n# 用 curl 测（注意：用 HTTP 不是 HTTPS，否则 urllib3 会误诊协议）\ncurl -x ddc.oxylabs.io:8001 -U \"user-XXX:password\" http://ip.oxylabs.io/location\n\n# 用 python requests 测\npython3 -c \"\nimport requests\np = {'http': 'http://user-XXX:pass@ddc.oxylabs.io:8001',\n     'https': 'http://user-XXX:pass@ddc.oxylabs.io:8001'}\nr = requests.get('https://ip.oxylabs.io/location', proxies=p, timeout=20)\nprint(r.status_code, r.text[:200])\n\"\n```\n\n返回 200 + JSON（如 `{\"ip\":\"205.188.202.202\",\"country\":\"US\",...}`）= 代理层通了。\n返回 503 = Oxylabs 主动拒（看 response code 速查表）。\n返回 `No route to host` = **你直接打了直连 IP**（45.x.x.x 段），改回 `ddc.oxylabs.io` 域名。\n\n## Oxylabs Response Code 速查\n\n| Code | 含义 | 怎么修 |\n|------|------|--------|\n| 400 | 请求格式错 | 检查 URL 格式 |\n| 403 | **restricted target** | 目标站（如 amazon.com）在此产品黑名单。换 ISP Proxies 或 Web Scraper API |\n| 404 | 域名解析不到 | 域名错或资源下架 |\n| 407 | 凭证错 / 没白名单 | 检查 user/pass，或在 dashboard 加白名单出口 IP |\n| 429 | 并发超限 | 降速或买更多带宽 |\n| 503 | 目标 DNS 失败 | 目标站从代理侧不可达。先直连目标确认它活着 |\n| 504 | 代理超时（60s） | 目标慢，重试或加超时 |\n\n## 历史教训\n\n**之前（2026-07-01）这个文件写的是反的**：\"用直连 IP，不要用域名\"。那次\"成功\"是 1 次侥幸（stealth 模式 + Amazon 当时没识别），不是配置正确。\n\n**2026-07-02 复测**：\n- 直连 IP `45.73.183.199:8001` → `OSError: [Errno 113] No route to host`（VPS 路由层）\n- 域名 `ddc.oxylabs.io:8001` → 503（按官方 response code 表 = 目标 DNS 失败，**不是格式问题**，是 Amazon 在 Oxylabs 黑名单）\n\n下次遇到\"代理挂了\"先分两类：\n1. 代理层通不通？→ 跑 `curl -x ddc.oxylabs.io:8001 ... http://ip.oxylabs.io/location` 验证\n2. 代理通了之后目标让不让爬？→ 看返回页是 \"Sorry! Something went wrong!\" 还是真数据\n\nFile v4.0.1:references/oxylabs-product-matrix.md\n\n# Oxylabs 代理产品矩阵 — 哪个产品对应哪个入口域名\n\n## 速查表\n\n| Oxylabs 产品 | 入口域名 | Amazon 可用? | 适用场景 | 备注 |\n|---|---|---|---|---|\n| **Datacenter Proxies** (DC) | `dc.oxylabs.io` | ❌ | 大流量、低成本、对反爬不严 | 与 DDC 不同产品 |\n| **Dedicated Datacenter Proxies** (DDC) | `ddc.oxylabs.io` | ❌ | 个人独享 IP、对反爬不严 | 文档说\"你不会直接访问 IP\" |\n| **ISP Proxies** ⭐ | `isn.oxylabs.io` 或 `pr.oxylabs.io` | ✅ **推荐** | Amazon 爬取、电商 | 住宅 ISP 段，Amazon 不拦 |\n| **Dedicated ISP Proxies** | `pr.oxylabs.io` | ✅ | 长期项目、高频爬 Amazon | 比 ISP 更稳定更贵 |\n| **Residential Proxies** | `pr.oxylabs.io` | ✅ | 任何目标站、轮换 IP | 按 GB 流量计费 |\n| **Mobile Proxies** | `mob.oxylabs.io` | ✅ | 反爬最严目标 | 4G/5G 移动 IP，最难被识别 |\n| **Web Scraper API** (非代理) | `https://realtime.oxylabs.io/v1/queries` | ✅ | 不写爬虫代码，HTTP POST 拿 JSON | $1.50/1K 请求起 |\n\n## `proxies.json` 模板\n\n### ISP Proxies（Amazon 推荐）\n```json\n{\n  \"proxies\": [\n    \"http://user-XXX:pass@isn.oxylabs.io:8001\",\n    \"http://user-XXX:pass@isn.oxylabs.io:8002\",\n    \"http://user-XXX:pass@isn.oxylabs.io:8003\"\n  ]\n}\n```\n\n### DDC（用 `ddc.oxylabs.io` 域名）\n```json\n{\n  \"proxies\": [\n    \"http://user-XXX:pass@ddc.oxylabs.io:8001\",\n    \"http://user-XXX:pass@ddc.oxylabs.io:8002\",\n    \"http://user-XXX:pass@ddc.oxylabs.io:8003\"\n  ]\n}\n```\n\n⚠️ **不要把 dashboard 上的 \"Assigned IP\" 当入口用**（如 `45.73.183.199:8001`）。VPS 上直连会 `No route to host`。\n\n### Web Scraper API（不走代理）\n不是 `proxies.json` 配的，是**直接发 HTTP POST**：\n```bash\ncurl 'https://realtime.oxylabs.io/v1/queries' \\\n  -u \"API_USER:API_PASS\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"source\":\"amazon_search\",\"query\":\"cable management\",\"parse\":true}'\n```\n返回完整 JSON（含 title/price/asin/bsr/...），无爬虫代码。\n\n## 判断\"该买哪个\"的决策\n\n| 你的情况 | 买什么 |\n|---|---|\n| 一次爬 100-1000 个 Amazon 产品 | ISP Proxies 最低套餐 |\n| 每天/每周跑 Amazon 监控 | Dedicated ISP Proxies |\n| 完全不写代码，只想要数据 | Web Scraper API |\n| 爬非 Amazon 站（普通电商/新闻） | Residential Proxies |\n| 预算紧 + 目标站反爬弱 | Datacenter Proxies |\n| 反爬最严（机票/Sneaker/抢购） | Mobile Proxies |\n\n## 切换代理的最小动作\n\n1. 改 `~/.openclaw/skills/amazon-scraper/config/proxies.json`（按上面模板）\n2. 直接 `docker run` 即可（**不需要重新 build 镜像** — `proxies.json` 是运行时读取）\n\n⚠️ 反例：改 `assets/amazon_handler.js` 后才需要 `docker build -t amazon-scraper .` 重新 build。\n\n## 凭证安全\n\n`proxies.json` 里的 user/pass 是**明文凭证**。打包 skill 发给别人前先参考 `sharing-checklist.md`（先问真实/占位/空数组 三选一）。\n\nFile v4.0.1:references/oxylabs-proxy-format.md\n\n# Oxylabs Dedicated Datacenter Proxies — connection cheat sheet\n\n## Correct proxy URL format\n\nFor **Oxylabs Dedicated Datacenter Proxies (DDC, self-service)**, the proxy URL is:\n\n```\nhttp://USERNAME:PASSWORD@ddc.oxylabs.io:PORT\n```\n\n- **Entry point**: `ddc.oxylabs.io` (NOT direct IP, NOT `pr.oxylabs.io`)\n- **Port**: `8001` and up (each port = one assigned IP)\n- Username: the proxy user created in Oxylabs dashboard (e.g. `yourname_AB12C`)\n\n**Common mistake**: trying to connect to the assigned IP directly (e.g. `45.73.183.199:8001`). Oxylabs explicitly states: *\"you will not directly access your IPs\"*. Use the entry point domain, not the IP.\n\n`proxies.json` example for 3 DDC ports:\n\n```json\n{\n  \"proxies\": [\n    \"http://user-ACCOUNT_ID:PASSWORD@ddc.oxylabs.io:8001\",\n    \"http://user-ACCOUNT_ID:PASSWORD@ddc.oxylabs.io:8002\",\n    \"http://user-ACCOUNT_ID:PASSWORD@ddc.oxylabs.io:8003\"\n  ]\n}\n```\n\n## Response code quick reference (from Oxylabs docs)\n\n| Code | Meaning | What to do |\n|------|---------|------------|\n| 400 | Bad request format | Check the URL format above |\n| 403 | **Restricted target** | The site (e.g. amazon.com) is on Oxylabs' restricted list for this proxy product. Switch proxy product (e.g. ISP Proxies) or use Web Scraper API instead |\n| 407 | Auth failed or IP not whitelisted | Wrong username/password, or you switched VPS and didn't re-whitelist the new exit IP |\n| 429 | Thread / concurrent session limit exceeded | Slow down, or buy more bandwidth |\n| 503 | **DNS failure to target** | Target site unreachable from proxy. Test target directly to confirm |\n| 504 | Proxy timeout (60s) | Target is slow; retry or extend timeout |\n\n## Verifying proxy works (in 30 seconds)\n\nBefore scraping, test the proxy chain with a simple reachability check:\n\n```bash\ncurl -x ddc.oxylabs.io:8001 -U \"USERNAME:PASSWORD\" http://ip.oxylabs.io/location\n```\n\n- **200 + JSON with your assigned IP** → proxy works\n- **503** → either wrong credentials OR target DNS issue (rare; mostly 503 = restricted target like Amazon)\n- **No route to host / connection refused** → VPS can't reach Oxylabs' datacenter IP range (likely GFW / VPS provider blocking 45.x.x.x). Use the entry point domain `ddc.oxylabs.io` instead, which uses DNS to find a routable IP\n\n## When DDC doesn't work for your target (Amazon specifically)\n\nAmazon aggressively blocks datacenter ASN ranges. If you get HTTP 200 but the page says *\"Sorry! Something went wrong!\"* — that's Amazon's soft-block, not a proxy config error. The proxy is working, but Amazon has flagged the IP.\n\n**What works for Amazon** (in order of cost):\n1. **Oxylabs ISP Proxies** (`isn.oxylabs.io` or `pr.oxylabs.io:8001`) — same company, residential-grade IPs, passes Amazon\n2. **Oxylabs Web Scraper API** (HTTP POST, not a proxy) — `https://realtime.oxylabs.io/v1/queries` with `source: \"amazon_search\"` etc. — no scraping code needed, ~$1.50/1K requests\n3. **Other providers' ISP pools** (Smartproxy, IPIDEA, Bright Data ISP)\n\nDDC is the cheapest product but is fundamentally the wrong tool for Amazon-scale scraping.\n\n## Rotating through multiple proxies\n\nIn `proxies.json` you can list multiple URLs. The `amazon_handler.js` script in this skill auto-rotates through them per request and falls back on failure. No extra config needed beyond a JSON array.\n\nFile v4.0.1:references/product-form-fusion.md\n\n# Amazon 产品形态融合创意（视觉化选品思路）\n\n## 场景\n用户提出类似\"亚马逊 X 品类首页 Top 5 产品外观形态融合，能产生什么新产品\"的需求 —— 这是一种**视觉化选品** + **形态组合创意**工作流。\n\n## 典型流程\n\n### 1. 抓数据\n用 `amazon_handler.js` 抓亚马逊类目 / 搜索结果：\n\n```bash\n# 两种入口\n# (a) 类目页（标题全有）\ndocker run --rm amazon-scraper node assets/amazon_handler.js \\\n  \"https://www.amazon.com/gp/bestsellers/electronics/5180508011\"\n\n# (b) 关键词搜索（经常 title=null，备 image URL）\ndocker run --rm amazon-scraper node assets/amazon_handler.js \\\n  \"https://www.amazon.com/s?k=cable+management+organizer\"\n```\n\n### 2. 取 Top N + 提取 image URL\n```python\nimport json\ndata = json.load(open('/tmp/cm.json'))\nfor p in (data.get('products') or [])[:5]:\n    print(f\"ASIN: {p['asin']}\")\n    print(f\"Img:  {p['image']}\")\n    print(f\"Price: ${p['price']}  Rating: {p['rating']}★  Reviews: {p.get('reviews')}\")\n```\n\n### 3. 关键 fallback：搜索页 title 缺失时\n如果 `p.get('title')` 是 `None`（**搜索页极常见**），**用 `p['image']` URL + `vision_analyze` 看图识物**：\n\n```python\nvision_analyze(\n    image_url=p['image'],\n    question=\"这是一个亚马逊 [类目] 类的产品图片。请详细描述产品外观、形态、材质、颜色、尺寸感。\"\n)\n```\n\n这是**唯一可靠**识别搜索结果产品形态的方法。\n\n### 4. 形态融合创意结构\n\n每张图分析完，整理出 Top N 产品的\"形态词条\"：\n- 形态 1（扎带 / 卷状 / 尼龙）\n- 形态 2（卡扣 / 圆形 / 黑色塑料）\n- 形态 3（理线槽 / 长条 / 梳齿）\n- 形态 4（金属托盘 / 网格 / 桌下）\n- 形态 5（PVC 软槽 / 自粘 / 桌边）\n\n然后做\"5 → 1 融合\"创意，**输出 3-5 个候选产品外观方案**，每个包含：\n- 命名（一句话）\n- 形态融合说明（哪 5 个怎么组合）\n- 外观描述（材质、颜色、尺寸感、造型比喻）\n- 功能描述（解决什么用户痛点）\n- 推荐指数 + 理由（客单价 / A+ 好做度 / 安装便捷 / 视觉冲击）\n\n### 5. 推荐筛选标准\n- **客单价提升空间**：单一功能 $7-15 → 多合一 $25-40\n- **视觉冲击**：黑色磨砂金属 / 极简铝型材 > 普通黑塑料\n- **安装便捷**：磁吸 / 3M 胶 / 0 工具 = 转化率高\n- **配件生态**：模块化 = 复购\n- **目标人群**：苹果/华为桌面办公 / 电竞 RGB / 家庭办公升级\n\n## Pitfalls\n- ❌ 搜索页 `title=null` 时以为抓失败 → 实际上数据齐了，**用 image URL + vision 即可**\n- ❌ 强行用 `/s?k=...` 拿 BSR Top → 搜索页没排名概念，混淆数据\n- ❌ 创意堆砌\"什么都加上\" → 一个产品主形态要明确，最多 2-3 个融合点\n- ❌ 给传统线下/工业产品做这种融合 → 亚马逊 C 端才吃这套\n- ✅ Top 5 足够（再多了信息冗余，融合时反而抓不到重点）\n- ✅ 形态描述尽量用视觉化语言（\"鹅卵石\"、\"桌面港湾\"、\"科技树\"）便于后面做产品图\n- ✅ 推荐 Top 1 时给出\"客单价 / 视觉 / 安装 / 配件\"4 维理由，不只是\"我觉得好\"\n\n## 相关 reference\n- 主 SKILL.md \"搜索页 title 缺失\" fallback 章节\n- 选品决策树（SKILL.md \"Agent 调用决策树\"）\n\nArchive v4.0.0: 32 files, 91509 bytes\n\nFiles: assets/amazon_handler.js (28708b), assets/fingerprint.js (6399b), assets/main_handler.js (4284b), assets/proxy.js (3264b), CHANGELOG.md (6941b), config/proxies.example.json (307b), Dockerfile.sh (1682b), package.json (634b), references/batch-parallel-scraping.md (3831b), references/bsr-top100-strategy.md (3378b), references/cdp-fallback-strategy.md (6430b), references/cron-price-monitor-pattern.md (3339b), references/full-extract-mode.md (3769b), references/new-layout-fixes-2026-07.md (4077b), references/oxylabs-ddc-format.md (4308b), references/oxylabs-product-matrix.md (2985b), references/oxylabs-proxy-format.md (3326b), references/product-form-fusion.md (3323b), references/reviews-strategy.md (13443b), references/search-page-title-fix.md (3232b), references/sharing-checklist.md (3084b), scripts/asin_monitor.py (22063b), scripts/asin_price_monitor.py (3727b), scripts/cdp_fallback_scrape.py (9804b), scripts/full_extract.js (17497b), scripts/pickleball_price_monitor.py (6801b), scripts/scrape_reviews.py (9436b), scripts/scraper_client.py (2291b), scripts/setup.sh (1894b), skill-card.md (2084b), SKILL.md (19936b), _meta.json (133b)\n\nFile v4.0.0:SKILL.md\n\n---\nname: amazon-scraper\ndescription: >\n  Containerized Amazon.com scraper (Docker + Playwright) for BSR/new-releases/movers rankings,\n  keyword search results, and product detail pages. Requires a paid ISP/residential proxy.\n\n  Use when the request is specifically about Amazon product or marketplace data:\n  亚马逊/Amazon, ASIN, BSR, Best Sellers, 畅销榜, 新品榜, 飙升榜,\n  选品, 竞品分析, 类目分析, listing 分析, 月销量 (bought in past month),\n  评分分布, 评论分析, Amazon 关键词搜索结果, Amazon 产品详情.\n\n  Do NOT use for general-purpose web scraping just because the user said\n  爬取/抓取/采集/scrape/crawl. Every run spends metered proxy bandwidth, and the\n  generic mode exists only as a fallback for pages related to an Amazon task.\n  For unrelated sites prefer web_search/web_fetch, or ask first.\nmetadata:\n  openclaw:\n    requires:\n      bins:\n        - docker\n---\n\n# Amazon Scraper\n\nDocker 容器化爬虫，Playwright Chromium。不使用 stealth 插件。支持亚马逊榜单/搜索/详情及通用动态页。\n\n## 第 0 步（开跑前必须先做，不过就停）\n\n先确认出口，再碰亚马逊：\n\n```bash\nexport AMAZON_PROXIES=\"http://USER:PASS@HOST:PORT\"   # 从密码管理器取，不要贴进对话\ncurl -s -x \"$AMAZON_PROXIES\" http://api.ipify.org\n```\n\n返回的纯文本必须等于你配置的代理 host。不是这个 IP、超时、407、403，都停，不要开爬，也不要重建后直接跑。\n\n- **凭证不进仓库、不进镜像、不进命令回显。** 用 `-e AMAZON_PROXIES`（值从环境取，不要写在命令行里）或挂载 `config/proxies.json`。该文件已在 `.gitignore` / `.dockerignore` 里。\n- 这一步只证明代理层通。裸 curl 打亚马逊拿到 500/202，不算第 0 步失败。\n- 第 0 步过了，才允许 `docker run amazon-scraper`。\n\n## 系统要求\n\n- **Docker Engine 20.10+**（必须已安装并运行）\n- **磁盘空间**：~2GB（镜像 + Playwright 浏览器二进制文件）\n- **内存**：建议 2GB+（Playwright 运行时需要）\n\n## 快速开始\n\n首次使用：在 skill 目录下执行一键构建脚本：\n\n```bash\nbash scripts/setup.sh\n```\n\n脚本会自动完成：构建 `amazon-scraper` 镜像 + 创建 `~/scrapes` 输出目录。\n\n\n## 模式选择规则\n\n### 1. Amazon模式 (`amazon_handler.js`)\n**自动触发条件:** URL包含 `amazon.com`，或用户提到亚马逊/Amazon/ASIN/BSR/选品/竞品/畅销榜/类目分析等关键词\n\n根据URL自动识别页面类型：\n\n| URL特征 | 页面类型 | 可获取字段 |\n|---|---|---|\n| `/gp/bestsellers/` | 畅销榜 | rank, title, asin, price, rating, reviews, image, url |\n| `/zg/new-releases/` | 新品榜 | 同上 |\n| `/zg/movers-and-shakers/` | 飙升榜 | 同上 |\n| `/s?k=` 或 `/s/` | 搜索结果 | title, asin, price, rating, reviews, image, url, **boughtPastMonth**, sponsored |\n| `/dp/` 或 `/gp/product/` | 产品详情 | title, asin, price, rating, reviews, brand, bsr, **boughtPastMonth**, **seller**, dateFirstAvailable, category, bullets, details |\n\n**⚠️ 重要规则:**\n- **Best Sellers 页面没有月销量(boughtPastMonth)数据** — 亚马逊不在榜单页显示此信息\n- **要获取月销量，必须用搜索页(`/s?k=关键词`)或产品详情页(`/dp/ASIN`)**\n- 如果用户同时需要排名+月销量，建议：先爬 Best Sellers 拿排名，再用搜索页补月销\n- **BSR URL 必须使用 `/gp/bestsellers/`**，`/zgbs/` 会返回 Page Not Found\n- **BSR 单 URL 只能拿 60 个产品**（2 页限制）。拿 Top 100 用搜索页 `--pages 5` 或多子类目 BSR 合并。详见 `references/bsr-top100-strategy.md`\n- **评论（rating/title/body/date/helpful/verified）不在这张表里**：需要登录态 + CDP，走 `scripts/scrape_reviews.py`（`/portal/customer-reviews/ASIN` 瀑布流）。先读 `references/reviews-strategy.md` 顶部的账号/合规代价说明\n- 视觉化选品 fallback：把 `image` URL 喂给 `vision_analyze`，让它\"看图识物\"。参考 `references/product-form-fusion.md`\n\n```bash\n# 畅销榜（有排名，无月销，最多 60 个）\ndocker run --rm amazon-scraper node assets/amazon_handler.js \"https://www.amazon.com/gp/bestsellers/electronics\"\n\n# 搜索结果（有月销，多页可达 100+）\ndocker run --rm amazon-scraper node assets/amazon_handler.js \"https://www.amazon.com/s?k=feather+duster\"\n\n# 产品详情（最全字段：BSR、品牌、卖点、月销）\ndocker run --rm amazon-scraper node assets/amazon_handler.js \"https://www.amazon.com/dp/B001TQ6IHS\"\n\n# 多页爬取（搜索页建议 5 页拿 Top 100）\ndocker run --rm amazon-scraper node assets/amazon_handler.js \"URL\" --pages 2\n\n# 保存结果到文件（必须挂载 /data 才能在主机读到）\ndocker run --rm -v ~/scrapes:/data amazon-scraper node assets/amazon_handler.js \"URL\" --output result.json\n\n# 用自己的代理覆盖内置配置\ndocker run --rm -e AMAZON_PROXIES=\"http://user:***@host:8001,...\" amazon-scraper node assets/amazon_handler.js \"URL\"\n```\n\n**输出状态（必须检查，不要只看 products 长度）**\n\n| `status` | 退出码 | 含义 | 该怎么办 |\n|---|---|---|---|\n| `SUCCESS` | 0 | 所有页都抓到了 | 正常使用数据 |\n| `PARTIAL` | 3 | 部分页被拦/超时，`failures[]` 列出来了 | 可以用已拿到的，但**不要**把缺的那部分当成\"不存在\" |\n| `ERROR` | 2 / 1 | 全部失败 | 丢弃结果。先查第 0 步出口，再看是不是被软拦截 |\n\n软拦截的典型表现是 HTTP 200 + 整页没渲染，于是所有字段都是 null。现在详情页遇到这种情况会算失败并换代理重试，不会再以 `SUCCESS` 混进结果。**写入数据库/生成报表前先判 `status`** —— 把软拦截当成\"价格没变\"比没有数据更糟。\n\n**输出格式:** JSON\n```json\n{\n  \"status\": \"SUCCESS\",\n  \"failedJobs\": [],\n  \"failures\": [],\n  \"type\": \"bestsellers|search|product-detail\",\n  \"category\": \"品类名\",\n  \"totalProducts\": 30,\n  \"scrapedAt\": \"ISO时间\",\n  \"products\": [\n    {\n      \"rank\": 1,\n      \"title\": \"产品名\",\n      \"asin\": \"B001TQ6IHS\",\n      \"price\": 9.94,\n      \"priceStr\": \"$9.94\",\n      \"rating\": 4.6,\n      \"reviews\": 20547,\n      \"boughtPastMonth\": \"1K+\",\n      \"image\": \"https://...\",\n      \"url\": \"https://...\",\n      \"sponsored\": false\n    }\n  ]\n}\n```\n\n### 2. 通用模式 (`main_handler.js`)\n**触发条件:** 非Amazon的URL，或用户提到爬取/抓取任意网页内容\n\n- 和 Amazon 模式同一套 Playwright（无 stealth）\n- 内置代理已预配置，无需额外设置\n- 支持 `--output` 文件保存\n- 可通过环境变量覆盖内置代理\n- Playwright打开页面，等待JS加载完成\n- 提取 `document.body.innerText`（纯文本，去广告噪音）\n- 输出上限10000字符\n- 输出: `{status:\"SUCCESS\", type:\"GENERIC\", title, data}`\n\n```bash\n# 通用爬取（代理已内置）\ndocker run --rm amazon-scraper node assets/main_handler.js \"https://任意网址\"\n\n# 保存文件\ndocker run --rm -v ~/scrapes:/data \\\n  amazon-scraper node assets/main_handler.js \"https://任意网址\" --output page.json\n```\n\n## Agent调用决策树\n\n```\n用户给了URL?\n├─ 包含 amazon.com → 用 amazon_handler.js\n│   ├─ 需要月销量? → 建议用搜索URL(/s?k=) 或详情页(/dp/)\n│   ├─ 需要排名? → 用畅销榜URL(/gp/bestsellers/)\n│   └─ 需要 Top 100? → 搜索页 --pages 5 (或看 references/bsr-top100-strategy.md)\n└─ 其他网站 → 用 main_handler.js (通用模式)\n\n用户没给URL，只说了需求?\n├─ \"爬亚马逊XX品类Top\" / \"XX类目排行\" / \"XX畅销榜\" → 构造 https://www.amazon.com/gp/bestsellers/品类\n├─ \"搜亚马逊XX\" / \"XX关键词搜索\" / \"找XX产品\" → 构造 https://www.amazon.com/s?k=关键词\n├─ \"分析某个ASIN\" / \"看看这个产品\" / \"XX的详情\" → 构造 https://www.amazon.com/dp/ASIN\n├─ \"XX的月销量\" / \"XX卖了多少\" / \"XX销量怎么样\" → 用搜索页或详情页（有boughtPastMonth）\n├─ \"竞品分析\" / \"竞品调研\" / \"对手在卖什么\" → 先搜索再逐个爬详情\n├─ \"选品\" / \"什么好卖\" / \"品类机会\" / \"市场调研\" → Best Sellers + 搜索结合\n├─ \"Top 100\" / \"Top 200\" / \"拿 100 个产品\" → /s?k=关键词 --pages 5 (一次拿 100+)\n├─ \"抓评论\" / \"获取评价\" / \"reviews\" → 需登录态 Chrome + CDP，用 scripts/scrape_reviews.py（瀑布流模式）\n└─ 其他网页 → 先web_search找到URL，再用通用模式爬\n```\n\n## 常见用户意图 → 操作映射\n\n| 用户说 | 操作 |\n|---|---|\n| \"帮我看看亚马逊XX品类\" | 爬 /gp/bestsellers/品类 畅销榜 |\n| \"XX在亚马逊卖得怎么样\" | 搜索 /s?k=XX 看月销 |\n| \"分析一下这个ASIN: BXXXXXXXXX\" | 爬 /dp/ASIN 详情页 |\n| \"XX品类有什么机会\" | 畅销榜 + 搜索 综合分析 |\n| \"帮我爬这个链接\" | 判断URL类型，选对应handler |\n| \"帮我抓XX网站的内容\" | 通用模式 |\n| \"搜一下XX的竞品\" | 搜索页爬取 + 分析 |\n| \"XX月销多少\" / \"XX一个月卖多少\" | 搜索页或详情页 |\n| \"帮我看看top 100\" / \"热门产品\" | **/s?k=关键词 --pages 5**（BSR 单 URL 只到 60） |\n| \"新品有哪些\" / \"最近上了什么新品\" | /zg/new-releases/ |\n| \"什么产品涨得快\" / \"飙升榜\" | /zg/movers-and-shakers/ |\n| \"抓评论\" / \"获取评价\" / \"XX的评论\" | **scripts/scrape_reviews.py**（需已登录 Chrome + CDP，瀑布流模式） |\n\n## 代理配置\n\n代理配置存放于 `config/proxies.json`，格式为 JSON 数组：\n\n```json\n{\n  \"proxies\": [\n    \"http://user-XXX:password@ENTRY_POINT:PORT\",\n    \"http://user-XXX:password@ENTRY_POINT:PORT\"\n  ]\n}\n```\n\n优先级（`assets/proxy.js`，环境变量优先）：\n1. `AMAZON_PROXIES`（逗号分隔多条）\n2. `AMAZON_PROXY`（单条）\n3. `AMAZON_PROXY_FILE` 指向的文件，否则 `config/proxies.json`\n\n一条都没配就**直接报错退出**，不会悄悄用宿主机 IP 去爬（真要这么做得显式 `AMAZON_ALLOW_DIRECT=1`）。日志只打 `host:port`，不打凭证。\n\n凭证**不再烘进镜像**（`Dockerfile` 不 `COPY config/`）。运行时二选一：\n\n```bash\ndocker run --rm -e AMAZON_PROXIES amazon-scraper node assets/amazon_handler.js \"URL\"\ndocker run --rm -v \"$PWD/config/proxies.json:/app/config/proxies.json:ro\" amazon-scraper node assets/amazon_handler.js \"URL\"\n```\n\n改凭证不需要重 build，只有改 `assets/` 才需要。协议只接受 `http://` / `https://`，`socks5://` 会在启动时报错。\n\n并发被代理条数卡死：`concurrency = min(请求值, 任务数, proxies.length)`。只有 1 条时 `--concurrency 5` 实际是 1，并且现在会打一行 WARNING 说明被降到了几。不要开 5 个容器打同一个 IP。\n\n### 入口域名速查（Oxylabs 各产品）\n\n| 产品 | 入口域名 | Amazon 适用? |\n|---|---|---|\n| Datacenter / DDC | `ddc.oxylabs.io` | ❌（被 Amazon ASN 黑名单） |\n| **ISP Proxies** ⭐ | `isn.oxylabs.io` 或 `pr.oxylabs.io` | ✅ **推荐** |\n| Residential | `pr.oxylabs.io` | ✅ |\n| Mobile | `mob.oxylabs.io` | ✅（最严目标） |\n\n> 📖 **完整产品矩阵 / 凭证格式 / 怎么从 dashboard 拿入口域名**：[references/oxylabs-product-matrix.md](references/oxylabs-product-matrix.md)\n> 📖 **DDC 故障排查 / response code 速查**：[references/oxylabs-ddc-format.md](references/oxylabs-ddc-format.md)\n> 📖 **Oxylabs Response Code + 怎么判断\"代理层通 vs 目标站拦\"**：[references/oxylabs-proxy-format.md](references/oxylabs-proxy-format.md)\n\n### ⚠️ 代理配置陷阱（来自真实踩坑）\n\n1. **不要把 dashboard 上的 \"Assigned IP\" 当入口**：那是出口 IP，不是入口。直接打 `45.x.x.x:8001` 在 VPS 上会 `No route to host`。**必须用 `ddc.oxylabs.io` 域名 + 端口**。详见 `references/oxylabs-ddc-format.md`。\n\n2. **DDC / Datacenter 代理对 Amazon 不可用**：Amazon 把整个 datacenter ASN 段都标为高风险，配置 100% 正确也会拿到 \"Sorry! Something went wrong!\" 软拦截页。**Amazon 爬取用 ISP Proxies**，不要用 DDC。详见 `references/oxylabs-product-matrix.md`。\n\n3. **凭证不区分协议前缀**：`proxies.json` 里 `http://` 和 `https://` 都能用（爬虫内部走 HTTP CONNECT）。别误用 `socks5://`。\n\n4. **凭证过期/被封 = 静默 0 results**：如果 Amazon 返回 `totalProducts: 0` 且无明显错误，第一反应是代理被 Amazon 软拦截（datacenter）或凭证挂了。先单独跑 `curl -x ddc.oxylabs.io:PORT -U user:pass http://ip.oxylabs.io/location` 验证代理层通不通。\n\n5. **改 `proxies.json` 不再需要重 build**：`Dockerfile` 已经不 `COPY config/`，凭证走 `-e AMAZON_PROXIES` 或挂载 `/app/config/proxies.json`。改 `assets/` 下的代码仍然要 `docker build -t amazon-scraper <skill 目录>`。\n6. **`--output` 是容器内路径**：必须 `-v /host/path:/data` 挂载，否则文件在容器内拿不到。\n7. **`Accept-Encoding: identity` 头导致详情页加载失败**：原 `createContext` 的 `extraHTTPHeaders` 和 `setExtraHTTPHeaders` 里设了 `Accept-Encoding: identity`，会导致 Amazon 新版详情页 JS 不渲染（`#productTitle` 等 selector 全部 timeout）。已移除该头和整个 `setExtraHTTPHeaders` 调用，详情页恢复正常。\n8. **新版 Amazon 详情页 DOM 结构变化**（2026年7月实测）：\n   - BSR 不在 `body.innerText` 正则里了，在 `#prodDetails tr` 的 `th/td` 行里（key=\"Best Sellers Rank\", val=\"#84 in Electronics...\"）\n   - brand 从 `#bylineInfo` 拿到的是 \"Brand: XXX\" 或 \"Visit the XXX Store\"，需清理前缀和后缀\n   - seller 从 `#sellerProfileTriggerId` 提取（如 \"BeataTap-SKALON\"），但部分 ASIN 的 `#sellerProfileTriggerId` 会抓到 Amazon 比价组件的垃圾文本（\"different sellers.\\nShow details...\"），需在分析层过滤\n   - `dateFirstAvailable` 在新版页面 **100% 缺失**（Amazon 不再展示此字段），详情页正则和 `#prodDetails tr` 都拿不到\n   - 详情页需要 `waitForSelector('#productTitle, #dp', { timeout: 15000 })` + `waitForTimeout(3000)` 才能拿到完整数据\n9. **datacenter 代理对 Amazon 详情页 100% 软拦截**（返回空页面），ISP 代理正常。用 `isp.oxylabs.io` 这类 ISP/住宅入口，不要用 `ddc.*` / `disp.*`。\n10. **搜索页 `--pages 5` 实际只返回 ~26 个唯一 ASIN**（不是 100）：Amazon 搜索结果含大量广告/重复，拿真正 Top 100 需要更多页或关键词变体。\n11. **搜索页翻页参数是 `&page=N` 不是 `&pg=N`**（2026-07-04 修复）：原代码用 `&pg=N` 翻页，Amazon 不认这个参数，每页都返回第一页数据（全是重复），所以 `--pages 10` 只拿到 29 个唯一 ASIN。改成 `&page=N` 后 10 页拿到 234 个唯一 ASIN。如果发现多页爬取结果去重后远少于预期，先检查翻页参数。\n12. **批量只能按当前代理条数来**：并发被代码卡成 `min(请求值, 任务数, 代理条数)`，只有 1 条出口时开 5 个容器就是 5 个浏览器打同一个 IP。超过约 10 分钟的批量必须后台跑（前台 600s 会杀掉容器）。要并行，先加不同出口 IP，再给每个容器 `-e AMAZON_PROXIES` 指定各自的出口。合并按 `_source`，不按返回的 `asin`。\n13. **CDP Fallback 补漏方案（2026-07-04 验证）**：即使安全加速模式仍有 ~25% ASIN 被软拦截返回 null。**解法**：用本地 Chrome CDP（port 9222）串行补爬失败的 ASIN，成功率 100%。脚本 `scripts/cdp_fallback_scrape.py`，详见 `references/cdp-fallback-strategy.md`。完整流程：Docker 批量（快但有失败）→ 提取失败 ASIN → CDP 串行补漏（慢但 100% 成功）→ 合并结果。\n\n## 反爬能力\n- **出口**：按配置的代理条数轮询；失败换下一条；页间隔 1.5s\n- **UA**：Linux Chrome 119，对齐 Playwright 1.40 内核（不再写 Mac/Win）\n- **端口绑死**：分辨率 / 时区 / 核数；cookie 落 `/data/fingerprint-state/{namespace}/{port}.json`（需 `-v ~/scrapes:/data`）。通用模式按目标域名分 namespace，不和亚马逊共用一个 cookie 罐\n- ⚠️ `BY_PORT` 的指纹表只覆盖端口 8001–8005。代理端口不在这个范围时会走 host 哈希的 fallback，所谓\"同端口同指纹\"实际没生效 —— 换代理商后要同步更新这张表\n- **先开首页再跳目标**，同一 page 带 referer\n- **不授权 geolocation**\n- **无 stealth 插件**（特征库公开，已关掉做 A/B）\n- **WebRTC**：`--force-webrtc-ip-handling-policy=disable_non_proxied_udp`，API 在，不走宿主机 UDP\n- 轻量 `mouse.move`，不是拟人鼠标库\n- Docker 用完即毁；`--no-sandbox` 仍在（容器必需）\n\n这不是红手指分控箱：没有一人一号、没有真实字体/显卡。\n\n## 局限\n- 通用模式输出上限10000字符\n- Amazon BSR 单 URL 最多 60 个产品（2 页限制），拿 Top 100 用搜索页\n- Amazon 单页搜索结果最多约 30-50 个产品\n- 不支持需要登录的页面\n- Docker 容器启动有冷启动时间（Playwright Chromium）\n- Amazon 对 datacenter 代理段主动风控，**Amazon 爬取必须用 ISP / Residential / Mobile 代理**\n- **评论抓取不在 Docker 内**：需已登录的 Chrome + CDP（`scripts/scrape_reviews.py`），非匿名页面，详见 `references/reviews-strategy.md`\n- **ASIN 详情页返回所有字段为 null**：`status: SUCCESS` + `products[0]` 存在但 `title`/`price`/`rating` 全为 null。两种原因：(1) ASIN 已下架/不可用（换一个有效 ASIN 验证即可区分）；(2) 代理被亚马逊软拦截（503/captcha 页）。用 `curl -x <proxy> -s -o /dev/null -w \"%{http_code}\" \"https://www.amazon.com/dp/<ASIN>\"` 检查，正常应返回 200，503 = 代理被拦\n\n## 相关 references\n- `references/new-layout-fixes-2026-07.md` — Amazon 新版页面 DOM 变化 + 修复详情（seller 字段、BSR 提取、brand 清理、proxy 切换）\n- `references/oxylabs-product-matrix.md` — 哪个 Oxylabs 产品对应哪个入口域名，Amazon 该买哪个\n- `references/oxylabs-ddc-format.md` — DDC 故障排查、response code 速查、\"代理通 vs 目标拦\"判别\n- `references/oxylabs-proxy-format.md` — 通用 Oxylabs 代理格式 + 验证命令\n- `references/bsr-top100-strategy.md` — BSR 60 个上限 + 拿 Top 100 的三种策略\n- `references/search-page-title-fix.md` — 搜索页 `title: null` 根因 + 已合入的修复 + 重 build 步骤\n- `references/product-form-fusion.md` — 视觉化选品 / 产品形态融合创意（vision_analyze fallback）\n- `references/sharing-checklist.md` — 打包分享前的凭证安全检查\n- `references/reviews-strategy.md` — 评论页抓取策略（瀑布流模式 + CDP page WS + 选择器参考）\n- `references/batch-parallel-scraping.md` — 批量详情页并行爬取模式（5 容器 × 独立代理端口 × 低并发）\n- `scripts/scrape_reviews.py` — 可直接运行的评论抓取脚本（CDP + 瀑布流 + SQLite 存储）\n- `scripts/asin_price_monitor.py` — 每日竞品ASIN价格监控脚本，设计用于 cron job（no_agent=True），stdout 推送到 Discord/Telegram。含价格变化检测（对比历史数据）。修改 ASINS 列表增删监控目标\n- `scripts/cdp_fallback_scrape.py` — CDP 补漏脚本：当 Docker 代理批量爬取被软拦截时，用本地 Chrome CDP 串行补爬失败 ASIN，成功率 100%。详见 `references/cdp-fallback-strategy.md`\n- `scripts/pickleball_price_monitor.py` — 多ASIN价格监控模板（Docker爬虫+SQLite+飞书多维表格+Discord推送），设计用于 cron no_agent 模式\n- `references/cron-price-monitor-pattern.md` — Cron价格监控集成模式：Hermes cron + 飞书多维表格 + Discord，含时区陷阱和 lark-base 字段类型注意事项\n- `references/cdp-fallback-strategy.md` — CDP Fallback 完整策略：前置条件、使用方法、Docker+CDP 两阶段工作流\n\nFile v4.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn7ewmmms6dthpk632rzrc05v981ah4f\",\n  \"slug\": \"amazon-scraper\",\n  \"version\": \"4.0.0\",\n  \"publishedAt\": 1790925924089\n}\n\nFile v4.0.0:references/batch-parallel-scraping.md\n\n# Batch Parallel Scraping Pattern\n\n**当前不适用。** 活代理只有 1 条（出口 IP 和端口见你本地的 `config/proxies.json`，不要写进文档）。代码把并发卡成 `min(请求值, 任务数, proxies.length)`，所以 `--concurrency 5` 仍是 1。`-e AMAZON_PROXIES` 在 `proxies.json` 非空时不生效。下面的 5 容器 × Oxylabs 8001–8005 是旧方案，现在执行会让 5 个浏览器打同一个 IP。\n\n大批量在只有 1 条 IP 时：一个容器，`--asins`，`--output` 必须配 `-v 宿主机目录:/data`，超过约 10 分钟改后台。先用 `http://api.ipify.org` 确认出口 IP 等于你配置的代理 host，再跑亚马逊。裸 curl 拿到 500/202 不算爬虫失败，以 handler 的 JSON 为准。\n\n旧方案（要有 5 个不同出口才能用，且必须挂载覆盖 `/app/config/proxies.json`，不能靠环境变量）：\n\nSafe high-throughput detail page scraping using multiple Docker containers with dedicated proxy ports.\n\n## Problem\nSingle container `--asins \"100_asins\" --concurrency 5` is slow (~20 min for 100 ASINs). But naive parallelism (5 containers × concurrency 5 = 25 total concurrent requests) causes ISP proxy soft-blocks — 76% of ASINs return all-null data.\n\n## Safe Pattern: 5 Containers × 2-3 Concurrency\n\n```bash\nPROXY_USER=\"user-XXX\"\nPROXY_PASS=\"XXX\"\n\n# Split 100 ASINs into 5 batches of 20\n# Each batch → its own Docker container with a dedicated proxy port\n\nfor i in 0 1 2 3 4; do\n  PORT=$((8001 + i))\n  BATCH=\"<20 ASINs comma-separated>\"\n  docker run --rm -v ~/scrapes:/data \\\n    -e AMAZON_PROXIES=\"http://${PROXY_USER}:${PROXY_PASS}@isp.oxylabs.io:${PORT}\" \\\n    amazon-scraper node assets/amazon_handler.js \\\n    --asins \"$BATCH\" --concurrency 2 \\\n    --output \"batch_${i}.json\" \\\n    > \"batch_${i}.log\" 2>&1 &\ndone\nwait\n```\n\n## Key Parameters\n\n| Parameter | Safe Value | Risk Value | Notes |\n|---|---|---|---|\n| Containers | 5 | >8 | One per proxy port (8001-8005) |\n| Concurrency per container | 2-3 | ≥5 | Each proxy port handles 2-3 concurrent connections |\n| Total concurrent | 10-15 | ≥25 | >15 triggers Amazon soft-blocks on ISP proxies |\n| ASINs per batch | 20 | >30 | Bigger batches = longer single-container runtime |\n\n## Performance\n\n| Mode | 100 ASINs | Success Rate |\n|---|---|---|\n| Single container, concurrency 5 | ~20 min | ~90% |\n| 5 containers × concurrency 3 | ~5 min | ~25% (too aggressive) |\n| 5 containers × concurrency 2 | ~5-7 min | ~25% (still aggressive on retry) |\n| 5 containers × concurrency 2, then retry failed with concurrency 1 | ~10 min | ~50% cumulative |\n\n## Retry Strategy for Failed ASINs\n\nAfter the first pass, collect ASINs that returned all-null (proxy soft-blocked), then retry with lower concurrency:\n\n```python\n# Collect failed ASINs\nfor p in all_products:\n    if not (p.get('title') or p.get('brand')):\n        failed_asins.append(p['asin'])\n```\n\nRetry with `--concurrency 1` or `--concurrency 2` on the same 5-container pattern. Typical improvement: +10-15% success on retry.\n\n## Merging Results\n\nAfter all batches + retries complete, merge by ASIN — prefer rows with actual data (title/brand not null):\n\n```python\ndetail_map = {}\nfor prefix in ['batch_', 'retry_']:\n    for i in range(5):\n        d = json.load(open(f'{prefix}{i}.json'))\n        for p in d['products']:\n            if p.get('title') or p.get('brand'):  # has real data\n                detail_map[p['asin']] = p\n```\n\n## When This Pattern is Needed\n- 100+ ASIN detail pages needed (monopoly analysis, competitive research)\n- Time constraint (user wants results in <10 min, not 20+)\n- Search page data already collected (just need brand/seller/bsr/bullets/details)\n\n## When to Use Single Container Instead\n- <30 ASINs (single container concurrency 3-5 is fine)\n- No time pressure\n- Want maximum success rate per ASIN\n\nFile v4.0.0:references/bsr-top100-strategy.md\n\n# Amazon BSR Top 100 抓取策略\n\n## 核心限制（必读）\n\n**`/gp/bestsellers/` URL 只能拿到 2 页 = 60 个产品。** Amazon 官方限制。`?pg=3` 会返回 \"Page Not Found\"。\n\n这意味着：\n- 拿 Top 30 ✅ 直接 `/gp/bestsellers/{category}`\n- 拿 Top 50/60 ✅ `/gp/bestsellers/{category} --pages 2`\n- **拿 Top 100 ❌ 单个 BSR URL 不行**\n\n## 三种 Top 100 策略\n\n### 策略 A：搜索页替代（最快、推荐）⭐\n\n```bash\ndocker run --rm -v /tmp/top100:/data amazon-scraper \\\n  node assets/amazon_handler.js \\\n  \"https://www.amazon.com/s?k=cable+management\" \\\n  --pages 5 \\\n  --output /data/cm.json\n```\n\n- 一次调用，~3 分钟\n- 5 页 = 100+ 个产品（去重后约 60-80 个独立 ASIN）\n- **数据是搜索算法排序的，不是 BSR 严格排名**（混了广告位）\n- 对\"市场分析\"够用\n\n**适合**：快速拿数据做品类分析、价格带分析、视觉调研。\n\n### 策略 B：多子类目 BSR 合并\n\nBSR 父类目下钻到 5 个子类目，每个拿 Top 30 = 150 个产品：\n\n```python\n# 主类目 electronics 没有子节点\n# 子类目节点 ID（在 URL 里能看到）\nsubcats = [\n    \"electronics/172541\",      # Audio & Video\n    \"electronics/281407\",      # Computers & Accessories\n    \"electronics/2407745011\",  # Wearable Technology\n    \"electronics/13896617011\", # Computer & Accessories\n    \"electronics/3024167031\",  # Cell Phones\n]\n# 拼 URL: https://www.amazon.com/gp/bestsellers/{subcat}\n```\n\n每个子 BSR 限 2 页 = 30 个。**5 × 30 = 150，去重 ~120 个**。\n\n**适合**：要做严格的\"畅销榜\"分析（不被广告位污染）。\n\n### 策略 C：单 session 串行（最稳但最慢）\n\n把 BSR Top 30 拿到 ASIN，逐个爬详情，3 个代理并发：\n\n```bash\n# /tmp/top100.sh\nwhile read asin; do\n  docker run --rm amazon-scraper node assets/amazon_handler.js \\\n    \"https://www.amazon.com/dp/${asin}\" \\\n    --output \"/tmp/details/${asin}.json\" 2>/dev/null\n  sleep 3\ndone < /tmp/asins.txt\n```\n\n**耗时**：30 个 ASIN × 15-25 秒 = 7-12 分钟（单容器，1 代理）\n\n**适合**：要拿详情做品牌/BSR/详情页分析。\n\n## 提速方案\n\n| 方案 | 速度 | 限制 |\n|---|---|---|\n| 单容器 `--pages 5` | 1x | skill 本身 |\n| 多 Docker 容器并发 | Nx（N=容器数） | 需 N 个不同代理 |\n| 改 handler.js 加并发 | 内部可控 | 需重新 build 镜像 |\n\n**多容器并发的现实约束**：你的代理数 = 最大并发数。\n- 3 个 DDC IP → 最多 3 个并发 → 3x 提速\n- 5 个 ISP 端口 → 最多 5 个并发 → 5x 提速\n\nskill 自带轮询：handler 内部已支持 1 个容器内多代理轮询 + 故障切换，无需额外配置。\n\n## 速度与限流\n\n- 单个 session：~30 秒/页（含 15 秒冷启动 + 3-5 秒 waitFor）\n- Amazon 限流：30-60 请求/小时/同一 IP 是安全线，超了会触发 503\n- 跑 Top 100 一次消耗约 5-10 个\"请求单位\"\n\n## 输出文件路径\n\n⚠️ **坑**：`--output` 是**容器内路径**。要保存到主机必须挂载卷：\n\n```bash\n# 错：文件在容器里\ndocker run --rm amazon-scraper node .../amazon_handler.js \"URL\" --output /tmp/x.json\n\n# 对：挂载 /data\ndocker run --rm -v /tmp/results:/data amazon-scraper node .../amazon_handler.js \\\n  \"URL\" --output /data/x.json\n# 文件在主机的 /tmp/results/x.json\n```\n\n代码里 `--output` 的路径会拼到 `/data/` 前缀下。\n\nFile v4.0.0:references/cdp-fallback-strategy.md\n\n# CDP Fallback 爬取策略\n\n## 问题场景\n\n批量爬取 Amazon 详情页时（`--asins` 模式），Docker 代理方案在以下情况会被 Amazon 软拦截：\n- 总并发 ≥ 15（5 容器 × concurrency 3）\n- 单代理端口并发 ≥ 3\n- 短时间内同 IP 大量请求\n\n软拦截表现：`status: SUCCESS` 但 `title/brand/seller/bsr` 全部 null，页面未渲染。\n\n## 解决方案：Chrome CDP 直连\n\n**核心思路**：绕过 Docker 代理，用 VPS 本地 Chrome 浏览器（CDP 协议）逐个串行爬取，配合 2-4 秒随机延迟。\n\n**实测结果（2026-07-04）**：\n- Docker 代理方案：75 个失败 ASIN，成功率 0%（全部被拦）\n- CDP 直连方案：75 个失败 ASIN，**成功率 100%**（0 失败）\n- 耗时：75 个 ASIN × ~5 秒/个 = 约 6 分钟\n\n## 前置条件\n\n1. Chrome 已启动并监听 CDP：\n\n```bash\nexport DISPLAY=:99\nXvfb :99 -screen 0 1920x1080x24 &>/dev/null &\n/opt/google/chrome/chrome --disable-gpu --no-first-run \\\n  --no-default-browser-check \\\n  --remote-debugging-port=9222 \\\n  --remote-debugging-address=127.0.0.1 \\\n  --user-data-dir=\"$HOME/.cache/amazon-scraper-chrome\" \\\n  \"https://www.amazon.com\"\n```\n\n> ### ⚠️ 不要加 `--remote-allow-origins=*`\n>\n> CDP **没有任何认证机制**：谁能连上 9222，谁就完全控制这个浏览器 —— 读 cookie、以登录用户身份下单、改收货地址、导出 session。本方案的前提恰恰是这个 Chrome 带着真实 Amazon 登录态，所以它的调试口就等于账号凭证。\n>\n> - `--remote-allow-origins=*` 关掉了 WebSocket 的 Origin 校验，于是**你在这个浏览器里打开的任意网页**都能连上 9222 接管它。只在确实遇到 Origin 报错时，针对具体来源写白名单，不要用 `*`。\n> - 必须显式 `--remote-debugging-address=127.0.0.1`。绑到 `0.0.0.0` 的 VPS 等于开了一个公网无密码浏览器后门；即使绑回环，也要确认没有端口转发或 docker 规则把它暴露出去（`ss -lntp | grep 9222` 自查）。\n> - 不要用 `--no-sandbox`。VPS 上常以 root 跑 Chrome，关掉沙箱意味着一个渲染器漏洞就能拿到 root。需要在容器里跑就改用非 root 用户 + `--user-ns` 之类的方案。\n> - profile 不要放 `/tmp`（全局可写，其他本地用户可读你的 cookie）。放 `$HOME/.cache/...` 并保持 `chmod 700`。\n> - 用完把这个 Chrome 关掉，别长期挂着一个带登录态的调试口。\n\n本目录的两个 CDP 脚本都会**自己新开一个标签页**并在结束时关掉，不会劫持你正在用的标签页。\n\n2. Python 依赖：\n```bash\npip install websocket-client\n```\n\n## 使用方法\n\n```bash\n# 方式1：直接传 ASIN 列表\npython3 ~/.hermes/skills/amazon-scraper/scripts/cdp_fallback_scrape.py \\\n  --asins \"B07XXX,B08YYY,B09ZZZ\" \\\n  --output cdp_results.json\n\n# 方式2：从 JSON 文件读 ASIN 列表\npython3 ~/.hermes/skills/amazon-scraper/scripts/cdp_fallback_scrape.py \\\n  --asin-file failed_asins.json \\\n  --output cdp_results.json\n```\n\n## 完整工作流：Docker 批量 + CDP 补漏\n\n```bash\n# Step 1: Docker 批量爬取（快但有失败）\n# 每个容器一个独立出口端口。没有 N 个不同出口就不要开 N 个容器。\n# 凭证从环境变量取，不要写进命令行（会进 shell history 和 ps 输出）。\nread -rsp 'proxy user: ' PROXY_USER; echo\nread -rsp 'proxy pass: ' PROXY_PASS; echo\n\nfor i in 0 1 2 3 4; do\n  PORT=$((8001 + i))\n  AMAZON_PROXIES=\"http://${PROXY_USER}:${PROXY_PASS}@isp.oxylabs.io:${PORT}\" \\\n  docker run --rm -v ~/scrapes:/data \\\n    -e AMAZON_PROXIES \\\n    amazon-scraper node assets/amazon_handler.js \\\n    --asins \"$BATCH_$i\" --concurrency 2 \\\n    --output \"batch_${i}.json\" &\ndone\nwait\n\n# Step 2: 找出失败的 ASIN（title/brand 全 null）\npython3 -c \"\nimport json\nfailed = []\nfor i in range(5):\n    d = json.load(open(f'batch_{i}.json'))\n    for p in d.get('products', []):\n        if not (p.get('title') or p.get('brand')):\n            failed.append(p['asin'])\njson.dump(failed, open('failed_asins.json', 'w'))\nprint(f'{len(failed)} failed ASINs')\n\"\n\n# Step 3: CDP 补漏（串行，100% 成功率）\npython3 ~/.hermes/skills/amazon-scraper/scripts/cdp_fallback_scrape.py \\\n  --asin-file failed_asins.json \\\n  --output cdp_results.json\n\n# Step 4: 合并所有结果\npython3 -c \"\nimport json\nall_details = {}\nfor i in range(5):\n    d = json.load(open(f'batch_{i}.json'))\n    for p in d.get('products', []):\n        if p.get('title') or p.get('brand'):\n            all_details[p['asin']] = p\nfor p in json.load(open('cdp_results.json')):\n    if p.get('title') or p.get('brand'):\n        all_details[p['asin']] = p\njson.dump(list(all_details.values()), open('merged_details.json', 'w'), ensure_ascii=False, indent=2)\nprint(f'Merged: {len(all_details)} ASINs with full data')\n\"\n```\n\n## 为什么 CDP 方案不会被拦截\n\n1. **真实浏览器指纹**：Chrome 150 的真实 User-Agent、Canvas、WebGL 指纹，比 playwright-extra stealth 更难检测\n2. **VPS 本地 IP**：不走代理，用 VPS 的数据中心 IP 直连。虽然 Amazon 对 datacenter IP 有风控，但**单 IP 低频串行请求（每 3-5 秒一个）不会触发**\n3. **已登录态**：Chrome profile 可能有 Amazon 登录 cookie（如果之前登录过），进一步降低风控。\n\n   ⚠️ 这条是双刃剑，而且代价不对称：带登录态抓取会把抓取行为直接绑到一个真实账号上。被判定为自动化访问时，封的是这个买家账号（连带历史订单、Prime、礼品卡余额），如果它和卖家账号有关联信息，还可能牵连卖家账号。详情页（`/dp/`）匿名可见，**不需要登录态就能抓，就不要用登录态抓**。只有评论页这种必须登录的场景才值得权衡，并且应该用一个专用的、与主业务无关的账号。\n4. **CDP 协议**：通过 Chrome DevTools Protocol 控制真实浏览器，不是 headless 模式\n\n## 注意事项\n\n- **串行慢但稳**：每个 ASIN 约 5 秒（3 秒页面加载 + 2-4 秒随机延迟），100 个 ASIN 约 8-10 分钟\n- **不要并发**：CDP 方案的核心优势就是串行低频，并发会破坏这个优势\n- **Chrome 需保持运行**：脚本运行期间不要关闭 Chrome 或断开 Xvfb\n- **进度保存**：每 10 个 ASIN 自动保存一次结果到 output 文件\n- **适用范围**：仅适用于详情页（`/dp/ASIN`），搜索页/BSR 页仍用 Docker 方案\n\nFile v4.0.0:references/cron-price-monitor-pattern.md\n\n# Cron 价格监控集成模式\n\n> Hermes cron + amazon-scraper + 飞书多维表格 + Discord 推送\n\n## 架构\n\n```\nHermes Cron (每6小时)\n  └─ no_agent=True, deliver=discord\n  └─ script: pickleball_price_monitor.py\n      ├─ Docker amazon-scraper 抓取ASIN价格\n      ├─ SQLite 存储历史价格\n      ├─ 飞书多维表格写入每次检查记录 (lark-cli base +record-batch-create)\n      └─ 价格变化时 → stdout → Discord推送\n          无变化时 → 静默 (empty stdout, 不推送)\n```\n\n## 创建飞书多维表格\n\n```bash\nlark-cli base +base-create --name \"监控表名\" --table-name \"价格记录\" \\\n  --fields '[\n    {\"name\":\"ASIN\",\"type\":\"text\"},\n    {\"name\":\"商品标题\",\"type\":\"text\"},\n    {\"name\":\"当前价格\",\"type\":\"number\"},\n    {\"name\":\"上次价格\",\"type\":\"number\"},\n    {\"name\":\"价格变化\",\"type\":\"number\"},\n    {\"name\":\"变化幅度\",\"type\":\"text\"},\n    {\"name\":\"检查时间\",\"type\":\"datetime\"},\n    {\"name\":\"状态\",\"type\":\"text\"}\n  ]' --as user\n```\n\n返回 `base_token` 和 `table.id`，填入脚本的 `BASE_TOKEN` 和 `TABLE_ID`。\n\n## 飞书写入格式 (lark-cli base +record-batch-create)\n\n`--json` 必须是 `{\"fields\":[...],\"rows\":[[...]]}` 格式，不是对象数组：\n\n```python\nbatch_json = json.dumps({\n    \"fields\": [\"ASIN\", \"商品标题\", \"当前价格\", \"上次价格\", \"价格变化\", \"变化幅度\", \"检查时间\", \"状态\"],\n    \"rows\": [\n        [\"B0F6XSV7XB\", \"商品标题\", 42.99, 45.99, -3.00, \"↓ -6.5%\", 1720000000000, \"降价\"],\n        # ...\n    ]\n}, ensure_ascii=False)\n```\n\n- `检查时间` datetime 字段需**毫秒时间戳**（秒级 × 1000）\n- `ensure_ascii=False` 保留中文\n- 单批最多 200 行\n\n## 创建 Cron 任务\n\n```bash\nhermes cron create \"every 6h\" \\\n  --name \"Pickleball价格监控\" \\\n  --script pickleball_price_monitor.py \\\n  --no-agent \\\n  --deliver discord\n```\n\n或通过 cronjob 工具：\n```\ncronjob(action=\"create\", schedule=\"every 6h\", deliver=\"discord\", \n        no_agent=True, script=\"pickleball_price_monitor.py\", name=\"监控名\")\n```\n\n## ⚠️ Hermes Cron 时区陷阱\n\n**Hermes cron 用 UTC 时间，不是北京时间！**\n\n| 用户说 | 错误配置 | 正确配置 (UTC) | 实际北京时间 |\n|---|---|---|---|\n| \"每天9点\" | `0 9 * * *` | `0 1 * * *` | 09:00 |\n| \"每天凌晨3点\" | `0 3 * * *` | `0 19 * * *` (前一天) | 03:00 |\n| \"每天20:30\" | `30 20 * * *` | `0 12 * * *` | 20:30 |\n| \"每6小时\" | `every 6h` | `every 6h` | 每6小时（不受时区影响） |\n\n公式：**UTC = 北京时间 - 8小时**\n\nduration 格式（`every 6h`, `30m`）不受时区影响，只有 cron 表达式（`0 9 * * *`）才需要换算。\n\n## no_agent 模式行为\n\n- `no_agent=True`：纯脚本执行，零 LLM token 消耗\n- 脚本 stdout 非空 → 推送到 Discord\n- 脚本 stdout 为空 → **静默**，不推送（watchdog 模式）\n- 脚本 stderr → 日志，不推送\n- Hermes 有 3 分钟硬中断保护\n\n## 修改监控 ASIN 列表\n\n编辑脚本顶部的 `ASINS` 列表：\n```python\nASINS = [\"B0F6XSV7XB\", \"B0FTQWG86Q\", \"B0G6CTNVQT\"]\n```\n\n修改后无需重启 cron，下次执行自动生效（脚本每次运行时读取）。\n\n## 依赖\n\n- `amazon-scraper` Docker 镜像（已构建）\n- `lark-cli` 已安装并认证（`lark-cli auth login`）\n- Python 3.11+（sqlite3 内置）\n\nFile v4.0.0:references/full-extract-mode.md\n\n# Full Extract Mode — 43字段全量抓取\n\n> `scripts/full_extract.js` — 独立于 `amazon_handler.js` 的全量提取脚本\n\n## 什么时候用\n\n- 用户说\"爬取页面里所有内容\"、\"最大化字段\"、\"所有可见字段\"、\"完整抓取\"\n- 需要原来 15 个字段之外的数据：划线价、评分分布、全部图片、视频、A+内容、评论摘要、变体、优惠券、交付时间等\n- 需要分析竞品 Listing 全貌（不只看价格/评分）\n\n## 用法\n\n```bash\ndocker run --rm \\\n  -v /root/.hermes/skills/amazon-scraper/scripts/full_extract.js:/app/full_extract.js \\\n  amazon-scraper node /app/full_extract.js \"https://www.amazon.com/dp/ASIN/\"\n```\n\n## 抓取的 43 个字段\n\n| 分类 | 字段 | 说明 |\n|---|---|---|\n| **基本信息** | title, asin, brand, brandUrl | 标题/ASIN/品牌/品牌店铺链接 |\n| **价格** | prices.main, prices.listPrice, prices.dealPrice, prices.buybox, prices.allPriceElements | 主价/划线价/优惠价/buybox/所有价格元素 |\n| **评分评论** | rating, ratingText, reviews, reviewsText, ratingHistogram | 评分/评分数/评分分布直方图 |\n| **图片** | images[main/thumbnail/hires/aplus] | 全部图片分类标记 |\n| **视频** | videos[video] | 含poster帧 |\n| **五点描述** | bullets, bulletsCount | 完整五点 |\n| **A+内容** | aplusContent, aplusLength | A+文本和长度 |\n| **详情表** | details (20+ keys), detailsCount | 技术规格键值对 |\n| **BSR** | bsr, bsrFullText | BSR数字+含分类排名全文 |\n| **分类** | category, categoryFull | 面包屑/完整路径 |\n| **销量** | boughtPastMonth | 月销量 |\n| **日期** | dateFirstAvailable | 上架日期（常为null） |\n| **卖家** | seller, shipsFrom, fulfilledBy | 卖家/发货方/FBA |\n| **库存** | availability, isPrime, delivery | 库存状态/Prime标记/预计送达 |\n| **促销** | coupon, subscribeAndSave | 优惠券/订阅优惠 |\n| **变体** | variations (allOptions, allAsins) | 颜色/尺寸选项+变体ASIN |\n| **推荐** | frequentlyBoughtTogether, carousels | FBT/关联推荐 |\n| **Q&A** | customerQA | 客户问答 |\n| **评论** | topReviews[8] | 顶部评论（作者/评分/标题/日期/正文/verified/有用数） |\n| **安全** | safetyWarning, newerVersion | 安全警告/新版提示 |\n| **元数据** | metaTags, canonicalUrl, pageUrl | 页面meta/规范URL |\n| **结构化** | structuredData (JSON-LD) | 结构化数据 |\n| **全文** | fullPageText | 整页纯文本（截断2万字） |\n\n## 与 amazon_handler.js 对比\n\n| 维度 | amazon_handler.js | full_extract.js |\n|---|---|---|\n| 字段数 | 15 | 43 |\n| 价格 | 仅 price/priceStr | 划线价/deal价/buybox/全部价格元素 |\n| 图片 | 仅1张主图 | 全部图片（main/thumb/hires/aplus分类） |\n| 视频 | 无 | 有（含poster） |\n| 评论 | 仅 rating + reviews 数 | 8条顶部评论（含作者/日期/正文/verified） |\n| A+内容 | 无 | 有 |\n| 变体 | 无 | 有（选项+变体ASIN） |\n| 交付/库存 | 无 | availability/isPrime/delivery |\n| 优惠券 | 无 | 有 |\n| 输出大小 | ~3KB | ~85KB |\n\n## 注意事项\n\n1. **proxy日志混入stdout首行**：`full_extract.js` 的 stdout 第一行可能是 `Using proxy: ...`，解析JSON时从第一个 `{` 开始截取\n2. **容器内代理路径**：硬编码为 `/app/config/proxies.json`，不是 `__dirname/../config/`\n3. **懒加载**：需要8次滚动 + 1秒等待才能加载全部图片和A+内容\n4. **偶发超时**：`waitForSelector('#productTitle, #dp', { timeout: 15000 })` 偶尔超时，重试即可（代理轮换）\n5. **dateFirstAvailable 常为 null**：Amazon 新版页面不再展示此字段\n6. **canonicalUrl 可能指向变体ASIN**：如 `B0F6XSV7XB` 的 canonical 指向 `B0GY92641V`（父ASIN）\n\nFile v4.0.0:references/new-layout-fixes-2026-07.md\n\n# Amazon New Layout Fixes (2026-07)\n\nSession-specific details of the DOM structure changes and fixes applied to `amazon_handler.js`.\n\n## Problem: Detail pages returning all-null fields\n\n### Root Cause\n1. `Accept-Encoding: identity` in `extraHTTPHeaders` + `setExtraHTTPHeaders` caused Amazon's JS to not render the product page content. Removing the header and the entire `setExtraHTTPHeaders` call fixed it.\n2. Old `waitForTimeout(3000)` was too short for new JS-heavy pages. Added `waitForSelector('#productTitle, #dp', { timeout: 15000 })` for product-detail pages.\n\n### New Amazon Detail Page DOM Structure (July 2026)\n\n| Field | Old Selector/Regex | New Selector | Notes |\n|---|---|---|---|\n| title | `#productTitle` | `#productTitle` | Unchanged |\n| brand | `#bylineInfo` | `#bylineInfo` | Returns \"Brand: XXX\" or \"Visit the XXX Store\" — must strip prefixes |\n| seller | N/A | `#sellerProfileTriggerId` | New field, e.g. \"BeataTap-SKALON\" |\n| bsr | `body.innerText` regex | `#prodDetails tr` th/td scan | Regex `Best Sellers Rank.*?#([\\d,]+)` no longer matches; must scan `#prodDetails tr` rows |\n| dateFirstAvailable | `body.innerText` regex | N/A — **100% missing** | Amazon no longer shows this on most product pages |\n| details | `#productDetails_techSpec_section_1 tr` | `#prodDetails tr` | New layout uses `#prodDetails` instead of `#productDetails_techSpec_section_1` |\n\n### Brand Cleanup Regex\n```js\nbrand = brand.replace(/^Brand:\\s*/i, '').replace(/^Visit the\\s+/i, '').replace(/\\s+Store$/i, '').trim();\n```\n\n### BSR Extraction (New Layout)\n```js\ndocument.querySelectorAll('#prodDetails tr').forEach(row => {\n    const cells = row.querySelectorAll('th, td');\n    if (cells.length >= 2) {\n        const key = cells[0].textContent.trim().replace(/[:\\s]+$/, '');\n        const val = cells[1].textContent.trim();\n        if (key.match(/Best Sellers Rank/i)) {\n            const m = val.match(/#([\\d,]+)/);\n            if (m) bsr = parseInt(m[1].replace(/,/g, ''));\n        }\n    }\n});\n```\n\n### Seller Extraction (3 methods, in order)\n1. `#sellerProfileTriggerId` — most reliable (e.g. \"BeataTap-SKALON\")\n2. `body.innerText.match(/Sold by\\s+([\\w\\s\\-\\.]+)/i)` — fallback\n3. `#prodDetails tr` row with \"Ships from\" or \"Sold by\" key — last resort\n\n### Seller Data Garbage Filter (in analyze.py)\nSome ASINs' `#sellerProfileTriggerId` captures Amazon comparison widget text:\n```\n\"different sellers.\\nShow details\\nExplore more from across the store\\nPage 1 of 7\\n...\"\n```\nFilter rule: if seller contains `\\n`, \"Show details\", \"Explore more\", or \"Page 1 of\" → set to null.\n\n## Proxy: Datacenter → ISP\n\n| Proxy Type | Entry Domain | Amazon Works? |\n|---|---|---|\n| Datacenter (DDC) | `disp.oxylabs.io` | ❌ 100% soft-blocked (empty page) |\n| ISP | `isp.oxylabs.io` | ✅ Working |\n\nChanged `config/proxies.json` from `disp.oxylabs.io` to `isp.oxylabs.io`. Must `docker build` after changing proxies.json since it's COPY'd into the image.\n\n## Search Page Limitation\n\n`--pages 5` returns ~26 unique ASINs, not 100. Amazon search results contain many sponsored/duplicate listings. To get closer to Top 100:\n- Use more pages (--pages 10) → returns ~234 unique ASINs (enough for Top 100)\n- Or use keyword variations and merge results\n\n## Search Page Pagination Fix (2026-07-04)\n\n**Bug**: `amazon_handler.js` used `&pg=${pg}` for search page pagination. Amazon does NOT recognize `&pg=N` — it silently returns page 1 data for every page. This caused `--pages 10` to return only 29 unique ASINs (all from page 1, just duplicated 10 times).\n\n**Fix**: Changed `&pg=N` to `&page=N` in the URL construction:\n```js\n// Before (broken):\nif (pg > 1) url = job.url.includes('?') ? `${job.url}&pg=${pg}` : `${job.url}?pg=${pg}`;\n// After (fixed):\nif (pg > 1) url = job.url.includes('?') ? `${job.url}&page=${pg}` : `${job.url}?page=${pg}`;\n```\n\n**Verification**: `--pages 10` with `&page=N` returns 234 unique ASINs (vs 29 with `&pg=N`).\n\n**Diagnosis tip**: If multi-page search scraping returns far fewer unique ASINs than expected, check the pagination parameter first.\n\nFile v4.0.0:references/oxylabs-ddc-format.md\n\n# Oxylabs DDC — 工作格式与故障排查\n\n## 结论（2026-07-02 实测 + 官方文档）\n\n**Oxylabs DDC 配爬虫，必须用入口域名 `ddc.oxylabs.io` + 端口，不要直接打 \"Assigned IP\"。**\n\n| 形式 | 结果 |\n|---|---|\n| `http://user-XXX:pass@ddc.oxylabs.io:8001` | ✅ 官方推荐格式，代理层通（前提：目标站没被 Oxylabs 标 restricted） |\n| `http://user-XXX:pass@45.73.183.199:8001` | ❌ 直连 IP。VPS 上 `socket.connect()` 直接 `No route to host`（45.x.x.x 段路由不通） |\n\n每个 user 关联的 \"Assigned IP\" 在 Oxylabs dashboard 的 `My Products → Dedicated Datacenter Proxies → Proxy list` 里能看到。**那个 IP 是 Oxylabs 帮你分配的出口 IP，不是入口**。文档原文：\n\n> \"Please note that you will use ports for making requests with your IPs, meaning you will not directly access your IPs.\"\n\n> \"Entry point: The gateway to connect to your proxies — this value never changes and is always `ddc.oxylabs.io`.\"\n\n## ⚠️ 关于\"用 DDC 爬 Amazon\"本身\n\n**DDC（datacenter 代理）对 Amazon 大规模爬取基本不可用。** 不管用域名还是直连 IP，Amazon 都会：\n\n- 整段 datacenter ASN 范围（AS14061 / AS204957 / AS8100 / AS6079 等）标为高风险\n- 返回 \"Sorry! Something went wrong!\" 软拦截页面（totalProducts: 0，HTTP 200，伪装成正常）\n- 偶尔能跑通 1-2 个请求（stealth 模式 + 时机），**不可重复**\n\n`proxies.json` 配置 100% 正确 ≠ 拿到数据。**剩下的不是配置问题，是 Amazon 反爬针对 datacenter 段的问题**。\n\n**对 Amazon 真正可用的方案**（按成本升序）：\n\n1. **Oxylabs ISP Proxies** — 入口 `isn.oxylabs.io` 或 `pr.oxylabs.io:8001`，住宅 ISP 段 IP，Amazon 不拦。**2026-07-02 实测跑通 cable management Top 100**。\n2. **Oxylabs Web Scraper API** — 不是代理，是 HTTP POST API，Oxylabs 自己爬好你拿 JSON。`https://realtime.oxylabs.io/v1/queries` + `source: \"amazon_search\"`。无爬虫代码，$1.50/1K 请求。\n3. **其他 ISP 池** — Smartproxy / IPIDEA / Bright Data ISP 套餐。\n\n## 故障排查命令\n\n单独验证代理还活着（**先确认代理本身通**，再考虑目标站拦不拦）：\n\n```bash\n# 用 curl 测（注意：用 HTTP 不是 HTTPS，否则 urllib3 会误诊协议）\ncurl -x ddc.oxylabs.io:8001 -U \"user-XXX:password\" http://ip.oxylabs.io/location\n\n# 用 python requests 测\npython3 -c \"\nimport requests\np = {'http': 'http://user-XXX:pass@ddc.oxylabs.io:8001',\n     'https': 'http://user-XXX:pass@ddc.oxylabs.io:8001'}\nr = requests.get('https://ip.oxylabs.io/location', proxies=p, timeout=20)\nprint(r.status_code, r.text[:200])\n\"\n```\n\n返回 200 + JSON（如 `{\"ip\":\"205.188.202.202\",\"country\":\"US\",...}`）= 代理层通了。\n返回 503 = Oxylabs 主动拒（看 response code 速查表）。\n返回 `No route to host` = **你直接打了直连 IP**（45.x.x.x 段），改回 `ddc.oxylabs.io` 域名。\n\n## Oxylabs Response Code 速查\n\n| Code | 含义 | 怎么修 |\n|------|------|--------|\n| 400 | 请求格式错 | 检查 URL 格式 |\n| 403 | **restricted target** | 目标站（如 amazon.com）在此产品黑名单。换 ISP Proxies 或 Web Scraper API |\n| 404 | 域名解析不到 | 域名错或资源下架 |\n| 407 | 凭证错 / 没白名单 | 检查 user/pass，或在 dashboard 加白名单出口 IP |\n| 429 | 并发超限 | 降速或买更多带宽 |\n| 503 | 目标 DNS 失败 | 目标站从代理侧不可达。先直连目标确认它活着 |\n| 504 | 代理超时（60s） | 目标慢，重试或加超时 |\n\n## 历史教训\n\n**之前（2026-07-01）这个文件写的是反的**：\"用直连 IP，不要用域名\"。那次\"成功\"是 1 次侥幸（stealth 模式 + Amazon 当时没识别），不是配置正确。\n\n**2026-07-02 复测**：\n- 直连 IP `45.73.183.199:8001` → `OSError: [Errno 113] No route to host`（VPS 路由层）\n- 域名 `ddc.oxylabs.io:8001` → 503（按官方 response code 表 = 目标 DNS 失败，**不是格式问题**，是 Amazon 在 Oxylabs 黑名单）\n\n下次遇到\"代理挂了\"先分两类：\n1. 代理层通不通？→ 跑 `curl -x ddc.oxylabs.io:8001 ... http://ip.oxylabs.io/location` 验证\n2. 代理通了之后目标让不让爬？→ 看返回页是 \"Sorry! Something went wrong!\" 还是真数据\n\nFile v4.0.0:references/oxylabs-product-matrix.md\n\n# Oxylabs 代理产品矩阵 — 哪个产品对应哪个入口域名\n\n## 速查表\n\n| Oxylabs 产品 | 入口域名 | Amazon 可用? | 适用场景 | 备注 |\n|---|---|---|---|---|\n| **Datacenter Proxies** (DC) | `dc.oxylabs.io` | ❌ | 大流量、低成本、对反爬不严 | 与 DDC 不同产品 |\n| **Dedicated Datacenter Proxies** (DDC) | `ddc.oxylabs.io` | ❌ | 个人独享 IP、对反爬不严 | 文档说\"你不会直接访问 IP\" |\n| **ISP Proxies** ⭐ | `isn.oxylabs.io` 或 `pr.oxylabs.io` | ✅ **推荐** | Amazon 爬取、电商 | 住宅 ISP 段，Amazon 不拦 |\n| **Dedicated ISP Proxies** | `pr.oxylabs.io` | ✅ | 长期项目、高频爬 Amazon | 比 ISP 更稳定更贵 |\n| **Residential Proxies** | `pr.oxylabs.io` | ✅ | 任何目标站、轮换 IP | 按 GB 流量计费 |\n| **Mobile Proxies** | `mob.oxylabs.io` | ✅ | 反爬最严目标 | 4G/5G 移动 IP，最难被识别 |\n| **Web Scraper API** (非代理) | `https://realtime.oxylabs.io/v1/queries` | ✅ | 不写爬虫代码，HTTP POST 拿 JSON | $1.50/1K 请求起 |\n\n## `proxies.json` 模板\n\n### ISP Proxies（Amazon 推荐）\n```json\n{\n  \"proxies\": [\n    \"http://user-XXX:pass@isn.oxylabs.io:8001\",\n    \"http://user-XXX:pass@isn.oxylabs.io:8002\",\n    \"http://user-XXX:pass@isn.oxylabs.io:8003\"\n  ]\n}\n```\n\n### DDC（用 `ddc.oxylabs.io` 域名）\n```json\n{\n  \"proxies\": [\n    \"http://user-XXX:pass@ddc.oxylabs.io:8001\",\n    \"http://user-XXX:pass@ddc.oxylabs.io:8002\",\n    \"http://user-XXX:pass@ddc.oxylabs.io:8003\"\n  ]\n}\n```\n\n⚠️ **不要把 dashboard 上的 \"Assigned IP\" 当入口用**（如 `45.73.183.199:8001`）。VPS 上直连会 `No route to host`。\n\n### Web Scraper API（不走代理）\n不是 `proxies.json` 配的，是**直接发 HTTP POST**：\n```bash\ncurl 'https://realtime.oxylabs.io/v1/queries' \\\n  -u \"API_USER:API_PASS\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"source\":\"amazon_search\",\"query\":\"cable management\",\"parse\":true}'\n```\n返回完整 JSON（含 title/price/asin/bsr/...），无爬虫代码。\n\n## 判断\"该买哪个\"的决策\n\n| 你的情况 | 买什么 |\n|---|---|\n| 一次爬 100-1000 个 Amazon 产品 | ISP Proxies 最低套餐 |\n| 每天/每周跑 Amazon 监控 | Dedicated ISP Proxies |\n| 完全不写代码，只想要数据 | Web Scraper API |\n| 爬非 Amazon 站（普通电商/新闻） | Residential Proxies |\n| 预算紧 + 目标站反爬弱 | Datacenter Proxies |\n| 反爬最严（机票/Sneaker/抢购） | Mobile Proxies |\n\n## 切换代理的最小动作\n\n1. 改 `~/.hermes/skills/amazon-scraper/config/proxies.json`（按上面模板）\n2. 直接 `docker run` 即可（**不需要重新 build 镜像** — `proxies.json` 是运行时读取）\n\n⚠️ 反例：改 `assets/amazon_handler.js` 后才需要 `docker build -t amazon-scraper .` 重新 build。\n\n## 凭证安全\n\n`proxies.json` 里的 user/pass 是**明文凭证**。打包 skill 发给别人前先参考 `sharing-checklist.md`（先问真实/占位/空数组 三选一）。\n\nFile v4.0.0:references/oxylabs-proxy-format.md\n\n# Oxylabs Dedicated Datacenter Proxies — connection cheat sheet\n\n## Correct proxy URL format\n\nFor **Oxylabs Dedicated Datacenter Proxies (DDC, self-service)**, the proxy URL is:\n\n```\nhttp://USERNAME:PASSWORD@ddc.oxylabs.io:PORT\n```\n\n- **Entry point**: `ddc.oxylabs.io` (NOT direct IP, NOT `pr.oxylabs.io`)\n- **Port**: `8001` and up (each port = one assigned IP)\n- Username: the proxy user created in Oxylabs dashboard (e.g. `yourname_AB12C`)\n\n**Common mistake**: trying to connect to the assigned IP directly (e.g. `45.73.183.199:8001`). Oxylabs explicitly states: *\"you will not directly access your IPs\"*. Use the entry point domain, not the IP.\n\n`proxies.json` example for 3 DDC ports:\n\n```json\n{\n  \"proxies\": [\n    \"http://user-ACCOUNT_ID:PASSWORD@ddc.oxylabs.io:8001\",\n    \"http://user-ACCOUNT_ID:PASSWORD@ddc.oxylabs.io:8002\",\n    \"http://user-ACCOUNT_ID:PASSWORD@ddc.oxylabs.io:8003\"\n  ]\n}\n```\n\n## Response code quick reference (from Oxylabs docs)\n\n| Code | Meaning | What to do |\n|------|---------|------------|\n| 400 | Bad request format | Check the URL format above |\n| 403 | **Restricted target** | The site (e.g. amazon.com) is on Oxylabs' restricted list for this proxy product. Switch proxy product (e.g. ISP Proxies) or use Web Scraper API instead |\n| 407 | Auth failed or IP not whitelisted | Wrong username/password, or you switched VPS and didn't re-whitelist the new exit IP |\n| 429 | Thread / concurrent session limit exceeded | Slow down, or buy more bandwidth |\n| 503 | **DNS failure to target** | Target site unreachable from proxy. Test target directly to confirm |\n| 504 | Proxy timeout (60s) | Target is slow; retry or extend timeout |\n\n## Verifying proxy works (in 30 seconds)\n\nBefore scraping, test the proxy chain with a simple reachability check:\n\n```bash\ncurl -x ddc.oxylabs.io:8001 -U \"USERNAME:PASSWORD\" http://ip.oxylabs.io/location\n```\n\n- **200 + JSON with your assigned IP** → proxy works\n- **503** → either wrong credentials OR target DNS issue (rare; mostly 503 = restricted target like Amazon)\n- **No route to host / connection refused** → VPS can't reach Oxylabs' datacenter IP range (likely GFW / VPS provider blocking 45.x.x.x). Use the entry point domain `ddc.oxylabs.io` instead, which uses DNS to find a routable IP\n\n## When DDC doesn't work for your target (Amazon specifically)\n\nAmazon aggressively blocks datacenter ASN ranges. If you get HTTP 200 but the page says *\"Sorry! Something went wrong!\"* — that's Amazon's soft-block, not a proxy config error. The proxy is working, but Amazon has flagged the IP.\n\n**What works for Amazon** (in order of cost):\n1. **Oxylabs ISP Proxies** (`isn.oxylabs.io` or `pr.oxylabs.io:8001`) — same company, residential-grade IPs, passes Amazon\n2. **Oxylabs Web Scraper API** (HTTP POST, not a proxy) — `https://realtime.oxylabs.io/v1/queries` with `source: \"amazon_search\"` etc. — no scraping code needed, ~$1.50/1K requests\n3. **Other providers' ISP pools** (Smartproxy, IPIDEA, Bright Data ISP)\n\nDDC is the cheapest product but is fundamentally the wrong tool for Amazon-scale scraping.\n\n## Rotating through multiple proxies\n\nIn `proxies.json` you can list multiple URLs. The `amazon_handler.js` script in this skill auto-rotates through them per request and falls back on failure. No extra config needed beyond a JSON array.\n\nFile v4.0.0:references/product-form-fusion.md\n\n# Amazon 产品形态融合创意（视觉化选品思路）\n\n## 场景\n用户提出类似\"亚马逊 X 品类首页 Top 5 产品外观形态融合，能产生什么新产品\"的需求 —— 这是一种**视觉化选品** + **形态组合创意**工作流。\n\n## 典型流程\n\n### 1. 抓数据\n用 `amazon_handler.js` 抓亚马逊类目 / 搜索结果：\n\n```bash\n# 两种入口\n# (a) 类目页（标题全有）\ndocker run --rm amazon-scraper node assets/amazon_handler.js \\\n  \"https://www.amazon.com/gp/bestsellers/electronics/5180508011\"\n\n# (b) 关键词搜索（经常 title=null，备 image URL）\ndocker run --rm amazon-scraper node assets/amazon_handler.js \\\n  \"https://www.amazon.com/s?k=cable+management+organizer\"\n```\n\n### 2. 取 Top N + 提取 image URL\n```python\nimport json\ndata = json.load(open('/tmp/cm.json'))\nfor p in (data.get('products') or [])[:5]:\n    print(f\"ASIN: {p['asin']}\")\n    print(f\"Img:  {p['image']}\")\n    print(f\"Price: ${p['price']}  Rating: {p['rating']}★  Reviews: {p.get('reviews')}\")\n```\n\n### 3. 关键 fallback：搜索页 title 缺失时\n如果 `p.get('title')` 是 `None`（**搜索页极常见**），**用 `p['image']` URL + `vision_analyze` 看图识物**：\n\n```python\nvision_analyze(\n    image_url=p['image'],\n    question=\"这是一个亚马逊 [类目] 类的产品图片。请详细描述产品外观、形态、材质、颜色、尺寸感。\"\n)\n```\n\n这是**唯一可靠**识别搜索结果产品形态的方法。\n\n### 4. 形态融合创意结构\n\n每张图分析完，整理出 Top N 产品的\"形态词条\"：\n- 形态 1（扎带 / 卷状 / 尼龙）\n- 形态 2（卡扣 / 圆形 / 黑色塑料）\n- 形态 3（理线槽 / 长条 / 梳齿）\n- 形态 4（金属托盘 / 网格 / 桌下）\n- 形态 5（PVC 软槽 / 自粘 / 桌边）\n\n然后做\"5 → 1 融合\"创意，**输出 3-5 个候选产品外观方案**，每个包含：\n- 命名（一句话）\n- 形态融合说明（哪 5 个怎么组合）\n- 外观描述（材质、颜色、尺寸感、造型比喻）\n- 功能描述（解决什么用户痛点）\n- 推荐指数 + 理由（客单价 / A+ 好做度 / 安装便捷 / 视觉冲击）\n\n### 5. 推荐筛选标准\n- **客单价提升空间**：单一功能 $7-15 → 多合一 $25-40\n- **视觉冲击**：黑色磨砂金属 / 极简铝型材 > 普通黑塑料\n- **安装便捷**：磁吸 / 3M 胶 / 0 工具 = 转化率高\n- **配件生态**：模块化 = 复购\n- **目标人群**：苹果/华为桌面办公 / 电竞 RGB / 家庭办公升级\n\n## Pitfalls\n- ❌ 搜索页 `title=null` 时以为抓失败 → 实际上数据齐了，**用 image URL + vision 即可**\n- ❌ 强行用 `/s?k=...` 拿 BSR Top → 搜索页没排名概念，混淆数据\n- ❌ 创意堆砌\"什么都加上\" → 一个产品主形态要明确，最多 2-3 个融合点\n- ❌ 给传统线下/工业产品做这种融合 → 亚马逊 C 端才吃这套\n- ✅ Top 5 足够（再多了信息冗余，融合时反而抓不到重点）\n- ✅ 形态描述尽量用视觉化语言（\"鹅卵石\"、\"桌面港湾\"、\"科技树\"）便于后面做产品图\n- ✅ 推荐 Top 1 时给出\"客单价 / 视觉 / 安装 / 配件\"4 维理由，不只是\"我觉得好\"\n\n## 相关 reference\n- 主 SKILL.md \"搜索页 title 缺失\" fallback 章节\n- 选品决策树（SKILL.md \"Agent 调用决策树\"）\n\nArchive v3.4.0: 7 files, 12998 bytes\n\nFiles: assets/amazon_handler.js (18795b), assets/main_handler.js (5454b), Dockerfile.sh (247b), package.json (858b), scripts/setup.sh (2227b), SKILL.md (8096b), _meta.json (133b)\n\nFile v3.4.0:SKILL.md\n\n---\nname: amazon-scraper\ndescription: >\n  High-performance containerized Amazon scraper (Docker + playwright-extra + Stealth plugin).\n  Bypasses Amazon headless detection. Supports Amazon BSR, search results, and product detail pages. Also includes a generic mode for other dynamic web pages.\n  Use when user mentions any of these:\n  爬虫, 爬取, 抓取, 采集, 数据采集, 爬数据, 抓数据, 获取数据,\n  scrape, crawl, extract, fetch data, pull data,\n  亚马逊, Amazon, ASIN, BSR, Best Sellers, 畅销榜, 热销榜, 新品榜, 飙升榜, 排行榜,\n  选品, 竞品分析, 竞品调研, 市场调研, 品类分析, 类目分析, 产品调研,\n  月销量, bought in past month, 销量, 评论数, 价格对比,\n  网页内容, 网站数据, 页面抓取, 动态页面,\n  关键词搜索, 搜索结果, search results,\n  产品详情, 产品信息, listing数据, listing分析,\n  top 100, top sellers, 热门产品, 爆款, 跑量款,\n  价格带, 评分分布, review分析, 评论分析\n---\n\n# Amazon Scraper\n\nDocker容器化爬虫，基于 playwright-extra + Stealth 插件，专为绕过亚马逊反爬检测优化，同时支持通用动态网页爬取。\n\n## ⚙️ 系统要求\n\n- **Docker Engine 20.10+**（必须已安装并运行）\n- **磁盘空间**：~2GB（镜像 + Playwright 浏览器二进制文件）\n- **内存**：建议 2GB+（Playwright 运行时需要）\n\n## 快速开始\n\n首次使用：在 skill 目录下执行一键构建脚本：\n\n```bash\nbash scripts/setup.sh\n```\n\n脚本会自动完成：构建 `amazon-scraper` 镜像 + 创建 `~/scrapes` 输出目录。\n\n## 模式选择规则\n\n### 1. Amazon模式 (`amazon_handler.js`)\n**自动触发条件:** URL包含 `amazon.com`，或用户提到亚马逊/Amazon/ASIN/BSR/选品/竞品/畅销榜/类目分析等关键词\n\n根据URL自动识别页面类型：\n\n| URL特征 | 页面类型 | 可获取字段 |\n|---|---|---|\n| `/gp/bestsellers/` | 畅销榜 | rank, title, asin, price, rating, reviews, image, url |\n| `/zg/new-releases/` | 新品榜 | 同上 |\n| `/zg/movers-and-shakers/` | 飙升榜 | 同上 |\n| `/s?k=` 或 `/s/` | 搜索结果 | title, asin, price, rating, reviews, image, url, **boughtPastMonth**, sponsored |\n| `/dp/` 或 `/gp/product/` | 产品详情 | title, asin, price, rating, reviews, brand, bsr, **boughtPastMonth**, dateFirstAvailable, category, bullets, details, image |\n\n**⚠️ 重要规则:**\n- **Best Sellers页面没有月销量(boughtPastMonth)数据** — 亚马逊不在榜单页显示此信息\n- **要获取月销量，必须用搜索页(`/s?k=关键词`)或产品详情页(`/dp/ASIN`)**\n- 如果用户同时需要排名+月销量，建议：先爬Best Sellers拿排名，再用搜索页补月销\n- **BSR URL 必须使用 `/gp/bestsellers/`**，`/zgbs/` 会返回 Page Not Found\n\n```bash\n# 畅销榜（有排名，无月销）\ndocker run -t --rm amazon-scraper node assets/amazon_handler.js \"https://www.amazon.com/gp/bestsellers/electronics\"\n\n# 搜索结果（有月销，无排名）\ndocker run -t --rm amazon-scraper node assets/amazon_handler.js \"https://www.amazon.com/s?k=feather+duster\"\n\n# 产品详情（最全字段：BSR、品牌、卖点、月销）\ndocker run -t --rm amazon-scraper node assets/amazon_handler.js \"https://www.amazon.com/dp/B001TQ6IHS\"\n\n# 多页爬取\ndocker run -t --rm amazon-scraper node assets/amazon_handler.js \"URL\" --pages 2\n\n# 保存结果到文件\ndocker run -t --rm -v ~/scrapes:/data amazon-scraper node assets/amazon_handler.js \"URL\" --output result.json\n\n# 多代理轮询（每页自动切换代理）\ndocker run -t --rm -v ~/scrapes:/data \\\n  -e AMAZON_PROXIES=\"http://user:pass@host:8001,http://user:pass@host:8002,http://user:pass@host:8003\" \\\n  amazon-scraper node assets/amazon_handler.js \"URL\" --pages 3 --output result.json\n```\n\n**输出格式:** JSON\n```json\n{\n  \"status\": \"SUCCESS\",\n  \"type\": \"bestsellers|search|product-detail\",\n  \"category\": \"品类名\",\n  \"totalProducts\": 30,\n  \"scrapedAt\": \"ISO时间\",\n  \"products\": [\n    {\n      \"rank\": 1,\n      \"title\": \"产品名\",\n      \"asin\": \"B001TQ6IHS\",\n      \"price\": 9.94,\n      \"priceStr\": \"$9.94\",\n      \"rating\": 4.6,\n      \"reviews\": 20547,\n      \"boughtPastMonth\": \"1K+\",\n      \"image\": \"https://...\",\n      \"url\": \"https://...\"\n    }\n  ]\n}\n```\n\n### 2. 通用模式 (`main_handler.js`)\n**触发条件:** 非Amazon的URL，或用户提到爬取/抓取任意网页内容\n\n- 基于和 Amazon 模式相同的 `playwright-extra` + Stealth 架构\n- 支持同样的代理配置（`AMAZON_PROXY` / `AMAZON_PROXIES`）\n- 支持 `--output` 文件保存\n- Playwright打开页面，等待JS加载完成\n- 提取 `document.body.innerText`（纯文本，去广告噪音）\n- 输出上限10000字符\n- 输出: `{status:\"SUCCESS\", type:\"GENERIC\", title, data}`\n\n```bash\n# 通用爬取\ndocker run -t --rm amazon-scraper node assets/main_handler.js \"https://任意网址\"\n\n# 带代理 + 保存文件\ndocker run -t --rm -v ~/scrapes:/data \\\n  -e AMAZON_PROXY=\"http://user:pass@host:7777\" \\\n  amazon-scraper node assets/main_handler.js \"https://任意网址\" --output page.json\n```\n\n## Agent调用决策树\n\n```\n用户给了URL?\n├─ 包含 amazon.com → 用 amazon_handler.js\n│   ├─ 需要月销量? → 建议用搜索URL(/s?k=) 或详情页(/dp/)\n│   └─ 需要排名? → 用畅销榜URL(/gp/bestsellers/)\n└─ 其他网站 → 用 main_handler.js (通用模式)\n\n用户没给URL，只说了需求?\n├─ \"爬亚马逊XX品类Top\" / \"XX类目排行\" / \"XX畅销榜\" → 构造 https://www.amazon.com/gp/bestsellers/品类\n├─ \"搜亚马逊XX\" / \"XX关键词搜索\" / \"找XX产品\" → 构造 https://www.amazon.com/s?k=关键词\n├─ \"分析某个ASIN\" / \"看看这个产品\" / \"XX的详情\" → 构造 https://www.amazon.com/dp/ASIN\n├─ \"XX的月销量\" / \"XX卖了多少\" / \"XX销量怎么样\" → 用搜索页或详情页（有boughtPastMonth）\n├─ \"竞品分析\" / \"竞品调研\" / \"对手在卖什么\" → 先搜索再逐个爬详情\n├─ \"选品\" / \"什么好卖\" / \"品类机会\" / \"市场调研\" → Best Sellers + 搜索结合\n└─ 其他网页 → 先web_search找到URL，再用通用模式爬\n```\n\n## 常见用户意图 → 操作映射\n\n| 用户说 | 操作 |\n|---|---|\n| \"帮我看看亚马逊XX品类\" | 爬 /gp/bestsellers/品类 畅销榜 |\n| \"XX在亚马逊卖得怎么样\" | 搜索 /s?k=XX 看月销 |\n| \"分析一下这个ASIN: BXXXXXXXXX\" | 爬 /dp/ASIN 详情页 |\n| \"XX品类有什么机会\" | 畅销榜 + 搜索 综合分析 |\n| \"帮我爬这个链接\" | 判断URL类型，选对应handler |\n| \"帮我抓XX网站的内容\" | 通用模式 |\n| \"搜一下XX的竞品\" | 搜索页爬取 + 分析 |\n| \"XX月销多少\" / \"XX一个月卖多少\" | 搜索页或详情页 |\n| \"帮我看看top 100\" / \"热门产品\" | Best Sellers畅销榜 |\n| \"新品有哪些\" / \"最近上了什么新品\" | /zg/new-releases/ |\n| \"什么产品涨得快\" / \"飙升榜\" | /zg/movers-and-shakers/ |\n\n## 代理配置\n\n支持两种环境变量注入方式：\n\n| 变量 | 用途 | 格式 |\n|---|---|---|\n| `AMAZON_PROXY` | 单代理 | `http://user:pass@host:port` |\n| `AMAZON_PROXIES` | 多代理轮询 | `http://u:p@h1:8001,http://u:p@h2:8002,...` |\n\n- **轮询**：多页爬取时每页自动切换下一个代理\n- **故障切换**：单页失败时自动重试列表中下一个代理\n- `-cc-US` 用户名后缀可指定国家 IP\n- `-sessid-xxx` 可保持 sticky session\n\n## 反爬能力\n- **playwright-extra + puppeteer-extra-plugin-stealth** — 自动修改 navigator、WebGL、Canvas 等 headless 特征\n- **Chrome 123 UserAgent** — 模拟真实 Mac Chrome 浏览器\n- **完整浏览器指纹 headers** — Accept-Encoding: identity, Sec-Ch-Ua, Sec-Fetch-* 等\n- **1920x1080 viewport** — 避免移动端/小屏检测\n- 自动滚动加载懒加载内容\n- Docker沙箱隔离，每次启动全新浏览器上下文\n- 代理轮询分散请求源 IP\n\n## 局限\n- 通用模式输出上限10000字符\n- Amazon单页最多约30-50个产品\n- 不支持需要登录的页面\n- Docker容器启动有~15秒冷启动时间（含 stealth 插件初始化）\n\nFile v3.4.0:_meta.json\n\n{\n  \"ownerId\": \"kn7ewmmms6dthpk632rzrc05v981ah4f\",\n  \"slug\": \"amazon-scraper\",\n  \"version\": \"3.4.0\",\n  \"publishedAt\": 1776851706599\n}\n\nFile v3.4.0:package.json\n\n{\n  \"name\": \"amazon-scraper\",\n  \"version\": \"3.4.0\",\n  \"description\": \"High-performance containerized Amazon scraper with stealth mode. Bypasses headless detection. Supports BSR, search, product detail, and generic pages. Built on Docker + playwright-extra + Stealth.\",\n  \"main\": \"assets/amazon_handler.js\",\n  \"scripts\": {\n    \"test\": \"node assets/main_handler.js https://www.amazon.com/gp/bestsellers/electronics\",\n    \"setup\": \"bash scripts/setup.sh\"\n  },\n  \"keywords\": [\n    \"amazon\",\n    \"scraper\",\n    \"asin\",\n    \"bsr\",\n    \"playwright\",\n    \"openclaw\",\n    \"docker\"\n  ],\n  \"author\": \"Joseph\",\n  \"license\": \"MIT\",\n  \"dependencies\": {\n    \"playwright\": \"^1.40.0\",\n    \"playwright-extra\": \"^4.3.6\",\n    \"puppeteer-extra-plugin-stealth\": \"^2.11.2\"\n  },\n  \"openclaw\": {\n    \"requires\": {\n      \"docker\": true\n    },\n    \"requiredBinaries\": [\"docker\"]\n  }\n}\n\nArchive v3.3.4: 7 files, 9492 bytes\n\nFiles: assets/amazon_handler.js (14680b), assets/main_handler.js (1379b), Dockerfile.sh (247b), package.json (863b), scripts/setup.sh (698b), SKILL.md (6559b), _meta.json (133b)\n\nFile v3.3.4:SKILL.md\n\n---\nname: amazon-scraper\ndescription: >\n  High-performance containerized Amazon scraper (Docker + playwright-extra + Stealth plugin).\n  Bypasses Amazon headless detection. Supports Amazon BSR, search results, and product detail pages. Also includes a generic mode for other dynamic web pages.\n  Use when user mentions any of these:\n  爬虫, 爬取, 抓取, 采集, 数据采集, 爬数据, 抓数据, 获取数据,\n  scrape, crawl, extract, fetch data, pull data,\n  亚马逊, Amazon, ASIN, BSR, Best Sellers, 畅销榜, 热销榜, 新品榜, 飙升榜, 排行榜,\n  选品, 竞品分析, 竞品调研, 市场调研, 品类分析, 类目分析, 产品调研,\n  月销量, bought in past month, 销量, 评论数, 价格对比,\n  网页内容, 网站数据, 页面抓取, 动态页面,\n  关键词搜索, 搜索结果, search results,\n  产品详情, 产品信息, listing数据, listing分析,\n  top 100, top sellers, 热门产品, 爆款, 跑量款,\n  价格带, 评分分布, review分析, 评论分析\n---\n\n# Amazon Scraper\n\nDocker容器化爬虫，基于 playwright-extra + Stealth 插件，专为绕过亚马逊反爬检测优化，同时支持通用动态网页爬取。\n\n## ⚙️ 系统要求\n\n- **Docker Engine 20.10+**（必须已安装并运行）\n- **磁盘空间**：~2GB（镜像 + Playwright 浏览器二进制文件）\n- **内存**：建议 2GB+（Playwright 运行时需要）\n\n## 快速开始\n\n首次使用：在 skill 目录下执行一键构建脚本：\n\n```bash\nbash scripts/setup.sh\n```\n\n脚本会自动完成：构建 `clawd-crawlee` 镜像 + 创建 `~/scrapes` 输出目录。\n\n## 模式选择规则\n\n### 1. Amazon模式 (`amazon_handler.js`)\n**自动触发条件:** URL包含 `amazon.com`，或用户提到亚马逊/Amazon/ASIN/BSR/选品/竞品/畅销榜/类目分析等关键词\n\n根据URL自动识别页面类型：\n\n| URL特征 | 页面类型 | 可获取字段 |\n|---|---|---|\n| `/zgbs/` 或 `/bestsellers/` | 畅销榜 | rank, title, asin, price, rating, reviews, image, url |\n| `/zg/new-releases/` | 新品榜 | 同上 |\n| `/zg/movers-and-shakers/` | 飙升榜 | 同上 |\n| `/s?k=` 或 `/s/` | 搜索结果 | title, asin, price, rating, reviews, image, url, **boughtPastMonth**, sponsored |\n| `/dp/` 或 `/gp/product/` | 产品详情 | title, asin, price, rating, reviews, brand, bsr, **boughtPastMonth**, dateFirstAvailable, category, bullets, details, image |\n\n**⚠️ 重要规则:**\n- **Best Sellers页面没有月销量(boughtPastMonth)数据** — 亚马逊不在榜单页显示此信息\n- **要获取月销量，必须用搜索页(`/s?k=关键词`)或产品详情页(`/dp/ASIN`)**\n- 如果用户同时需要排名+月销量，建议：先爬Best Sellers拿排名，再用搜索页补月销\n\n```bash\n# 畅销榜（有排名，无月销）\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"https://www.amazon.com/zgbs/electronics\"\n\n# 搜索结果（有月销，无排名）\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"https://www.amazon.com/s?k=feather+duster\"\n\n# 产品详情（最全字段：BSR、品牌、卖点、月销）\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"https://www.amazon.com/dp/B001TQ6IHS\"\n\n# 多页爬取\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"URL\" --pages 2\n```\n\n**输出格式:** JSON\n```json\n{\n  \"status\": \"SUCCESS\",\n  \"type\": \"bestsellers|search|product-detail\",\n  \"category\": \"品类名\",\n  \"totalProducts\": 30,\n  \"scrapedAt\": \"ISO时间\",\n  \"products\": [\n    {\n      \"rank\": 1,\n      \"title\": \"产品名\",\n      \"asin\": \"B001TQ6IHS\",\n      \"price\": 9.94,\n      \"priceStr\": \"$9.94\",\n      \"rating\": 4.6,\n      \"reviews\": 20547,\n      \"boughtPastMonth\": \"1K+\",\n      \"image\": \"https://...\",\n      \"url\": \"https://...\"\n    }\n  ]\n}\n```\n\n### 2. 通用模式 (`main_handler.js`)\n**触发条件:** 非Amazon的URL，或用户提到爬取/抓取任意网页内容\n\n- Playwright打开页面，等待JS加载完成\n- 提取 `document.body.innerText`（纯文本，去广告噪音）\n- 输出上限10000字符\n- 输出: `{status:\"SUCCESS\", type:\"GENERIC\", title, data}`\n\n```bash\ndocker run -t --rm clawd-crawlee node assets/main_handler.js \"https://任意网址\"\n```\n\n## Agent调用决策树\n\n```\n用户给了URL?\n├─ 包含 amazon.com → 用 amazon_handler.js\n│   ├─ 需要月销量? → 建议用搜索URL(/s?k=) 或详情页(/dp/)\n│   └─ 需要排名? → 用畅销榜URL(/zgbs/)\n└─ 其他网站 → 用 main_handler.js (通用模式)\n\n用户没给URL，只说了需求?\n├─ \"爬亚马逊XX品类Top\" / \"XX类目排行\" / \"XX畅销榜\" → 构造 https://www.amazon.com/zgbs/品类\n├─ \"搜亚马逊XX\" / \"XX关键词搜索\" / \"找XX产品\" → 构造 https://www.amazon.com/s?k=关键词\n├─ \"分析某个ASIN\" / \"看看这个产品\" / \"XX的详情\" → 构造 https://www.amazon.com/dp/ASIN\n├─ \"XX的月销量\" / \"XX卖了多少\" / \"XX销量怎么样\" → 用搜索页或详情页（有boughtPastMonth）\n├─ \"竞品分析\" / \"竞品调研\" / \"对手在卖什么\" → 先搜索再逐个爬详情\n├─ \"选品\" / \"什么好卖\" / \"品类机会\" / \"市场调研\" → Best Sellers + 搜索结合\n└─ 其他网页 → 先web_search找到URL，再用通用模式爬\n```\n\n## 常见用户意图 → 操作映射\n\n| 用户说 | 操作 |\n|---|---|\n| \"帮我看看亚马逊XX品类\" | 爬 /zgbs/品类 畅销榜 |\n| \"XX在亚马逊卖得怎么样\" | 搜索 /s?k=XX 看月销 |\n| \"分析一下这个ASIN: BXXXXXXXXX\" | 爬 /dp/ASIN 详情页 |\n| \"XX品类有什么机会\" | 畅销榜 + 搜索 综合分析 |\n| \"帮我爬这个链接\" | 判断URL类型，选对应handler |\n| \"帮我抓XX网站的内容\" | 通用模式 |\n| \"搜一下XX的竞品\" | 搜索页爬取 + 分析 |\n| \"XX月销多少\" / \"XX一个月卖多少\" | 搜索页或详情页 |\n| \"帮我看看top 100\" / \"热门产品\" | Best Sellers畅销榜 |\n| \"新品有哪些\" / \"最近上了什么新品\" | /zg/new-releases/ |\n| \"什么产品涨得快\" / \"飙升榜\" | /zg/movers-and-shakers/ |\n\n## 反爬能力\n- **playwright-extra-plugin-stealth** — 自动修改 navigator、WebGL、Canvas 等 headless 特征\n- **Chrome 120 UserAgent** — 模拟真实 Windows Chrome 浏览器\n- **1920x1080 viewport** — 避免移动端/小屏检测\n- 自动滚动加载懒加载内容\n- Docker沙箱隔离，每次启动全新浏览器上下文\n\n## 局限\n- 通用模式输出上限10000字符\n- Amazon单页最多约30-50个产品\n- 不支持需要登录的页面\n- Docker容器启动有~15秒冷启动时间（含 stealth 插件初始化）\n\nFile v3.3.4:_meta.json\n\n{\n  \"ownerId\": \"kn7ewmmms6dthpk632rzrc05v981ah4f\",\n  \"slug\": \"amazon-scraper\",\n  \"version\": \"3.3.4\",\n  \"publishedAt\": 1776834476425\n}\n\nFile v3.3.4:package.json\n\n{\n  \"name\": \"amazon-scraper\",\n  \"version\": \"3.3.4\",\n  \"description\": \"High-performance containerized Amazon scraper with stealth mode. Bypasses headless detection. Supports BSR, search, product detail, and generic pages. Built on Docker + playwright-extra + Stealth.\",\n  \"main\": \"assets/amazon_handler.js\",\n  \"scripts\": {\n    \"test\": \"node assets/main_handler.js https://www.amazon.com/zgbs/electronics\",\n    \"setup\": \"bash scripts/setup.sh\"\n  },\n  \"keywords\": [\n    \"amazon\",\n    \"scraper\",\n    \"asin\",\n    \"bsr\",\n    \"crawlee\",\n    \"playwright\",\n    \"openclaw\",\n    \"docker\"\n  ],\n  \"author\": \"Joseph\",\n  \"license\": \"MIT\",\n  \"dependencies\": {\n    \"playwright\": \"^1.40.0\",\n    \"playwright-extra\": \"^4.3.6\",\n    \"puppeteer-extra-plugin-stealth\": \"^2.11.2\"\n  },\n  \"openclaw\": {\n    \"requires\": {\n      \"docker\": true\n    },\n    \"requiredBinaries\": [\"docker\"]\n  }\n}\n\nArchive v3.3.3: 7 files, 9479 bytes\n\nFiles: assets/amazon_handler.js (14681b), assets/main_handler.js (1379b), Dockerfile.sh (247b), package.json (864b), scripts/setup.sh (698b), SKILL.md (6559b), _meta.json (133b)\n\nFile v3.3.3:SKILL.md\n\n---\nname: amazon-scraper\ndescription: >\n  High-performance containerized Amazon scraper (Docker + playwright-extra + Stealth plugin).\n  Bypasses Amazon headless detection. Supports Amazon BSR, search results, and product detail pages. Also includes a generic mode for other dynamic web pages.\n  Use when user mentions any of these:\n  爬虫, 爬取, 抓取, 采集, 数据采集, 爬数据, 抓数据, 获取数据,\n  scrape, crawl, extract, fetch data, pull data,\n  亚马逊, Amazon, ASIN, BSR, Best Sellers, 畅销榜, 热销榜, 新品榜, 飙升榜, 排行榜,\n  选品, 竞品分析, 竞品调研, 市场调研, 品类分析, 类目分析, 产品调研,\n  月销量, bought in past month, 销量, 评论数, 价格对比,\n  网页内容, 网站数据, 页面抓取, 动态页面,\n  关键词搜索, 搜索结果, search results,\n  产品详情, 产品信息, listing数据, listing分析,\n  top 100, top sellers, 热门产品, 爆款, 跑量款,\n  价格带, 评分分布, review分析, 评论分析\n---\n\n# Amazon Scraper\n\nDocker容器化爬虫，基于 playwright-extra + Stealth 插件，专为绕过亚马逊反爬检测优化，同时支持通用动态网页爬取。\n\n## ⚙️ 系统要求\n\n- **Docker Engine 20.10+**（必须已安装并运行）\n- **磁盘空间**：~2GB（镜像 + Playwright 浏览器二进制文件）\n- **内存**：建议 2GB+（Playwright 运行时需要）\n\n## 快速开始\n\n首次使用：在 skill 目录下执行一键构建脚本：\n\n```bash\nbash scripts/setup.sh\n```\n\n脚本会自动完成：构建 `clawd-crawlee` 镜像 + 创建 `~/scrapes` 输出目录。\n\n## 模式选择规则\n\n### 1. Amazon模式 (`amazon_handler.js`)\n**自动触发条件:** URL包含 `amazon.com`，或用户提到亚马逊/Amazon/ASIN/BSR/选品/竞品/畅销榜/类目分析等关键词\n\n根据URL自动识别页面类型：\n\n| URL特征 | 页面类型 | 可获取字段 |\n|---|---|---|\n| `/zgbs/` 或 `/bestsellers/` | 畅销榜 | rank, title, asin, price, rating, reviews, image, url |\n| `/zg/new-releases/` | 新品榜 | 同上 |\n| `/zg/movers-and-shakers/` | 飙升榜 | 同上 |\n| `/s?k=` 或 `/s/` | 搜索结果 | title, asin, price, rating, reviews, image, url, **boughtPastMonth**, sponsored |\n| `/dp/` 或 `/gp/product/` | 产品详情 | title, asin, price, rating, reviews, brand, bsr, **boughtPastMonth**, dateFirstAvailable, category, bullets, details, image |\n\n**⚠️ 重要规则:**\n- **Best Sellers页面没有月销量(boughtPastMonth)数据** — 亚马逊不在榜单页显示此信息\n- **要获取月销量，必须用搜索页(`/s?k=关键词`)或产品详情页(`/dp/ASIN`)**\n- 如果用户同时需要排名+月销量，建议：先爬Best Sellers拿排名，再用搜索页补月销\n\n```bash\n# 畅销榜（有排名，无月销）\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"https://www.amazon.com/zgbs/electronics\"\n\n# 搜索结果（有月销，无排名）\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"https://www.amazon.com/s?k=feather+duster\"\n\n# 产品详情（最全字段：BSR、品牌、卖点、月销）\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"https://www.amazon.com/dp/B001TQ6IHS\"\n\n# 多页爬取\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"URL\" --pages 2\n```\n\n**输出格式:** JSON\n```json\n{\n  \"status\": \"SUCCESS\",\n  \"type\": \"bestsellers|search|product-detail\",\n  \"category\": \"品类名\",\n  \"totalProducts\": 30,\n  \"scrapedAt\": \"ISO时间\",\n  \"products\": [\n    {\n      \"rank\": 1,\n      \"title\": \"产品名\",\n      \"asin\": \"B001TQ6IHS\",\n      \"price\": 9.94,\n      \"priceStr\": \"$9.94\",\n      \"rating\": 4.6,\n      \"reviews\": 20547,\n      \"boughtPastMonth\": \"1K+\",\n      \"image\": \"https://...\",\n      \"url\": \"https://...\"\n    }\n  ]\n}\n```\n\n### 2. 通用模式 (`main_handler.js`)\n**触发条件:** 非Amazon的URL，或用户提到爬取/抓取任意网页内容\n\n- Playwright打开页面，等待JS加载完成\n- 提取 `document.body.innerText`（纯文本，去广告噪音）\n- 输出上限10000字符\n- 输出: `{status:\"SUCCESS\", type:\"GENERIC\", title, data}`\n\n```bash\ndocker run -t --rm clawd-crawlee node assets/main_handler.js \"https://任意网址\"\n```\n\n## Agent调用决策树\n\n```\n用户给了URL?\n├─ 包含 amazon.com → 用 amazon_handler.js\n│   ├─ 需要月销量? → 建议用搜索URL(/s?k=) 或详情页(/dp/)\n│   └─ 需要排名? → 用畅销榜URL(/zgbs/)\n└─ 其他网站 → 用 main_handler.js (通用模式)\n\n用户没给URL，只说了需求?\n├─ \"爬亚马逊XX品类Top\" / \"XX类目排行\" / \"XX畅销榜\" → 构造 https://www.amazon.com/zgbs/品类\n├─ \"搜亚马逊XX\" / \"XX关键词搜索\" / \"找XX产品\" → 构造 https://www.amazon.com/s?k=关键词\n├─ \"分析某个ASIN\" / \"看看这个产品\" / \"XX的详情\" → 构造 https://www.amazon.com/dp/ASIN\n├─ \"XX的月销量\" / \"XX卖了多少\" / \"XX销量怎么样\" → 用搜索页或详情页（有boughtPastMonth）\n├─ \"竞品分析\" / \"竞品调研\" / \"对手在卖什么\" → 先搜索再逐个爬详情\n├─ \"选品\" / \"什么好卖\" / \"品类机会\" / \"市场调研\" → Best Sellers + 搜索结合\n└─ 其他网页 → 先web_search找到URL，再用通用模式爬\n```\n\n## 常见用户意图 → 操作映射\n\n| 用户说 | 操作 |\n|---|---|\n| \"帮我看看亚马逊XX品类\" | 爬 /zgbs/品类 畅销榜 |\n| \"XX在亚马逊卖得怎么样\" | 搜索 /s?k=XX 看月销 |\n| \"分析一下这个ASIN: BXXXXXXXXX\" | 爬 /dp/ASIN 详情页 |\n| \"XX品类有什么机会\" | 畅销榜 + 搜索 综合分析 |\n| \"帮我爬这个链接\" | 判断URL类型，选对应handler |\n| \"帮我抓XX网站的内容\" | 通用模式 |\n| \"搜一下XX的竞品\" | 搜索页爬取 + 分析 |\n| \"XX月销多少\" / \"XX一个月卖多少\" | 搜索页或详情页 |\n| \"帮我看看top 100\" / \"热门产品\" | Best Sellers畅销榜 |\n| \"新品有哪些\" / \"最近上了什么新品\" | /zg/new-releases/ |\n| \"什么产品涨得快\" / \"飙升榜\" | /zg/movers-and-shakers/ |\n\n## 反爬能力\n- **playwright-extra-plugin-stealth** — 自动修改 navigator、WebGL、Canvas 等 headless 特征\n- **Chrome 120 UserAgent** — 模拟真实 Windows Chrome 浏览器\n- **1920x1080 viewport** — 避免移动端/小屏检测\n- 自动滚动加载懒加载内容\n- Docker沙箱隔离，每次启动全新浏览器上下文\n\n## 局限\n- 通用模式输出上限10000字符\n- Amazon单页最多约30-50个产品\n- 不支持需要登录的页面\n- Docker容器启动有~15秒冷启动时间（含 stealth 插件初始化）\n\nFile v3.3.3:_meta.json\n\n{\n  \"ownerId\": \"kn7ewmmms6dthpk632rzrc05v981ah4f\",\n  \"slug\": \"amazon-scraper\",\n  \"version\": \"3.3.3\",\n  \"publishedAt\": 1776834262551\n}\n\nFile v3.3.3:package.json\n\n{\n  \"name\": \"amazon-scraper\",\n  \"version\": \"3.3.3\",\n  \"description\": \"High-performance containerized Amazon scraper with stealth mode. Bypasses headless detection. Supports BSR, search, product detail, and generic pages. Built on Docker + playwright-extra + Stealth.\",\n  \"main\": \"assets/amazon_handler.js\",\n  \"scripts\": {\n    \"test\": \"node assets/main_handler.js https://www.amazon.com/zgbs/electronics\",\n    \"setup\": \"bash scripts/setup.sh\"\n  },\n  \"keywords\": [\n    \"amazon\",\n    \"scraper\",\n    \"asin\",\n    \"bsr\",\n    \"crawlee\",\n    \"playwright\",\n    \"openclaw\",\n    \"docker\"\n  ],\n  \"author\": \"Joseph\",\n  \"license\": \"MIT\",\n  \"dependencies\": {\n    \"playwright\": \"^1.40.0\",\n    \"playwright-extra\": \"^4.3.6\",\n    \"playwright-extra-plugin-stealth\": \"^2.11.2\"\n  },\n  \"openclaw\": {\n    \"requires\": {\n      \"docker\": true\n    },\n    \"requiredBinaries\": [\"docker\"]\n  }\n}\n\nArchive v3.3.2: 7 files, 9480 bytes\n\nFiles: assets/amazon_handler.js (14681b), assets/main_handler.js (1379b), Dockerfile.sh (247b), package.json (864b), scripts/setup.sh (698b), SKILL.md (6559b), _meta.json (133b)\n\nFile v3.3.2:SKILL.md\n\n---\nname: amazon-scraper\ndescription: >\n  High-performance containerized Amazon scraper (Docker + playwright-extra + Stealth plugin).\n  Bypasses Amazon headless detection. Supports Amazon BSR, search results, and product detail pages. Also includes a generic mode for other dynamic web pages.\n  Use when user mentions any of these:\n  爬虫, 爬取, 抓取, 采集, 数据采集, 爬数据, 抓数据, 获取数据,\n  scrape, crawl, extract, fetch data, pull data,\n  亚马逊, Amazon, ASIN, BSR, Best Sellers, 畅销榜, 热销榜, 新品榜, 飙升榜, 排行榜,\n  选品, 竞品分析, 竞品调研, 市场调研, 品类分析, 类目分析, 产品调研,\n  月销量, bought in past month, 销量, 评论数, 价格对比,\n  网页内容, 网站数据, 页面抓取, 动态页面,\n  关键词搜索, 搜索结果, search results,\n  产品详情, 产品信息, listing数据, listing分析,\n  top 100, top sellers, 热门产品, 爆款, 跑量款,\n  价格带, 评分分布, review分析, 评论分析\n---\n\n# Amazon Scraper\n\nDocker容器化爬虫，基于 playwright-extra + Stealth 插件，专为绕过亚马逊反爬检测优化，同时支持通用动态网页爬取。\n\n## ⚙️ 系统要求\n\n- **Docker Engine 20.10+**（必须已安装并运行）\n- **磁盘空间**：~2GB（镜像 + Playwright 浏览器二进制文件）\n- **内存**：建议 2GB+（Playwright 运行时需要）\n\n## 快速开始\n\n首次使用：在 skill 目录下执行一键构建脚本：\n\n```bash\nbash scripts/setup.sh\n```\n\n脚本会自动完成：构建 `clawd-crawlee` 镜像 + 创建 `~/scrapes` 输出目录。\n\n## 模式选择规则\n\n### 1. Amazon模式 (`amazon_handler.js`)\n**自动触发条件:** URL包含 `amazon.com`，或用户提到亚马逊/Amazon/ASIN/BSR/选品/竞品/畅销榜/类目分析等关键词\n\n根据URL自动识别页面类型：\n\n| URL特征 | 页面类型 | 可获取字段 |\n|---|---|---|\n| `/zgbs/` 或 `/bestsellers/` | 畅销榜 | rank, title, asin, price, rating, reviews, image, url |\n| `/zg/new-releases/` | 新品榜 | 同上 |\n| `/zg/movers-and-shakers/` | 飙升榜 | 同上 |\n| `/s?k=` 或 `/s/` | 搜索结果 | title, asin, price, rating, reviews, image, url, **boughtPastMonth**, sponsored |\n| `/dp/` 或 `/gp/product/` | 产品详情 | title, asin, price, rating, reviews, brand, bsr, **boughtPastMonth**, dateFirstAvailable, category, bullets, details, image |\n\n**⚠️ 重要规则:**\n- **Best Sellers页面没有月销量(boughtPastMonth)数据** — 亚马逊不在榜单页显示此信息\n- **要获取月销量，必须用搜索页(`/s?k=关键词`)或产品详情页(`/dp/ASIN`)**\n- 如果用户同时需要排名+月销量，建议：先爬Best Sellers拿排名，再用搜索页补月销\n\n```bash\n# 畅销榜（有排名，无月销）\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"https://www.amazon.com/zgbs/electronics\"\n\n# 搜索结果（有月销，无排名）\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"https://www.amazon.com/s?k=feather+duster\"\n\n# 产品详情（最全字段：BSR、品牌、卖点、月销）\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"https://www.amazon.com/dp/B001TQ6IHS\"\n\n# 多页爬取\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"URL\" --pages 2\n```\n\n**输出格式:** JSON\n```json\n{\n  \"status\": \"SUCCESS\",\n  \"type\": \"bestsellers|search|product-detail\",\n  \"category\": \"品类名\",\n  \"totalProducts\": 30,\n  \"scrapedAt\": \"ISO时间\",\n  \"products\": [\n    {\n      \"rank\": 1,\n      \"title\": \"产品名\",\n      \"asin\": \"B001TQ6IHS\",\n      \"price\": 9.94,\n      \"priceStr\": \"$9.94\",\n      \"rating\": 4.6,\n      \"reviews\": 20547,\n      \"boughtPastMonth\": \"1K+\",\n      \"image\": \"https://...\",\n      \"url\": \"https://...\"\n    }\n  ]\n}\n```\n\n### 2. 通用模式 (`main_handler.js`)\n**触发条件:** 非Amazon的URL，或用户提到爬取/抓取任意网页内容\n\n- Playwright打开页面，等待JS加载完成\n- 提取 `document.body.innerText`（纯文本，去广告噪音）\n- 输出上限10000字符\n- 输出: `{status:\"SUCCESS\", type:\"GENERIC\", title, data}`\n\n```bash\ndocker run -t --rm clawd-crawlee node assets/main_handler.js \"https://任意网址\"\n```\n\n## Agent调用决策树\n\n```\n用户给了URL?\n├─ 包含 amazon.com → 用 amazon_handler.js\n│   ├─ 需要月销量? → 建议用搜索URL(/s?k=) 或详情页(/dp/)\n│   └─ 需要排名? → 用畅销榜URL(/zgbs/)\n└─ 其他网站 → 用 main_handler.js (通用模式)\n\n用户没给URL，只说了需求?\n├─ \"爬亚马逊XX品类Top\" / \"XX类目排行\" / \"XX畅销榜\" → 构造 https://www.amazon.com/zgbs/品类\n├─ \"搜亚马逊XX\" / \"XX关键词搜索\" / \"找XX产品\" → 构造 https://www.amazon.com/s?k=关键词\n├─ \"分析某个ASIN\" / \"看看这个产品\" / \"XX的详情\" → 构造 https://www.amazon.com/dp/ASIN\n├─ \"XX的月销量\" / \"XX卖了多少\" / \"XX销量怎么样\" → 用搜索页或详情页（有boughtPastMonth）\n├─ \"竞品分析\" / \"竞品调研\" / \"对手在卖什么\" → 先搜索再逐个爬详情\n├─ \"选品\" / \"什么好卖\" / \"品类机会\" / \"市场调研\" → Best Sellers + 搜索结合\n└─ 其他网页 → 先web_search找到URL，再用通用模式爬\n```\n\n## 常见用户意图 → 操作映射\n\n| 用户说 | 操作 |\n|---|---|\n| \"帮我看看亚马逊XX品类\" | 爬 /zgbs/品类 畅销榜 |\n| \"XX在亚马逊卖得怎么样\" | 搜索 /s?k=XX 看月销 |\n| \"分析一下这个ASIN: BXXXXXXXXX\" | 爬 /dp/ASIN 详情页 |\n| \"XX品类有什么机会\" | 畅销榜 + 搜索 综合分析 |\n| \"帮我爬这个链接\" | 判断URL类型，选对应handler |\n| \"帮我抓XX网站的内容\" | 通用模式 |\n| \"搜一下XX的竞品\" | 搜索页爬取 + 分析 |\n| \"XX月销多少\" / \"XX一个月卖多少\" | 搜索页或详情页 |\n| \"帮我看看top 100\" / \"热门产品\" | Best Sellers畅销榜 |\n| \"新品有哪些\" / \"最近上了什么新品\" | /zg/new-releases/ |\n| \"什么产品涨得快\" / \"飙升榜\" | /zg/movers-and-shakers/ |\n\n## 反爬能力\n- **playwright-extra-plugin-stealth** — 自动修改 navigator、WebGL、Canvas 等 headless 特征\n- **Chrome 120 UserAgent** — 模拟真实 Windows Chrome 浏览器\n- **1920x1080 viewport** — 避免移动端/小屏检测\n- 自动滚动加载懒加载内容\n- Docker沙箱隔离，每次启动全新浏览器上下文\n\n## 局限\n- 通用模式输出上限10000字符\n- Amazon单页最多约30-50个产品\n- 不支持需要登录的页面\n- Docker容器启动有~15秒冷启动时间（含 stealth 插件初始化）\n\nFile v3.3.2:_meta.json\n\n{\n  \"ownerId\": \"kn7ewmmms6dthpk632rzrc05v981ah4f\",\n  \"slug\": \"amazon-scraper\",\n  \"version\": \"3.3.2\",\n  \"publishedAt\": 1776834212432\n}\n\nFile v3.3.2:package.json\n\n{\n  \"name\": \"amazon-scraper\",\n  \"version\": \"3.3.2\",\n  \"description\": \"High-performance containerized Amazon scraper with stealth mode. Bypasses headless detection. Supports BSR, search, product detail, and generic pages. Built on Docker + playwright-extra + Stealth.\",\n  \"main\": \"assets/amazon_handler.js\",\n  \"scripts\": {\n    \"test\": \"node assets/main_handler.js https://www.amazon.com/zgbs/electronics\",\n    \"setup\": \"bash scripts/setup.sh\"\n  },\n  \"keywords\": [\n    \"amazon\",\n    \"scraper\",\n    \"asin\",\n    \"bsr\",\n    \"crawlee\",\n    \"playwright\",\n    \"openclaw\",\n    \"docker\"\n  ],\n  \"author\": \"Joseph\",\n  \"license\": \"MIT\",\n  \"dependencies\": {\n    \"playwright-extra\": \"^4.3.6\",\n    \"playwright-extra-plugin-stealth\": \"^2.11.2\",\n    \"playwright\": \"^1.40.0\"\n  },\n  \"openclaw\": {\n    \"requires\": {\n      \"docker\": true\n    },\n    \"requiredBinaries\": [\"docker\"]\n  }\n}\n\nArchive v3.3.1: 7 files, 9384 bytes\n\nFiles: assets/amazon_handler.js (14681b), assets/main_handler.js (1379b), Dockerfile.sh (247b), package.json (864b), scripts/setup.sh (698b), SKILL.md (6392b), _meta.json (133b)\n\nFile v3.3.1:SKILL.md\n\n---\nname: amazon-scraper\ndescription: >\n  High-performance containerized Amazon scraper (Docker + playwright-extra + Stealth plugin).\n  Bypasses Amazon headless detection. Supports Amazon BSR, search results, and product detail pages. Also includes a generic mode for other dynamic web pages.\n  Use when user mentions any of these:\n  爬虫, 爬取, 抓取, 采集, 数据采集, 爬数据, 抓数据, 获取数据,\n  scrape, crawl, extract, fetch data, pull data,\n  亚马逊, Amazon, ASIN, BSR, Best Sellers, 畅销榜, 热销榜, 新品榜, 飙升榜, 排行榜,\n  选品, 竞品分析, 竞品调研, 市场调研, 品类分析, 类目分析, 产品调研,\n  月销量, bought in past month, 销量, 评论数, 价格对比,\n  网页内容, 网站数据, 页面抓取, 动态页面,\n  关键词搜索, 搜索结果, search results,\n  产品详情, 产品信息, listing数据, listing分析,\n  top 100, top sellers, 热门产品, 爆款, 跑量款,\n  价格带, 评分分布, review分析, 评论分析\n---\n\n# Amazon Scraper\n\nDocker容器化爬虫，基于 playwright-extra + Stealth 插件，专为绕过亚马逊反爬检测优化，同时支持通用动态网页爬取。\n\n## ⚙️ 系统要求\n\n- **Docker Engine 20.10+**（必须已安装并运行）\n- **磁盘空间**：~2GB（镜像 + Playwright 浏览器二进制文件）\n- **内存**：建议 2GB+（Playwright 运行时需要）\n\n## 快速开始\n\n首次使用：在 skill 目录下执行一键构建脚本：\n\n```bash\nbash scripts/setup.sh\n```\n\n脚本会自动完成：构建 `clawd-crawlee` 镜像 + 创建 `~/scrapes` 输出目录。\n\n## 模式选择规则\n\n### 1. Amazon模式 (`amazon_handler.js`)\n**自动触发条件:** URL包含 `amazon.com`，或用户提到亚马逊/Amazon/ASIN/BSR/选品/竞品/畅销榜/类目分析等关键词\n\n根据URL自动识别页面类型：\n\n| URL特征 | 页面类型 | 可获取字段 |\n|---|---|---|\n| `/zgbs/` 或 `/bestsellers/` | 畅销榜 | rank, title, asin, price, rating, reviews, image, url |\n| `/zg/new-releases/` | 新品榜 | 同上 |\n| `/zg/movers-and-shakers/` | 飙升榜 | 同上 |\n| `/s?k=` 或 `/s/` | 搜索结果 | title, asin, price, rating, reviews, image, url, **boughtPastMonth**, sponsored |\n| `/dp/` 或 `/gp/product/` | 产品详情 | title, asin, price, rating, reviews, brand, bsr, **boughtPastMonth**, dateFirstAvailable, category, bullets, details, image |\n\n**⚠️ 重要规则:**\n- **Best Sellers页面没有月销量(boughtPastMonth)数据** — 亚马逊不在榜单页显示此信息\n- **要获取月销量，必须用搜索页(`/s?k=关键词`)或产品详情页(`/dp/ASIN`)**\n- 如果用户同时需要排名+月销量，建议：先爬Best Sellers拿排名，再用搜索页补月销\n\n```bash\n# 畅销榜（有排名，无月销）\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"https://www.amazon.com/zgbs/electronics\"\n\n# 搜索结果（有月销，无排名）\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"https://www.amazon.com/s?k=feather+duster\"\n\n# 产品详情（最全字段：BSR、品牌、卖点、月销）\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"https://www.amazon.com/dp/B001TQ6IHS\"\n\n# 多页爬取\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"URL\" --pages 2\n```\n\n**输出格式:** JSON\n```json\n{\n  \"status\": \"SUCCESS\",\n  \"type\": \"bestsellers|search|product-detail\",\n  \"category\": \"品类名\",\n  \"totalProducts\": 30,\n  \"scrapedAt\": \"ISO时间\",\n  \"products\": [\n    {\n      \"rank\": 1,\n      \"title\": \"产品名\",\n      \"asin\": \"B001TQ6IHS\",\n      \"price\": 9.94,\n      \"priceStr\": \"$9.94\",\n      \"rating\": 4.6,\n      \"reviews\": 20547,\n      \"boughtPastMonth\": \"1K+\",\n      \"image\": \"https://...\",\n      \"url\": \"https://...\"\n    }\n  ]\n}\n```\n\n### 2. 通用模式 (`main_handler.js`)\n**触发条件:** 非Amazon的URL，或用户提到爬取/抓取任意网页内容\n\n- Playwright打开页面，等待JS加载完成\n- 提取 `document.body.innerText`（纯文本，去广告噪音）\n- 输出上限10000字符\n- 输出: `{status:\"SUCCESS\", type:\"GENERIC\", title, data}`\n\n```bash\ndocker run -t --rm clawd-crawlee node assets/main_handler.js \"https://任意网址\"\n```\n\n## Agent调用决策树\n\n```\n用户给了URL?\n├─ 包含 amazon.com → 用 amazon_handler.js\n│   ├─ 需要月销量? → 建议用搜索URL(/s?k=) 或详情页(/dp/)\n│   └─ 需要排名? → 用畅销榜URL(/zgbs/)\n└─ 其他网站 → 用 main_handler.js (通用模式)\n\n用户没给URL，只说了需求?\n├─ \"爬亚马逊XX品类Top\" / \"XX类目排行\" / \"XX畅销榜\" → 构造 https://www.amazon.com/zgbs/品类\n├─ \"搜亚马逊XX\" / \"XX关键词搜索\" / \"找XX产品\" → 构造 https://www.amazon.com/s?k=关键词\n├─ \"分析某个ASIN\" / \"看看这个产品\" / \"XX的详情\" → 构造 https://www.amazon.com/dp/ASIN\n├─ \"XX的月销量\" / \"XX卖了多少\" / \"XX销量怎么样\" → 用搜索页或详情页（有boughtPastMonth）\n├─ \"竞品分析\" / \"竞品调研\" / \"对手在卖什么\" → 先搜索再逐个爬详情\n├─ \"选品\" / \"什么好卖\" / \"品类机会\" / \"市场调研\" → Best Sellers + 搜索结合\n└─ 其他网页 → 先web_search找到URL，再用通用模式爬\n```\n\n## 常见用户意图 → 操作映射\n\n| 用户说 | 操作 |\n|---|---|\n| \"帮我看看亚马逊XX品类\" | 爬 /zgbs/品类 畅销榜 |\n| \"XX在亚马逊卖得怎么样\" | 搜索 /s?k=XX 看月销 |\n| \"分析一下这个ASIN: BXXXXXXXXX\" | 爬 /dp/ASIN 详情页 |\n| \"XX品类有什么机会\" | 畅销榜 + 搜索 综合分析 |\n| \"帮我爬这个链接\" | 判断URL类型，选对应handler |\n| \"帮我抓XX网站的内容\" | 通用模式 |\n| \"搜一下XX的竞品\" | 搜索页爬取 + 分析 |\n| \"XX月销多少\" / \"XX一个月卖多少\" | 搜索页或详情页 |\n| \"帮我看看top 100\" / \"热门产品\" | Best Sellers畅销榜 |\n| \"新品有哪些\" / \"最近上了什么新品\" | /zg/new-releases/ |\n| \"什么产品涨得快\" / \"飙升榜\" | /zg/movers-and-shakers/ |\n\n## 反爬能力\n- 每次清除Cookie，模拟全新用户\n- Docker沙箱隔离，无指纹追踪\n- Playwright模拟真实浏览器行为\n- 自动滚动加载懒加载内容\n- 支持重试（maxRetries: 2）\n\n## 局限\n- 通用模式输出上限10000字符\n- Amazon单页最多约30-50个产品\n- 不支持需要登录的页面\n- Docker容器启动有~10秒冷启动时间\n\nFile v3.3.1:_meta.json\n\n{\n  \"ownerId\": \"kn7ewmmms6dthpk632rzrc05v981ah4f\",\n  \"slug\": \"amazon-scraper\",\n  \"version\": \"3.3.1\",\n  \"publishedAt\": 1776834152331\n}\n\nFile v3.3.1:package.json\n\n{\n  \"name\": \"amazon-scraper\",\n  \"version\": \"3.3.1\",\n  \"description\": \"High-performance containerized Amazon scraper with stealth mode. Bypasses headless detection. Supports BSR, search, product detail, and generic pages. Built on Docker + playwright-extra + Stealth.\",\n  \"main\": \"assets/amazon_handler.js\",\n  \"scripts\": {\n    \"test\": \"node assets/main_handler.js https://www.amazon.com/zgbs/electronics\",\n    \"setup\": \"bash scripts/setup.sh\"\n  },\n  \"keywords\": [\n    \"amazon\",\n    \"scraper\",\n    \"asin\",\n    \"bsr\",\n    \"crawlee\",\n    \"playwright\",\n    \"openclaw\",\n    \"docker\"\n  ],\n  \"author\": \"Joseph\",\n  \"license\": \"MIT\",\n  \"dependencies\": {\n    \"playwright-extra\": \"^4.3.6\",\n    \"playwright-extra-plugin-stealth\": \"^2.11.2\",\n    \"playwright\": \"^1.40.0\"\n  },\n  \"openclaw\": {\n    \"requires\": {\n      \"docker\": true\n    },\n    \"requiredBinaries\": [\"docker\"]\n  }\n}\n\nArchive v3.3.0: 7 files, 9310 bytes\n\nFiles: assets/amazon_handler.js (14681b), assets/main_handler.js (1379b), Dockerfile.sh (247b), package.json (818b), scripts/setup.sh (698b), SKILL.md (6294b), _meta.json (133b)\n\nFile v3.3.0:SKILL.md\n\n---\nname: amazon-scraper\ndescription: >\n  High-performance containerized Amazon scraper (Docker + Crawlee + Playwright).\n  Supports Amazon BSR, search results, and product detail pages. Also includes a generic mode for other dynamic web pages.\n  Use when user mentions any of these:\n  爬虫, 爬取, 抓取, 采集, 数据采集, 爬数据, 抓数据, 获取数据,\n  scrape, crawl, extract, fetch data, pull data,\n  亚马逊, Amazon, ASIN, BSR, Best Sellers, 畅销榜, 热销榜, 新品榜, 飙升榜, 排行榜,\n  选品, 竞品分析, 竞品调研, 市场调研, 品类分析, 类目分析, 产品调研,\n  月销量, bought in past month, 销量, 评论数, 价格对比,\n  网页内容, 网站数据, 页面抓取, 动态页面,\n  关键词搜索, 搜索结果, search results,\n  产品详情, 产品信息, listing数据, listing分析,\n  top 100, top sellers, 热门产品, 爆款, 跑量款,\n  价格带, 评分分布, review分析, 评论分析\n---\n\n# Amazon Scraper\n\nDocker容器化爬虫，专为亚马逊数据采集优化，同时支持通用动态网页爬取。\n\n## ⚙️ 系统要求\n\n- **Docker Engine 20.10+**（必须已安装并运行）\n- **磁盘空间**：~2GB（镜像 + Playwright 浏览器二进制文件）\n- **内存**：建议 2GB+（Playwright 运行时需要）\n\n## 快速开始\n\n首次使用：在 skill 目录下执行一键构建脚本：\n\n```bash\nbash scripts/setup.sh\n```\n\n脚本会自动完成：构建 `clawd-crawlee` 镜像 + 创建 `~/scrapes` 输出目录。\n\n## 模式选择规则\n\n### 1. Amazon模式 (`amazon_handler.js`)\n**自动触发条件:** URL包含 `amazon.com`，或用户提到亚马逊/Amazon/ASIN/BSR/选品/竞品/畅销榜/类目分析等关键词\n\n根据URL自动识别页面类型：\n\n| URL特征 | 页面类型 | 可获取字段 |\n|---|---|---|\n| `/zgbs/` 或 `/bestsellers/` | 畅销榜 | rank, title, asin, price, rating, reviews, image, url |\n| `/zg/new-releases/` | 新品榜 | 同上 |\n| `/zg/movers-and-shakers/` | 飙升榜 | 同上 |\n| `/s?k=` 或 `/s/` | 搜索结果 | title, asin, price, rating, reviews, image, url, **boughtPastMonth**, sponsored |\n| `/dp/` 或 `/gp/product/` | 产品详情 | title, asin, price, rating, reviews, brand, bsr, **boughtPastMonth**, dateFirstAvailable, category, bullets, details, image |\n\n**⚠️ 重要规则:**\n- **Best Sellers页面没有月销量(boughtPastMonth)数据** — 亚马逊不在榜单页显示此信息\n- **要获取月销量，必须用搜索页(`/s?k=关键词`)或产品详情页(`/dp/ASIN`)**\n- 如果用户同时需要排名+月销量，建议：先爬Best Sellers拿排名，再用搜索页补月销\n\n```bash\n# 畅销榜（有排名，无月销）\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"https://www.amazon.com/zgbs/electronics\"\n\n# 搜索结果（有月销，无排名）\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"https://www.amazon.com/s?k=feather+duster\"\n\n# 产品详情（最全字段：BSR、品牌、卖点、月销）\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"https://www.amazon.com/dp/B001TQ6IHS\"\n\n# 多页爬取\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"URL\" --pages 2\n```\n\n**输出格式:** JSON\n```json\n{\n  \"status\": \"SUCCESS\",\n  \"type\": \"bestsellers|search|product-detail\",\n  \"category\": \"品类名\",\n  \"totalProducts\": 30,\n  \"scrapedAt\": \"ISO时间\",\n  \"products\": [\n    {\n      \"rank\": 1,\n      \"title\": \"产品名\",\n      \"asin\": \"B001TQ6IHS\",\n      \"price\": 9.94,\n      \"priceStr\": \"$9.94\",\n      \"rating\": 4.6,\n      \"reviews\": 20547,\n      \"boughtPastMonth\": \"1K+\",\n      \"image\": \"https://...\",\n      \"url\": \"https://...\"\n    }\n  ]\n}\n```\n\n### 2. 通用模式 (`main_handler.js`)\n**触发条件:** 非Amazon的URL，或用户提到爬取/抓取任意网页内容\n\n- Playwright打开页面，等待JS加载完成\n- 提取 `document.body.innerText`（纯文本，去广告噪音）\n- 输出上限10000字符\n- 输出: `{status:\"SUCCESS\", type:\"GENERIC\", title, data}`\n\n```bash\ndocker run -t --rm clawd-crawlee node assets/main_handler.js \"https://任意网址\"\n```\n\n## Agent调用决策树\n\n```\n用户给了URL?\n├─ 包含 amazon.com → 用 amazon_handler.js\n│   ├─ 需要月销量? → 建议用搜索URL(/s?k=) 或详情页(/dp/)\n│   └─ 需要排名? → 用畅销榜URL(/zgbs/)\n└─ 其他网站 → 用 main_handler.js (通用模式)\n\n用户没给URL，只说了需求?\n├─ \"爬亚马逊XX品类Top\" / \"XX类目排行\" / \"XX畅销榜\" → 构造 https://www.amazon.com/zgbs/品类\n├─ \"搜亚马逊XX\" / \"XX关键词搜索\" / \"找XX产品\" → 构造 https://www.amazon.com/s?k=关键词\n├─ \"分析某个ASIN\" / \"看看这个产品\" / \"XX的详情\" → 构造 https://www.amazon.com/dp/ASIN\n├─ \"XX的月销量\" / \"XX卖了多少\" / \"XX销量怎么样\" → 用搜索页或详情页（有boughtPastMonth）\n├─ \"竞品分析\" / \"竞品调研\" / \"对手在卖什么\" → 先搜索再逐个爬详情\n├─ \"选品\" / \"什么好卖\" / \"品类机会\" / \"市场调研\" → Best Sellers + 搜索结合\n└─ 其他网页 → 先web_search找到URL，再用通用模式爬\n```\n\n## 常见用户意图 → 操作映射\n\n| 用户说 | 操作 |\n|---|---|\n| \"帮我看看亚马逊XX品类\" | 爬 /zgbs/品类 畅销榜 |\n| \"XX在亚马逊卖得怎么样\" | 搜索 /s?k=XX 看月销 |\n| \"分析一下这个ASIN: BXXXXXXXXX\" | 爬 /dp/ASIN 详情页 |\n| \"XX品类有什么机会\" | 畅销榜 + 搜索 综合分析 |\n| \"帮我爬这个链接\" | 判断URL类型，选对应handler |\n| \"帮我抓XX网站的内容\" | 通用模式 |\n| \"搜一下XX的竞品\" | 搜索页爬取 + 分析 |\n| \"XX月销多少\" / \"XX一个月卖多少\" | 搜索页或详情页 |\n| \"帮我看看top 100\" / \"热门产品\" | Best Sellers畅销榜 |\n| \"新品有哪些\" / \"最近上了什么新品\" | /zg/new-releases/ |\n| \"什么产品涨得快\" / \"飙升榜\" | /zg/movers-and-shakers/ |\n\n## 反爬能力\n- 每次清除Cookie，模拟全新用户\n- Docker沙箱隔离，无指纹追踪\n- Playwright模拟真实浏览器行为\n- 自动滚动加载懒加载内容\n- 支持重试（maxRetries: 2）\n\n## 局限\n- 通用模式输出上限10000字符\n- Amazon单页最多约30-50个产品\n- 不支持需要登录的页面\n- Docker容器启动有~10秒冷启动时间\n\nFile v3.3.0:_meta.json\n\n{\n  \"ownerId\": \"kn7ewmmms6dthpk632rzrc05v981ah4f\",\n  \"slug\": \"amazon-scraper\",\n  \"version\": \"3.3.0\",\n  \"publishedAt\": 1776834093519\n}\n\nFile v3.3.0:package.json\n\n{\n  \"name\": \"amazon-scraper\",\n  \"version\": \"3.3.0\",\n  \"description\": \"High-performance containerized web scraper for Amazon (BSR, search, product detail) and generic dynamic pages. Built on Docker + Crawlee + Playwright.\",\n  \"main\": \"assets/amazon_handler.js\",\n  \"scripts\": {\n    \"test\": \"node assets/main_handler.js https://www.amazon.com/zgbs/electronics\",\n    \"setup\": \"bash scripts/setup.sh\"\n  },\n  \"keywords\": [\n    \"amazon\",\n    \"scraper\",\n    \"asin\",\n    \"bsr\",\n    \"crawlee\",\n    \"playwright\",\n    \"openclaw\",\n    \"docker\"\n  ],\n  \"author\": \"Joseph\",\n  \"license\": \"MIT\",\n  \"dependencies\": {\n    \"playwright-extra\": \"^4.3.6\",\n    \"playwright-extra-plugin-stealth\": \"^2.11.2\",\n    \"playwright\": \"^1.40.0\"\n  },\n  \"openclaw\": {\n    \"requires\": {\n      \"docker\": true\n    },\n    \"requiredBinaries\": [\"docker\"]\n  }\n}\n\nArchive v3.2.0: 7 files, 9225 bytes\n\nFiles: assets/amazon_handler.js (14735b), assets/main_handler.js (1379b), Dockerfile.sh (219b), package.json (759b), scripts/setup.sh (698b), SKILL.md (6294b), _meta.json (133b)\n\nFile v3.2.0:SKILL.md\n\n---\nname: amazon-scraper\ndescription: >\n  High-performance containerized Amazon scraper (Docker + Crawlee + Playwright).\n  Supports Amazon BSR, search results, and product detail pages. Also includes a generic mode for other dynamic web pages.\n  Use when user mentions any of these:\n  爬虫, 爬取, 抓取, 采集, 数据采集, 爬数据, 抓数据, 获取数据,\n  scrape, crawl, extract, fetch data, pull data,\n  亚马逊, Amazon, ASIN, BSR, Best Sellers, 畅销榜, 热销榜, 新品榜, 飙升榜, 排行榜,\n  选品, 竞品分析, 竞品调研, 市场调研, 品类分析, 类目分析, 产品调研,\n  月销量, bought in past month, 销量, 评论数, 价格对比,\n  网页内容, 网站数据, 页面抓取, 动态页面,\n  关键词搜索, 搜索结果, search results,\n  产品详情, 产品信息, listing数据, listing分析,\n  top 100, top sellers, 热门产品, 爆款, 跑量款,\n  价格带, 评分分布, review分析, 评论分析\n---\n\n# Amazon Scraper\n\nDocker容器化爬虫，专为亚马逊数据采集优化，同时支持通用动态网页爬取。\n\n## ⚙️ 系统要求\n\n- **Docker Engine 20.10+**（必须已安装并运行）\n- **磁盘空间**：~2GB（镜像 + Playwright 浏览器二进制文件）\n- **内存**：建议 2GB+（Playwright 运行时需要）\n\n## 快速开始\n\n首次使用：在 skill 目录下执行一键构建脚本：\n\n```bash\nbash scripts/setup.sh\n```\n\n脚本会自动完成：构建 `clawd-crawlee` 镜像 + 创建 `~/scrapes` 输出目录。\n\n## 模式选择规则\n\n### 1. Amazon模式 (`amazon_handler.js`)\n**自动触发条件:** URL包含 `amazon.com`，或用户提到亚马逊/Amazon/ASIN/BSR/选品/竞品/畅销榜/类目分析等关键词\n\n根据URL自动识别页面类型：\n\n| URL特征 | 页面类型 | 可获取字段 |\n|---|---|---|\n| `/zgbs/` 或 `/bestsellers/` | 畅销榜 | rank, title, asin, price, rating, reviews, image, url |\n| `/zg/new-releases/` | 新品榜 | 同上 |\n| `/zg/movers-and-shakers/` | 飙升榜 | 同上 |\n| `/s?k=` 或 `/s/` | 搜索结果 | title, asin, price, rating, reviews, image, url, **boughtPastMonth**, sponsored |\n| `/dp/` 或 `/gp/product/` | 产品详情 | title, asin, price, rating, reviews, brand, bsr, **boughtPastMonth**, dateFirstAvailable, category, bullets, details, image |\n\n**⚠️ 重要规则:**\n- **Best Sellers页面没有月销量(boughtPastMonth)数据** — 亚马逊不在榜单页显示此信息\n- **要获取月销量，必须用搜索页(`/s?k=关键词`)或产品详情页(`/dp/ASIN`)**\n- 如果用户同时需要排名+月销量，建议：先爬Best Sellers拿排名，再用搜索页补月销\n\n```bash\n# 畅销榜（有排名，无月销）\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"https://www.amazon.com/zgbs/electronics\"\n\n# 搜索结果（有月销，无排名）\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"https://www.amazon.com/s?k=feather+duster\"\n\n# 产品详情（最全字段：BSR、品牌、卖点、月销）\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"https://www.amazon.com/dp/B001TQ6IHS\"\n\n# 多页爬取\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"URL\" --pages 2\n```\n\n**输出格式:** JSON\n```json\n{\n  \"status\": \"SUCCESS\",\n  \"type\": \"bestsellers|search|product-detail\",\n  \"category\": \"品类名\",\n  \"totalProducts\": 30,\n  \"scrapedAt\": \"ISO时间\",\n  \"products\": [\n    {\n      \"rank\": 1,\n      \"title\": \"产品名\",\n      \"asin\": \"B001TQ6IHS\",\n      \"price\": 9.94,\n      \"priceStr\": \"$9.94\",\n      \"rating\": 4.6,\n      \"reviews\": 20547,\n      \"boughtPastMonth\": \"1K+\",\n      \"image\": \"https://...\",\n      \"url\": \"https://...\"\n    }\n  ]\n}\n```\n\n### 2. 通用模式 (`main_handler.js`)\n**触发条件:** 非Amazon的URL，或用户提到爬取/抓取任意网页内容\n\n- Playwright打开页面，等待JS加载完成\n- 提取 `document.body.innerText`（纯文本，去广告噪音）\n- 输出上限10000字符\n- 输出: `{status:\"SUCCESS\", type:\"GENERIC\", title, data}`\n\n```bash\ndocker run -t --rm clawd-crawlee node assets/main_handler.js \"https://任意网址\"\n```\n\n## Agent调用决策树\n\n```\n用户给了URL?\n├─ 包含 amazon.com → 用 amazon_handler.js\n│   ├─ 需要月销量? → 建议用搜索URL(/s?k=) 或详情页(/dp/)\n│   └─ 需要排名? → 用畅销榜URL(/zgbs/)\n└─ 其他网站 → 用 main_handler.js (通用模式)\n\n用户没给URL，只说了需求?\n├─ \"爬亚马逊XX品类Top\" / \"XX类目排行\" / \"XX畅销榜\" → 构造 https://www.amazon.com/zgbs/品类\n├─ \"搜亚马逊XX\" / \"XX关键词搜索\" / \"找XX产品\" → 构造 https://www.amazon.com/s?k=关键词\n├─ \"分析某个ASIN\" / \"看看这个产品\" / \"XX的详情\" → 构造 https://www.amazon.com/dp/ASIN\n├─ \"XX的月销量\" / \"XX卖了多少\" / \"XX销量怎么样\" → 用搜索页或详情页（有boughtPastMonth）\n├─ \"竞品分析\" / \"竞品调研\" / \"对手在卖什么\" → 先搜索再逐个爬详情\n├─ \"选品\" / \"什么好卖\" / \"品类机会\" / \"市场调研\" → Best Sellers + 搜索结合\n└─ 其他网页 → 先web_search找到URL，再用通用模式爬\n```\n\n## 常见用户意图 → 操作映射\n\n| 用户说 | 操作 |\n|---|---|\n| \"帮我看看亚马逊XX品类\" | 爬 /zgbs/品类 畅销榜 |\n| \"XX在亚马逊卖得怎么样\" | 搜索 /s?k=XX 看月销 |\n| \"分析一下这个ASIN: BXXXXXXXXX\" | 爬 /dp/ASIN 详情页 |\n| \"XX品类有什么机会\" | 畅销榜 + 搜索 综合分析 |\n| \"帮我爬这个链接\" | 判断URL类型，选对应handler |\n| \"帮我抓XX网站的内容\" | 通用模式 |\n| \"搜一下XX的竞品\" | 搜索页爬取 + 分析 |\n| \"XX月销多少\" / \"XX一个月卖多少\" | 搜索页或详情页 |\n| \"帮我看看top 100\" / \"热门产品\" | Best Sellers畅销榜 |\n| \"新品有哪些\" / \"最近上了什么新品\" | /zg/new-releases/ |\n| \"什么产品涨得快\" / \"飙升榜\" | /zg/movers-and-shakers/ |\n\n## 反爬能力\n- 每次清除Cookie，模拟全新用户\n- Docker沙箱隔离，无指纹追踪\n- Playwright模拟真实浏览器行为\n- 自动滚动加载懒加载内容\n- 支持重试（maxRetries: 2）\n\n## 局限\n- 通用模式输出上限10000字符\n- Amazon单页最多约30-50个产品\n- 不支持需要登录的页面\n- Docker容器启动有~10秒冷启动时间\n\nFile v3.2.0:_meta.json\n\n{\n  \"ownerId\": \"kn7ewmmms6dthpk632rzrc05v981ah4f\",\n  \"slug\": \"amazon-scraper\",\n  \"version\": \"3.2.0\",\n  \"publishedAt\": 1776831637963\n}\n\nFile v3.2.0:package.json\n\n{\n  \"name\": \"amazon-scraper\",\n  \"version\": \"3.2.0\",\n  \"description\": \"High-performance containerized web scraper for Amazon (BSR, search, product detail) and generic dynamic pages. Built on Docker + Crawlee + Playwright.\",\n  \"main\": \"assets/amazon_handler.js\",\n  \"scripts\": {\n    \"test\": \"node assets/main_handler.js https://www.amazon.com/zgbs/electronics\",\n    \"setup\": \"bash scripts/setup.sh\"\n  },\n  \"keywords\": [\n    \"amazon\",\n    \"scraper\",\n    \"asin\",\n    \"bsr\",\n    \"crawlee\",\n    \"playwright\",\n    \"openclaw\",\n    \"docker\"\n  ],\n  \"author\": \"Joseph\",\n  \"license\": \"MIT\",\n  \"dependencies\": {\n    \"crawlee\": \"^3.0.0\",\n    \"playwright\": \"^1.40.0\"\n  },\n  \"openclaw\": {\n    \"requires\": {\n      \"docker\": true\n    },\n    \"requiredBinaries\": [\"docker\"]\n  }\n}\n\nArchive v3.1.9: 6 files, 8828 bytes\n\nFiles: assets/amazon_handler.js (14735b), assets/main_handler.js (1379b), package.json (759b), scripts/setup.sh (422b), SKILL.md (6294b), _meta.json (133b)\n\nFile v3.1.9:SKILL.md\n\n---\nname: amazon-scraper\ndescription: >\n  High-performance containerized Amazon scraper (Docker + Crawlee + Playwright).\n  Supports Amazon BSR, search results, and product detail pages. Also includes a generic mode for other dynamic web pages.\n  Use when user mentions any of these:\n  爬虫, 爬取, 抓取, 采集, 数据采集, 爬数据, 抓数据, 获取数据,\n  scrape, crawl, extract, fetch data, pull data,\n  亚马逊, Amazon, ASIN, BSR, Best Sellers, 畅销榜, 热销榜, 新品榜, 飙升榜, 排行榜,\n  选品, 竞品分析, 竞品调研, 市场调研, 品类分析, 类目分析, 产品调研,\n  月销量, bought in past month, 销量, 评论数, 价格对比,\n  网页内容, 网站数据, 页面抓取, 动态页面,\n  关键词搜索, 搜索结果, search results,\n  产品详情, 产品信息, listing数据, listing分析,\n  top 100, top sellers, 热门产品, 爆款, 跑量款,\n  价格带, 评分分布, review分析, 评论分析\n---\n\n# Amazon Scraper\n\nDocker容器化爬虫，专为亚马逊数据采集优化，同时支持通用动态网页爬取。\n\n## ⚙️ 系统要求\n\n- **Docker Engine 20.10+**（必须已安装并运行）\n- **磁盘空间**：~2GB（镜像 + Playwright 浏览器二进制文件）\n- **内存**：建议 2GB+（Playwright 运行时需要）\n\n## 快速开始\n\n首次使用：在 skill 目录下执行一键构建脚本：\n\n```bash\nbash scripts/setup.sh\n```\n\n脚本会自动完成：构建 `clawd-crawlee` 镜像 + 创建 `~/scrapes` 输出目录。\n\n## 模式选择规则\n\n### 1. Amazon模式 (`amazon_handler.js`)\n**自动触发条件:** URL包含 `amazon.com`，或用户提到亚马逊/Amazon/ASIN/BSR/选品/竞品/畅销榜/类目分析等关键词\n\n根据URL自动识别页面类型：\n\n| URL特征 | 页面类型 | 可获取字段 |\n|---|---|---|\n| `/zgbs/` 或 `/bestsellers/` | 畅销榜 | rank, title, asin, price, rating, reviews, image, url |\n| `/zg/new-releases/` | 新品榜 | 同上 |\n| `/zg/movers-and-shakers/` | 飙升榜 | 同上 |\n| `/s?k=` 或 `/s/` | 搜索结果 | title, asin, price, rating, reviews, image, url, **boughtPastMonth**, sponsored |\n| `/dp/` 或 `/gp/product/` | 产品详情 | title, asin, price, rating, reviews, brand, bsr, **boughtPastMonth**, dateFirstAvailable, category, bullets, details, image |\n\n**⚠️ 重要规则:**\n- **Best Sellers页面没有月销量(boughtPastMonth)数据** — 亚马逊不在榜单页显示此信息\n- **要获取月销量，必须用搜索页(`/s?k=关键词`)或产品详情页(`/dp/ASIN`)**\n- 如果用户同时需要排名+月销量，建议：先爬Best Sellers拿排名，再用搜索页补月销\n\n```bash\n# 畅销榜（有排名，无月销）\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"https://www.amazon.com/zgbs/electronics\"\n\n# 搜索结果（有月销，无排名）\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"https://www.amazon.com/s?k=feather+duster\"\n\n# 产品详情（最全字段：BSR、品牌、卖点、月销）\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"https://www.amazon.com/dp/B001TQ6IHS\"\n\n# 多页爬取\ndocker run -t --rm clawd-crawlee node assets/amazon_handler.js \"URL\" --pages 2\n```\n\n**输出格式:** JSON\n```json\n{\n  \"status\": \"SUCCESS\",\n  \"type\": \"bestsellers|search|product-detail\",\n  \"category\": \"品类名\",\n  \"totalProducts\": 30,\n  \"scrapedAt\": \"ISO时间\",\n  \"products\": [\n    {\n      \"rank\": 1,\n      \"title\": \"产品名\",\n      \"asin\": \"B001TQ6IHS\",\n      \"price\": 9.94,\n      \"priceStr\": \"$9.94\",\n      \"rating\": 4.6,\n      \"reviews\": 20547,\n      \"boughtPastMonth\": \"1K+\",\n      \"image\": \"https://...\",\n      \"url\": \"https://...\"\n    }\n  ]\n}\n```\n\n### 2. 通用模式 (`main_handler.js`)\n**触发条件:** 非Amazon的URL，或用户提到爬取/抓取任意网页内容\n\n- Playwright打开页面，等待JS加载完成\n- 提取 `document.body.innerText`（纯文本，去广告噪音）\n- 输出上限10000字符\n- 输出: `{status:\"SUCCESS\", type:\"GENERIC\", title, data}`\n\n```bash\ndocker run -t --rm clawd-crawlee node assets/main_handler.js \"https://任意网址\"\n```\n\n## Agent调用决策树\n\n```\n用户给了URL?\n├─ 包含 amazon.com → 用 amazon_handler.js\n│   ├─ 需要月销量? → 建议用搜索URL(/s?k=) 或详情页(/dp/)\n│   └─ 需要排名? → 用畅销榜URL(/zgbs/)\n└─ 其他网站 → 用 main_handler.js (通用模式)\n\n用户没给URL，只说了需求?\n├─ \"爬亚马逊XX品类Top\" / \"XX类目排行\" / \"XX畅销榜\" → 构造 https://www.amazon.com/zgbs/品类\n├─ \"搜亚马逊XX\" / \"XX关键词搜索\" / \"找XX产品\" → 构造 https://www.amazon.com/s?k=关键词\n├─ \"分析某个ASIN\" / \"看看这个产品\" / \"XX的详情\" → 构造 https://www.amazon.com/dp/ASIN\n├─ \"XX的月销量\" / \"XX卖了多少\" / \"XX销量怎么样\" → 用搜索页或详情页（有boughtPastMonth）\n├─ \"竞品分析\" / \"竞品调研\" / \"对手在卖什么\" → 先搜索再逐个爬详情\n├─ \"选品\" / \"什么好卖\" / \"品类机会\" / \"市场调研\" → Best Sellers + 搜索结合\n└─ 其他网页 → 先web_search找到URL，再用通用模式爬\n```\n\n## 常见用户意图 → 操作映射\n\n| 用户说 | 操作 |\n|---|---|\n| \"帮我看看亚马逊XX品类\" | 爬 /zgbs/品类 畅销榜 |\n| \"XX在亚马逊卖得怎么样\" | 搜索 /s?k=XX 看月销 |\n| \"分析一下这个ASIN: BXXXXXXXXX\" | 爬 /dp/ASIN 详情页 |\n| \"XX品类有什么机会\" | 畅销榜 + 搜索 综合分析 |\n| \"帮我爬这个链接\" | 判断URL类型，选对应handler |\n| \"帮我抓XX网站的内容\" | 通用模式 |\n| \"搜一下XX的竞品\" | 搜索页爬取 + 分析 |\n| \"XX月销多少\" / \"XX一个月卖多少\" | 搜索页或详情页 |\n| \"帮我看看top 100\" / \"热门产品\" | Best Sellers畅销榜 |\n| \"新品有哪些\" / \"最近上了什么新品\" | /zg/new-releases/ |\n| \"什么产品涨得快\" / \"飙升榜\" | /zg/movers-and-shakers/ |\n\n## 反爬能力\n- 每次清除Cookie，模拟全新用户\n- Docker沙箱隔离，无指纹追踪\n- Playwright模拟真实浏览器行为\n- 自动滚动加载懒加载内容\n- 支持重试（maxRetries: 2）\n\n## 局限\n- 通用模式输出上限10000字符\n- Amazon单页最多约30-50个产品\n- 不支持需要登录的页面\n- Docker容器启动有~10秒冷启动时间\n\nFile v3.1.9:_meta.json\n\n{\n  \"ownerId\": \"kn7ewmmms6dthpk632rzrc05v981ah4f\",\n  \"slug\": \"amazon-scraper\",\n  \"version\": \"3.1.9\",\n  \"publishedAt\": 1776829381638\n}\n\nFile v3.1.9:package.json\n\n{\n  \"name\": \"amazon-scraper\",\n  \"version\": \"3.1.9\",\n  \"description\": \"High-performance containerized web scraper for Amazon (BSR, search, product detail) and generic dynamic pages. Built on Docker + Crawlee + Playwright.\",\n  \"main\": \"assets/amazon_handler.js\",\n  \"scripts\": {\n    \"test\": \"node assets/main_handler.js https://www.amazon.com/zgbs/electronics\",\n    \"setup\": \"bash scripts/setup.sh\"\n  },\n  \"keywords\": [\n    \"amazon\",\n    \"scraper\",\n    \"asin\",\n    \"bsr\",\n    \"crawlee\",\n    \"playwright\",\n    \"openclaw\",\n    \"docker\"\n  ],\n  \"author\": \"Joseph\",\n  \"license\": \"MIT\",\n  \"dependencies\": {\n    \"crawlee\": \"^3.0.0\",\n    \"playwright\": \"^1.40.0\"\n  },\n  \"openclaw\": {\n    \"requires\": {\n      \"docker\": true\n    },\n    \"requiredBinaries\": [\"docker\"]\n  }\n}","readmeExcerpt":"Skill: amazon-scraper Owner: jiafar Summary: Containerized Amazon.com scraper (Docker + Playwright) for BSR/new-releases/movers rankings, keyword search results, and product detail pages. Requires a paid ISP/residential proxy. Use when the request is specifically about Amazon product or marketplace data: 亚马逊/Amazon, ASIN, BSR, Best Sellers, 畅销榜, 新品榜, 飙升榜, 选品, 竞品分析, 类目分析, listing 分析, 月销量 (bought in past month), 评分分布, ","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"curl -s -x \"$AMAZON_PROXIES\" http://api.ipify.org"},{"language":"bash","snippet":"export AMAZON_PROXIES=\"http://USER:PASS@HOST:PORT\"   # 从密码管理器取，不要贴进对话\ncurl -s -x \"$AMAZON_PROXIES\" http://api.ipify.org"},{"language":"bash","snippet":"bash scripts/setup.sh"},{"language":"bash","snippet":"# 畅销榜（有排名，无月销，最多 60 个）\ndocker run --rm amazon-scraper node assets/amazon_handler.js \"https://www.amazon.com/gp/bestsellers/electronics\"\n\n# 搜索结果（有月销，多页可达 100+）\ndocker run --rm amazon-scraper node assets/amazon_handler.js \"https://www.amazon.com/s?k=feather+duster\"\n\n# 产品详情（最全字段：BSR、品牌、卖点、月销）\ndocker run --rm amazon-scraper node assets/amazon_handler.js \"https://www.amazon.com/dp/B001TQ6IHS\"\n\n# 多页爬取（搜索页建议 5 页拿 Top 100）\ndocker run --rm amazon-scraper node assets/amazon_handler.js \"URL\" --pages 2\n\n# 保存结果到文件（必须挂载 /data 才能在主机读到）\ndocker run --rm -v ~/scrapes:/data amazon-scraper node assets/amazon_handler.js \"URL\" --output result.json\n\n# 用自己的代理覆盖内置配置\ndocker run --rm -e AMAZON_PROXIES=\"http://user:***@host:8001,...\" amazon-scraper node assets/amazon_handler.js \"URL\""},{"language":"json","snippet":"{\n  \"status\": \"SUCCESS\",\n  \"failedJobs\": [],\n  \"failures\": [],\n  \"type\": \"bestsellers|search|product-detail\",\n  \"category\": \"品类名\",\n  \"totalProducts\": 30,\n  \"scrapedAt\": \"ISO时间\",\n  \"products\": [\n    {\n      \"rank\": 1,\n      \"title\": \"产品名\",\n      \"asin\": \"B001TQ6IHS\",\n      \"price\": 9.94,\n      \"priceStr\": \"$9.94\",\n      \"rating\": 4.6,\n      \"reviews\": 20547,\n      \"boughtPastMonth\": \"1K+\",\n      \"image\": \"https://...\",\n      \"url\": \"https://...\",\n      \"sponsored\": false\n    }\n  ]\n}"},{"language":"bash","snippet":"# 通用爬取（代理已内置）\ndocker run --rm amazon-scraper node assets/main_handler.js \"https://任意网址\"\n\n# 保存文件\ndocker run --rm -v ~/scrapes:/data \\\n  amazon-scraper node assets/main_handler.js \"https://任意网址\" --output page.json"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: amazon-scraper\ndescription: >\n  Containerized Amazon.com scraper (Docker + Playwright) for BSR/new-releases/movers rankings,\n  keyword search results, and product detail pages. Requires a paid ISP/residential proxy.\n\n  Use when the request is specifically about Amazon product or marketplace data:\n  亚马逊/Amazon, ASIN, BSR, Best Sellers, 畅销榜, 新品榜, 飙升榜,\n  选品, 竞品分析, 类目分析, listing 分析, 月销量 (bought in past month),\n  评分分布, 评论分析, Amazon 关键词搜索结果, Amazon 产品详情.\n\n  Do NOT use for general-purpose web scraping just because the user said\n  爬取/抓取/采集/scrape/crawl. Every run spends metered proxy bandwidth, and the\n  generic mode exists only as a fallback for pages related to an Amazon task.\n  For unrelated sites prefer web_search/web_fetch, or ask first.\nmetadata:\n  openclaw:\n    requires:\n      bins:\n        - docker\n---\n\n# Amazon Scraper\n\nDocker 容器化爬虫，Playwright Chromium。不使用 stealth 插件。支持亚马逊榜单/搜索/详情及通用动态页。\n\n## 第 0 步（开跑前必须先做，不过就停）\n\n先确认出口，再碰亚马逊：\n\n```bash\nexport AMAZON_PROXIES=\"http://USER:PASS@HOST:PORT\"   # 从密码管理器取，不要贴进对话\ncurl -s -x \"$AMAZON_PROXIES\" http://api.ipify.org\n```\n\n返回的纯文本必须等于你配置的代理 host。不是这个 IP、超时、407、403，都停，不要开爬，也不要重建后直接跑。\n\n- **凭证不进仓库、不进镜像、不进命令回显。** 用 `-e AMAZON_PROXIES`（值从环境取，不要写在命令行里）或挂载 `config/proxies.json`。该文件已在 `.gitignore` / `.dockerignore` 里。\n- 这一步只证明代理层通。裸 curl 打亚马逊拿到 500/202，不算第 0 步失败。\n- 第 0 步过了，才允许 `docker run amazon-scraper`。\n\n## 系统要求\n\n- **Docker Engine 20.10+**（必须已安装并运行）\n- **磁盘空间**：~2GB（镜像 + Playwright 浏览器二进制文件）\n- **内存**：建议 2GB+（Playwright 运行时需要）\n\n## 快速开始\n\n首次使用：在 skill 目录下执行一键构建脚本：\n\n```bash\nbash scripts/setup.sh\n```\n\n脚本会自动完成：构建 `amazon-scraper` 镜像 + 创建 `~/scrapes` 输出目录。\n\n\n## 模式选择规则\n\n### 1. Amazon模式 (`amazon_handler.js`)\n**自动触发条件:** URL包含 `amazon.com`，或用户提到亚马逊/Amazon/ASIN/BSR/选品/竞品/畅销榜/类目分析等关键词\n\n根据URL自动识别页面类型：\n\n| URL特征 | 页面类型 | 可获取字段 |\n|---|---|---|\n| `/gp/bestsellers/` | 畅销榜 | rank, title, asin, price, rating, reviews, image, url |\n| `/zg/new-releases/` | 新品榜 | 同上 |\n| `/zg/movers-and-shakers/` | 飙升榜 | 同上 |\n| `/s?k=` 或 `/s/` | 搜索结果 | title, asin, price, rating, reviews, image, url, **boughtPastMonth**, sponsored |\n| `/dp/` 或 `/gp/product/` | 产品详情 | title, asin, price, rating, reviews, brand, bsr, **boughtPastMonth**, **seller**, dateFirstAvailable, category, bullets, details |\n\n**⚠️ 重要规则:**\n- **Best Sellers 页面没有月销量(boughtPastMonth)数据** — 亚马逊不在榜单页显示此信息\n- **要获取月销量，必须用搜索页(`/s?k=关键词`)或产品详情页(`/dp/ASIN`)**\n- 如果用户同时需要排名+月销量，建议：先爬 Best Sellers 拿排名，再用搜索页补月销\n- **BSR URL 必须使用 `/gp/bestsellers/`**，`/zgbs/` 会返回 Page Not Found\n- **BSR 单 URL 只能拿 60 个产品**（2 页限制）。拿 Top 100 用搜索页 `--pages 5` 或多子类目 BSR 合并。详见 `references/bsr-top100-strategy.md`\n- **评论（rating/title/body/date/helpful/verified）不在这张表里**：需要登录态 + CDP，走 `scripts/scrape_reviews.py`（`/portal/customer-reviews/ASIN` 瀑布流）。先读 `references/reviews-strategy.md` 顶部的账号/合规代价说明\n- 视觉化选品 fallback：把 `image` URL 喂给 `vision_analyze`，让它\"看图识物\"。参考 `references/product-form-fusion.md`\n\n```bash\n# 畅销榜（有排名，无月销，最多 60 个）\ndocker run --rm amazon-scraper node assets/amazon_handler.js \"https://www.amazon.com/gp/bestsellers/"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7ewmmms6dthpk632rzrc05v981ah4f\",\n  \"slug\": \"amazon-scraper\",\n  \"version\": \"4.0.1\",\n  \"publishedAt\": 1790926130735\n}"},{"path":"references/batch-parallel-scraping.md","content":"# Batch Parallel Scraping Pattern\n\n**当前不适用。** 活代理只有 1 条（出口 IP 和端口见你本地的 `config/proxies.json`，不要写进文档）。代码把并发卡成 `min(请求值, 任务数, proxies.length)`，所以 `--concurrency 5` 仍是 1。`-e AMAZON_PROXIES` 在 `proxies.json` 非空时不生效。下面的 5 容器 × Oxylabs 8001–8005 是旧方案，现在执行会让 5 个浏览器打同一个 IP。\n\n大批量在只有 1 条 IP 时：一个容器，`--asins`，`--output` 必须配 `-v 宿主机目录:/data`，超过约 10 分钟改后台。先用 `http://api.ipify.org` 确认出口 IP 等于你配置的代理 host，再跑亚马逊。裸 curl 拿到 500/202 不算爬虫失败，以 handler 的 JSON 为准。\n\n旧方案（要有 5 个不同出口才能用，且必须挂载覆盖 `/app/config/proxies.json`，不能靠环境变量）：\n\nSafe high-throughput detail page scraping using multiple Docker containers with dedicated proxy ports.\n\n## Problem\nSingle container `--asins \"100_asins\" --concurrency 5` is slow (~20 min for 100 ASINs). But naive parallelism (5 containers × concurrency 5 = 25 total concurrent requests) causes ISP proxy soft-blocks — 76% of ASINs return all-null data.\n\n## Safe Pattern: 5 Containers × 2-3 Concurrency\n\n```bash\nPROXY_USER=\"user-XXX\"\nPROXY_PASS=\"XXX\"\n\n# Split 100 ASINs into 5 batches of 20\n# Each batch → its own Docker container with a dedicated proxy port\n\nfor i in 0 1 2 3 4; do\n  PORT=$((8001 + i))\n  BATCH=\"<20 ASINs comma-separated>\"\n  docker run --rm -v ~/scrapes:/data \\\n    -e AMAZON_PROXIES=\"http://${PROXY_USER}:${PROXY_PASS}@isp.oxylabs.io:${PORT}\" \\\n    amazon-scraper node assets/amazon_handler.js \\\n    --asins \"$BATCH\" --concurrency 2 \\\n    --output \"batch_${i}.json\" \\\n    > \"batch_${i}.log\" 2>&1 &\ndone\nwait\n```\n\n## Key Parameters\n\n| Parameter | Safe Value | Risk Value | Notes |\n|---|---|---|---|\n| Containers | 5 | >8 | One per proxy port (8001-8005) |\n| Concurrency per container | 2-3 | ≥5 | Each proxy port handles 2-3 concurrent connections |\n| Total concurrent | 10-15 | ≥25 | >15 triggers Amazon soft-blocks on ISP proxies |\n| ASINs per batch | 20 | >30 | Bigger batches = longer single-container runtime |\n\n## Performance\n\n| Mode | 100 ASINs | Success Rate |\n|---|---|---|\n| Single container, concurrency 5 | ~20 min | ~90% |\n| 5 containers × concurrency 3 | ~5 min | ~25% (too aggressive) |\n| 5 containers × concurrency 2 | ~5-7 min | ~25% (still aggressive on retry) |\n| 5 containers × concurrency 2, then retry failed with concurrency 1 | ~10 min | ~50% cumulative |\n\n## Retry Strategy for Failed ASINs\n\nAfter the first pass, collect ASINs that returned all-null (proxy soft-blocked), then retry with lower concurrency:\n\n```python\n# Collect failed ASINs\nfor p in all_products:\n    if not (p.get('title') or p.get('brand')):\n        failed_asins.append(p['asin'])\n```\n\nRetry with `--concurrency 1` or `--concurrency 2` on the same 5-container pattern. Typical improvement: +10-15% success on retry.\n\n## Merging Results\n\nAfter all batches + retries complete, merge by ASIN — prefer rows with actual data (title/brand not null):\n\n```python\ndetail_map = {}\nfor prefix in ['batch_', 'retry_']:\n    for i in range(5):\n        d = json.load(open(f'{prefix}{i}.json'))\n        for p in d['products']:\n            if p.get('title') or p.get('brand'):  # has real data\n  "},{"path":"references/bsr-top100-strategy.md","content":"# Amazon BSR Top 100 抓取策略\n\n## 核心限制（必读）\n\n**`/gp/bestsellers/` URL 只能拿到 2 页 = 60 个产品。** Amazon 官方限制。`?pg=3` 会返回 \"Page Not Found\"。\n\n这意味着：\n- 拿 Top 30 ✅ 直接 `/gp/bestsellers/{category}`\n- 拿 Top 50/60 ✅ `/gp/bestsellers/{category} --pages 2`\n- **拿 Top 100 ❌ 单个 BSR URL 不行**\n\n## 三种 Top 100 策略\n\n### 策略 A：搜索页替代（最快、推荐）⭐\n\n```bash\ndocker run --rm -v /tmp/top100:/data amazon-scraper \\\n  node assets/amazon_handler.js \\\n  \"https://www.amazon.com/s?k=cable+management\" \\\n  --pages 5 \\\n  --output /data/cm.json\n```\n\n- 一次调用，~3 分钟\n- 5 页 = 100+ 个产品（去重后约 60-80 个独立 ASIN）\n- **数据是搜索算法排序的，不是 BSR 严格排名**（混了广告位）\n- 对\"市场分析\"够用\n\n**适合**：快速拿数据做品类分析、价格带分析、视觉调研。\n\n### 策略 B：多子类目 BSR 合并\n\nBSR 父类目下钻到 5 个子类目，每个拿 Top 30 = 150 个产品：\n\n```python\n# 主类目 electronics 没有子节点\n# 子类目节点 ID（在 URL 里能看到）\nsubcats = [\n    \"electronics/172541\",      # Audio & Video\n    \"electronics/281407\",      # Computers & Accessories\n    \"electronics/2407745011\",  # Wearable Technology\n    \"electronics/13896617011\", # Computer & Accessories\n    \"electronics/3024167031\",  # Cell Phones\n]\n# 拼 URL: https://www.amazon.com/gp/bestsellers/{subcat}\n```\n\n每个子 BSR 限 2 页 = 30 个。**5 × 30 = 150，去重 ~120 个**。\n\n**适合**：要做严格的\"畅销榜\"分析（不被广告位污染）。\n\n### 策略 C：单 session 串行（最稳但最慢）\n\n把 BSR Top 30 拿到 ASIN，逐个爬详情，3 个代理并发：\n\n```bash\n# /tmp/top100.sh\nwhile read asin; do\n  docker run --rm amazon-scraper node assets/amazon_handler.js \\\n    \"https://www.amazon.com/dp/${asin}\" \\\n    --output \"/tmp/details/${asin}.json\" 2>/dev/null\n  sleep 3\ndone < /tmp/asins.txt\n```\n\n**耗时**：30 个 ASIN × 15-25 秒 = 7-12 分钟（单容器，1 代理）\n\n**适合**：要拿详情做品牌/BSR/详情页分析。\n\n## 提速方案\n\n| 方案 | 速度 | 限制 |\n|---|---|---|\n| 单容器 `--pages 5` | 1x | skill 本身 |\n| 多 Docker 容器并发 | Nx（N=容器数） | 需 N 个不同代理 |\n| 改 handler.js 加并发 | 内部可控 | 需重新 build 镜像 |\n\n**多容器并发的现实约束**：你的代理数 = 最大并发数。\n- 3 个 DDC IP → 最多 3 个并发 → 3x 提速\n- 5 个 ISP 端口 → 最多 5 个并发 → 5x 提速\n\nskill 自带轮询：handler 内部已支持 1 个容器内多代理轮询 + 故障切换，无需额外配置。\n\n## 速度与限流\n\n- 单个 session：~30 秒/页（含 15 秒冷启动 + 3-5 秒 waitFor）\n- Amazon 限流：30-60 请求/小时/同一 IP 是安全线，超了会触发 503\n- 跑 Top 100 一次消耗约 5-10 个\"请求单位\"\n\n## 输出文件路径\n\n⚠️ **坑**：`--output` 是**容器内路径**。要保存到主机必须挂载卷：\n\n```bash\n# 错：文件在容器里\ndocker run --rm amazon-scraper node .../amazon_handler.js \"URL\" --output /tmp/x.json\n\n# 对：挂载 /data\ndocker run --rm -v /tmp/results:/data amazon-scraper node .../amazon_handler.js \\\n  \"URL\" --output /data/x.json\n# 文件在主机的 /tmp/results/x.json\n```\n\n代码里 `--output` 的路径会拼到 `/data/` 前缀下。"},{"path":"references/cdp-fallback-strategy.md","content":"# CDP Fallback 爬取策略\n\n## 问题场景\n\n批量爬取 Amazon 详情页时（`--asins` 模式），Docker 代理方案在以下情况会被 Amazon 软拦截：\n- 总并发 ≥ 15（5 容器 × concurrency 3）\n- 单代理端口并发 ≥ 3\n- 短时间内同 IP 大量请求\n\n软拦截表现：`status: SUCCESS` 但 `title/brand/seller/bsr` 全部 null，页面未渲染。\n\n## 解决方案：Chrome CDP 直连\n\n**核心思路**：绕过 Docker 代理，用 VPS 本地 Chrome 浏览器（CDP 协议）逐个串行爬取，配合 2-4 秒随机延迟。\n\n**实测结果（2026-07-04）**：\n- Docker 代理方案：75 个失败 ASIN，成功率 0%（全部被拦）\n- CDP 直连方案：75 个失败 ASIN，**成功率 100%**（0 失败）\n- 耗时：75 个 ASIN × ~5 秒/个 = 约 6 分钟\n\n## 前置条件\n\n1. Chrome 已启动并监听 CDP：\n\n```bash\nexport DISPLAY=:99\nXvfb :99 -screen 0 1920x1080x24 &>/dev/null &\n/opt/google/chrome/chrome --disable-gpu --no-first-run \\\n  --no-default-browser-check \\\n  --remote-debugging-port=9222 \\\n  --remote-debugging-address=127.0.0.1 \\\n  --user-data-dir=\"$HOME/.cache/amazon-scraper-chrome\" \\\n  \"https://www.amazon.com\"\n```\n\n> ### ⚠️ 不要加 `--remote-allow-origins=*`\n>\n> CDP **没有任何认证机制**：谁能连上 9222，谁就完全控制这个浏览器 —— 读 cookie、以登录用户身份下单、改收货地址、导出 session。本方案的前提恰恰是这个 Chrome 带着真实 Amazon 登录态，所以它的调试口就等于账号凭证。\n>\n> - `--remote-allow-origins=*` 关掉了 WebSocket 的 Origin 校验，于是**你在这个浏览器里打开的任意网页**都能连上 9222 接管它。只在确实遇到 Origin 报错时，针对具体来源写白名单，不要用 `*`。\n> - 必须显式 `--remote-debugging-address=127.0.0.1`。绑到 `0.0.0.0` 的 VPS 等于开了一个公网无密码浏览器后门；即使绑回环，也要确认没有端口转发或 docker 规则把它暴露出去（`ss -lntp | grep 9222` 自查）。\n> - 不要用 `--no-sandbox`。VPS 上常以 root 跑 Chrome，关掉沙箱意味着一个渲染器漏洞就能拿到 root。需要在容器里跑就改用非 root 用户 + `--user-ns` 之类的方案。\n> - profile 不要放 `/tmp`（全局可写，其他本地用户可读你的 cookie）。放 `$HOME/.cache/...` 并保持 `chmod 700`。\n> - 用完把这个 Chrome 关掉，别长期挂着一个带登录态的调试口。\n\n本目录的两个 CDP 脚本都会**自己新开一个标签页**并在结束时关掉，不会劫持你正在用的标签页。\n\n2. Python 依赖：\n```bash\npip install websocket-client\n```\n\n## 使用方法\n\n```bash\n# 方式1：直接传 ASIN 列表\npython3 ~/.openclaw/skills/amazon-scraper/scripts/cdp_fallback_scrape.py \\\n  --asins \"B07XXX,B08YYY,B09ZZZ\" \\\n  --output cdp_results.json\n\n# 方式2：从 JSON 文件读 ASIN 列表\npython3 ~/.openclaw/skills/amazon-scraper/scripts/cdp_fallback_scrape.py \\\n  --asin-file failed_asins.json \\\n  --output cdp_results.json\n```\n\n## 完整工作流：Docker 批量 + CDP 补漏\n\n```bash\n# Step 1: Docker 批量爬取（快但有失败）\n# 每个容器一个独立出口端口。没有 N 个不同出口就不要开 N 个容器。\n# 凭证从环境变量取，不要写进命令行（会进 shell history 和 ps 输出）。\nread -rsp 'proxy user: ' PROXY_USER; echo\nread -rsp 'proxy pass: ' PROXY_PASS; echo\n\nfor i in 0 1 2 3 4; do\n  PORT=$((8001 + i))\n  AMAZON_PROXIES=\"http://${PROXY_USER}:${PROXY_PASS}@isp.oxylabs.io:${PORT}\" \\\n  docker run --rm -v ~/scrapes:/data \\\n    -e AMAZON_PROXIES \\\n    amazon-scraper node assets/amazon_handler.js \\\n    --asins \"$BATCH_$i\" --concurrency 2 \\\n    --output \"batch_${i}.json\" &\ndone\nwait\n\n# Step 2: 找出失败的 ASIN（title/brand 全 null）\npython3 -c \"\nimport json\nfailed = []\nfor i in range(5):\n    d = json.load(open(f'batch_{i}.json'))\n    for p in d.get('products', []):\n        if not (p.get('title') or p.get('brand')):\n            failed.append(p['asin'])\njson.dump(failed, open('failed_asins.json', 'w'))\nprint(f'{len(failed)} failed ASINs')\n\"\n\n# Step 3: CDP 补漏（串行，100% 成功率）\npython3 ~/.openclaw/skills/amazon-scraper/scripts/cdp_fallback_scrape.py \\\n  --asin-file failed_asins.json \\\n  --outp"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":1352,"uniquenessScore":43,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T05:12:46.315Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T05:12:46.315Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T12:55:56.038Z","emptyReason":null},"items":[{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-10T18:48:31.762Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}