{"id":"025abcd1-b2d5-4b59-b4f4-e9a4555ad194","entityType":"agent","slug":"clawhub-smallkeyboy-102-playwright-scraper-skill","name":"102 Playwright Scraper Skill","canonicalUrl":"https://www.xpersona.co/agent/clawhub-smallkeyboy-102-playwright-scraper-skill","canonicalPath":"/agent/clawhub-smallkeyboy-102-playwright-scraper-skill","generatedAt":"2026-10-11T17:42:16.059Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-11T14:54:26.941Z","emptyReason":null},"description":"Playwright-based web scraping OpenClaw Skill with anti-bot protection. Successfully tested on complex sites like Discuss.com.hk. Skill: 102 Playwright Scraper Skill Owner: smallkeyboy Summary: Playwright-based web scraping OpenClaw Skill with anti-bot protection. Successfully tested on complex sites like Discuss.com.hk. Tags: latest:1.0.0 Version history: v1.0.0 | 2026-04-16T07:09:40.650Z | auto **Playwright Scraper Skill v1.2.0:** - Added comprehensive usage guide and script descriptions, including detailed guidance for various website anti-b","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s170nr6nxfsrk9n96c9r21ydcx83j5pf:102-playwright-scraper-skill","sourceUrl":"https://clawhub.ai/smallkeyboy/102-playwright-scraper-skill","homepage":"https://clawhub.ai/smallkeyboy/skills/102-playwright-scraper-skill","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/smallkeyboy/102-playwright-scraper-skill","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/smallkeyboy/skills/102-playwright-scraper-skill","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":60,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Playwright-based web scraping OpenClaw Skill with anti-bot protection. Successfully tested on complex sites like Discuss.com.hk. Skill: 102 Playwright Scraper S"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T14:54:26.941Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T14:54:26.941Z","emptyReason":null},"stars":null,"forks":null,"downloads":1046,"packageName":null,"latestVersion":"1.0.0","tractionLabel":"1K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T14:54:26.934Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T14:54:26.941Z","lastCrawledAt":"2026-10-11T14:54:26.934Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T14:54:26.934Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.0","createdAt":"2026-04-16T07:09:40.650Z","changelog":"**Playwright Scraper Skill v1.2.0:** - Added comprehensive usage guide and script descriptions, including detailed guidance for various website anti-bot levels. - Introduced a use case matrix and performance comparison chart for choosing the best scraping method. - Documented anti-bot protection strategies employed in the provided scripts. - Included troubleshooting steps, environmental variable customization, and best practices for higher scraping success. - Outlined future improvement plans and provided helpful external references.","fileCount":15,"zipByteSize":19961}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s170nr6nxfsrk9n96c9r21ydcx83j5pf:102-playwright-scraper-skill","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-smallkeyboy-102-playwright-scraper-skill/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-smallkeyboy-102-playwright-scraper-skill/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-smallkeyboy-102-playwright-scraper-skill/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-smallkeyboy-102-playwright-scraper-skill/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-smallkeyboy-102-playwright-scraper-skill/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-smallkeyboy-102-playwright-scraper-skill/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T17:42:16.057Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-smallkeyboy-102-playwright-scraper-skill/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-smallkeyboy-102-playwright-scraper-skill/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-smallkeyboy-102-playwright-scraper-skill/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-smallkeyboy-102-playwright-scraper-skill/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-11T14:54:26.941Z","emptyReason":null},"readme":"Skill: 102 Playwright Scraper Skill\n\nOwner: smallkeyboy\n\nSummary: Playwright-based web scraping OpenClaw Skill with anti-bot protection. Successfully tested on complex sites like Discuss.com.hk.\n\nTags: latest:1.0.0\n\nVersion history:\n\nv1.0.0 | 2026-04-16T07:09:40.650Z | auto\n\n**Playwright Scraper Skill v1.2.0:**\n\n- Added comprehensive usage guide and script descriptions, including detailed guidance for various website anti-bot levels.\n- Introduced a use case matrix and performance comparison chart for choosing the best scraping method.\n- Documented anti-bot protection strategies employed in the provided scripts.\n- Included troubleshooting steps, environmental variable customization, and best practices for higher scraping success.\n- Outlined future improvement plans and provided helpful external references.\n\nArchive index:\n\nArchive v1.0.0: 15 files, 19961 bytes\n\nFiles: _meta.json (147b), CHANGELOG.md (1625b), CONTRIBUTING.md (3011b), examples/discuss-hk.sh (432b), examples/README.md (4207b), INSTALL.md (2203b), package-lock.json (1696b), package.json (531b), README_ZH.md (4267b), README.md (4471b), scripts/playwright-simple.js (1755b), scripts/playwright-stealth.js (5572b), skill-card.md (2395b), SKILL.md (6155b), test.sh (1110b)\n\nFile v1.0.0:SKILL.md\n\n---\nname: playwright-scraper-skill\ndescription: Playwright-based web scraping OpenClaw Skill with anti-bot protection. Successfully tested on complex sites like Discuss.com.hk.\nversion: 1.2.0\nauthor: Simon Chan\n---\n\n# Playwright Scraper Skill\n\nA Playwright-based web scraping OpenClaw Skill with anti-bot protection. Choose the best approach based on the target website's anti-bot level.\n\n---\n\n## 🎯 Use Case Matrix\n\n| Target Website | Anti-Bot Level | Recommended Method | Script |\n|---------------|----------------|-------------------|--------|\n| **Regular Sites** | Low | web_fetch tool | N/A (built-in) |\n| **Dynamic Sites** | Medium | Playwright Simple | `scripts/playwright-simple.js` |\n| **Cloudflare Protected** | High | **Playwright Stealth** ⭐ | `scripts/playwright-stealth.js` |\n| **YouTube** | Special | deep-scraper | Install separately |\n| **Reddit** | Special | reddit-scraper | Install separately |\n\n---\n\n## 📦 Installation\n\n```bash\ncd playwright-scraper-skill\nnpm install\nnpx playwright install chromium\n```\n\n---\n\n## 🚀 Quick Start\n\n### 1️⃣ Simple Sites (No Anti-Bot)\n\nUse OpenClaw's built-in `web_fetch` tool:\n\n```bash\n# Invoke directly in OpenClaw\nHey, fetch me the content from https://example.com\n```\n\n---\n\n### 2️⃣ Dynamic Sites (Requires JavaScript)\n\nUse **Playwright Simple**:\n\n```bash\nnode scripts/playwright-simple.js \"https://example.com\"\n```\n\n**Example output:**\n```json\n{\n  \"url\": \"https://example.com\",\n  \"title\": \"Example Domain\",\n  \"content\": \"...\",\n  \"elapsedSeconds\": \"3.45\"\n}\n```\n\n---\n\n### 3️⃣ Anti-Bot Protected Sites (Cloudflare etc.)\n\nUse **Playwright Stealth**:\n\n```bash\nnode scripts/playwright-stealth.js \"https://m.discuss.com.hk/#hot\"\n```\n\n**Features:**\n- Hide automation markers (`navigator.webdriver = false`)\n- Realistic User-Agent (iPhone, Android)\n- Random delays to mimic human behavior\n- Screenshot and HTML saving support\n\n---\n\n### 4️⃣ YouTube Video Transcripts\n\nUse **deep-scraper** (install separately):\n\n```bash\n# Install deep-scraper skill\nnpx clawhub install deep-scraper\n\n# Use it\ncd skills/deep-scraper\nnode assets/youtube_handler.js \"https://www.youtube.com/watch?v=VIDEO_ID\"\n```\n\n---\n\n## 📖 Script Descriptions\n\n### `scripts/playwright-simple.js`\n- **Use Case:** Regular dynamic websites\n- **Speed:** Fast (3-5 seconds)\n- **Anti-Bot:** None\n- **Output:** JSON (title, content, URL)\n\n### `scripts/playwright-stealth.js` ⭐\n- **Use Case:** Sites with Cloudflare or anti-bot protection\n- **Speed:** Medium (5-20 seconds)\n- **Anti-Bot:** Medium-High (hides automation, realistic UA)\n- **Output:** JSON + Screenshot + HTML file\n- **Verified:** 100% success on Discuss.com.hk\n\n---\n\n## 🎓 Best Practices\n\n### 1. Try web_fetch First\nIf the site doesn't have dynamic loading, use OpenClaw's `web_fetch` tool—it's fastest.\n\n### 2. Need JavaScript? Use Playwright Simple\nIf you need to wait for JavaScript rendering, use `playwright-simple.js`.\n\n### 3. Getting Blocked? Use Stealth\nIf you encounter 403 or Cloudflare challenges, use `playwright-stealth.js`.\n\n### 4. Special Sites Need Specialized Skills\n- YouTube → deep-scraper\n- Reddit → reddit-scraper\n- Twitter → bird skill\n\n---\n\n## 🔧 Customization\n\nAll scripts support environment variables:\n\n```bash\n# Set screenshot path\nSCREENSHOT_PATH=/path/to/screenshot.png node scripts/playwright-stealth.js URL\n\n# Set wait time (milliseconds)\nWAIT_TIME=10000 node scripts/playwright-simple.js URL\n\n# Enable headful mode (show browser)\nHEADLESS=false node scripts/playwright-stealth.js URL\n\n# Save HTML\nSAVE_HTML=true node scripts/playwright-stealth.js URL\n\n# Custom User-Agent\nUSER_AGENT=\"Mozilla/5.0 ...\" node scripts/playwright-stealth.js URL\n```\n\n---\n\n## 📊 Performance Comparison\n\n| Method | Speed | Anti-Bot | Success Rate (Discuss.com.hk) |\n|--------|-------|----------|-------------------------------|\n| web_fetch | ⚡ Fastest | ❌ None | 0% |\n| Playwright Simple | 🚀 Fast | ⚠️ Low | 20% |\n| **Playwright Stealth** | ⏱️ Medium | ✅ Medium | **100%** ✅ |\n| Puppeteer Stealth | ⏱️ Medium | ✅ Medium-High | ~80% |\n| Crawlee (deep-scraper) | 🐢 Slow | ❌ Detected | 0% |\n| Chaser (Rust) | ⏱️ Medium | ❌ Detected | 0% |\n\n---\n\n## 🛡️ Anti-Bot Techniques Summary\n\nLessons learned from our testing:\n\n### ✅ Effective Anti-Bot Measures\n1. **Hide `navigator.webdriver`** — Essential\n2. **Realistic User-Agent** — Use real devices (iPhone, Android)\n3. **Mimic Human Behavior** — Random delays, scrolling\n4. **Avoid Framework Signatures** — Crawlee, Selenium are easily detected\n5. **Use `addInitScript` (Playwright)** — Inject before page load\n\n### ❌ Ineffective Anti-Bot Measures\n1. **Only changing User-Agent** — Not enough\n2. **Using high-level frameworks (Crawlee)** — More easily detected\n3. **Docker isolation** — Doesn't help with Cloudflare\n\n---\n\n## 🔍 Troubleshooting\n\n### Issue: 403 Forbidden\n**Solution:** Use `playwright-stealth.js`\n\n### Issue: Cloudflare Challenge Page\n**Solution:**\n1. Increase wait time (10-15 seconds)\n2. Try `headless: false` (headful mode sometimes has higher success rate)\n3. Consider using proxy IPs\n\n### Issue: Blank Page\n**Solution:**\n1. Increase `waitForTimeout`\n2. Use `waitUntil: 'networkidle'` or `'domcontentloaded'`\n3. Check if login is required\n\n---\n\n## 📝 Memory & Experience\n\n### 2026-02-07 Discuss.com.hk Test Conclusions\n- ✅ **Pure Playwright + Stealth** succeeded (5s, 200 OK)\n- ❌ Crawlee (deep-scraper) failed (403)\n- ❌ Chaser (Rust) failed (Cloudflare)\n- ❌ Puppeteer standard failed (403)\n\n**Best Solution:** Pure Playwright + anti-bot techniques (framework-independent)\n\n---\n\n## 🚧 Future Improvements\n\n- [ ] Add proxy IP rotation\n- [ ] Implement cookie management (maintain login state)\n- [ ] Add CAPTCHA handling (2captcha / Anti-Captcha)\n- [ ] Batch scraping (parallel URLs)\n- [ ] Integration with OpenClaw's `browser` tool\n\n---\n\n## 📚 References\n\n- [Playwright Official Docs](https://playwright.dev/)\n- [puppeteer-extra-plugin-stealth](https://github.com/berstend/puppeteer-extra/tree/master/packages/puppeteer-extra-plugin-stealth)\n- [deep-scraper skill](https://clawhub.com/opsun/deep-scraper)\n\nFile v1.0.0:examples/README.md\n\n# Usage Examples\n\n## Basic Usage\n\n### 1. Quick Scrape (Example.com)\n\n```bash\nnode scripts/playwright-simple.js https://example.com\n```\n\n**Output:**\n```json\n{\n  \"title\": \"Example Domain\",\n  \"url\": \"https://example.com/\",\n  \"content\": \"Example Domain\\n\\nThis domain is for use...\",\n  \"metaDescription\": \"\",\n  \"elapsedSeconds\": \"3.42\"\n}\n```\n\n---\n\n### 2. Anti-Bot Protected Site (Discuss.com.hk)\n\n```bash\nnode scripts/playwright-stealth.js \"https://m.discuss.com.hk/#hot\"\n```\n\n**Output:**\n```json\n{\n  \"title\": \"香港討論區 discuss.com.hk\",\n  \"url\": \"https://m.discuss.com.hk/#hot\",\n  \"htmlLength\": 186345,\n  \"contentPreview\": \"...\",\n  \"cloudflare\": false,\n  \"screenshot\": \"./screenshot-1770467444364.png\",\n  \"data\": {\n    \"links\": [\n      {\n        \"text\": \"區議員周潔瑩疑消防通道違泊 道歉稱急於搬貨\",\n        \"href\": \"https://m.discuss.com.hk/index.php?action=thread&tid=32148378...\"\n      }\n    ]\n  },\n  \"elapsedSeconds\": \"19.59\"\n}\n```\n\n---\n\n## Advanced Usage\n\n### 3. Custom Wait Time\n\n```bash\nWAIT_TIME=15000 node scripts/playwright-stealth.js <URL>\n```\n\n### 4. Show Browser (Debug Mode)\n\n```bash\nHEADLESS=false node scripts/playwright-stealth.js <URL>\n```\n\n### 5. Save Screenshot and HTML\n\n```bash\nSCREENSHOT_PATH=/tmp/my-page.png \\\nSAVE_HTML=true \\\nnode scripts/playwright-stealth.js <URL>\n```\n\n### 6. Custom User-Agent\n\n```bash\nUSER_AGENT=\"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36\" \\\nnode scripts/playwright-stealth.js <URL>\n```\n\n---\n\n## Integration Examples\n\n### Using in Shell Scripts\n\n```bash\n#!/bin/bash\n# Run from playwright-scraper-skill directory\n\nURL=\"https://example.com\"\nOUTPUT_FILE=\"result.json\"\n\necho \"🕷️  Starting scrape: $URL\"\n\nnode scripts/playwright-stealth.js \"$URL\" > \"$OUTPUT_FILE\"\n\nif [ $? -eq 0 ]; then\n  echo \"✅ Success! Results saved to: $OUTPUT_FILE\"\nelse\n  echo \"❌ Failed\"\n  exit 1\nfi\n```\n\n### Batch Scraping Multiple URLs\n\n```bash\n#!/bin/bash\n\nURLS=(\n  \"https://example.com\"\n  \"https://example.org\"\n  \"https://example.net\"\n)\n\nfor url in \"${URLS[@]}\"; do\n  echo \"Scraping: $url\"\n  node scripts/playwright-stealth.js \"$url\" > \"output_$(date +%s).json\"\n  sleep 5  # Avoid IP blocking\ndone\n```\n\n---\n\n## Calling from Node.js\n\n```javascript\nconst { spawn } = require('child_process');\n\nfunction scrape(url) {\n  return new Promise((resolve, reject) => {\n    const proc = spawn('node', [\n      'scripts/playwright-stealth.js',\n      url\n    ]);\n    \n    let output = '';\n    \n    proc.stdout.on('data', (data) => {\n      output += data.toString();\n    });\n    \n    proc.on('close', (code) => {\n      if (code === 0) {\n        try {\n          // Extract JSON (last line)\n          const lines = output.trim().split('\\n');\n          const json = JSON.parse(lines[lines.length - 1]);\n          resolve(json);\n        } catch (e) {\n          reject(e);\n        }\n      } else {\n        reject(new Error(`Exit code: ${code}`));\n      }\n    });\n  });\n}\n\n// Usage\n(async () => {\n  const result = await scrape('https://example.com');\n  console.log(result.title);\n})();\n```\n\n---\n\n## Common Scenarios\n\n### Scraping News Articles\n\n```bash\nnode scripts/playwright-stealth.js \"https://news.example.com/article/123\"\n```\n\n### Scraping E-commerce Products\n\n```bash\nWAIT_TIME=10000 \\\nSAVE_HTML=true \\\nnode scripts/playwright-stealth.js \"https://shop.example.com/product/456\"\n```\n\n### Scraping Forum Posts\n\n```bash\nnode scripts/playwright-stealth.js \"https://forum.example.com/thread/789\"\n```\n\n---\n\n## Troubleshooting\n\n### Issue: Page Not Fully Loaded\n\n**Solution:** Increase wait time\n```bash\nWAIT_TIME=20000 node scripts/playwright-stealth.js <URL>\n```\n\n### Issue: Still Blocked by Cloudflare\n\n**Solution:** Use headful mode + manual wait\n```bash\nHEADLESS=false \\\nWAIT_TIME=30000 \\\nnode scripts/playwright-stealth.js <URL>\n```\n\n### Issue: Requires Login\n\n**Solution:** Manually login first, export cookies, then load\n(Future feature, currently not supported)\n\n---\n\n## Performance Tips\n\n1. **Parallel scraping:** Use `Promise.all()` or shell `&`\n2. **Delay requests:** `sleep 5` to avoid IP blocking\n3. **Use proxies:** Rotate IPs (future feature)\n4. **Cache results:** Avoid duplicate scraping\n\n---\n\nFor more information, see [SKILL.md](../SKILL.md)\n\nFile v1.0.0:README.md\n\n# Playwright Scraper Skill 🕷️\n\n[![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)\n[![Node.js](https://img.shields.io/badge/Node.js-18+-green.svg)](https://nodejs.org/)\n[![Playwright](https://img.shields.io/badge/Playwright-1.40+-blue.svg)](https://playwright.dev/)\n\n**[中文文檔](README_ZH.md)** | English\n\nA Playwright-based web scraping OpenClaw Skill with anti-bot protection. Successfully tested on complex websites like Discuss.com.hk.\n\n> 📦 **Installation:** See [INSTALL.md](INSTALL.md)  \n> 📚 **Full Documentation:** See [SKILL.md](SKILL.md)  \n> 💡 **Examples:** See [examples/README.md](examples/README.md)\n\n---\n\n## ✨ Features\n\n- ✅ **Pure Playwright** — Modern, powerful, easy to use\n- ✅ **Anti-Bot Protection** — Hides automation, realistic UA\n- ✅ **Verified** — 100% success on Discuss.com.hk\n- ✅ **Simple to Use** — One-line commands\n- ✅ **Customizable** — Environment variable support\n\n---\n\n## 🚀 Quick Start\n\n### Installation\n\n```bash\nnpm install\nnpx playwright install chromium\n```\n\n### Usage\n\n```bash\n# Quick scraping\nnode scripts/playwright-simple.js https://example.com\n\n# Stealth mode (recommended)\nnode scripts/playwright-stealth.js \"https://m.discuss.com.hk/#hot\"\n```\n\n---\n\n## 📖 Two Modes\n\n| Mode | Use Case | Speed | Anti-Bot |\n|------|----------|-------|----------|\n| **Simple** | Regular dynamic sites | Fast (3-5s) | None |\n| **Stealth** ⭐ | Sites with anti-bot | Medium (5-20s) | Medium-High |\n\n### Simple Mode\n\nFor sites without anti-bot protection:\n\n```bash\nnode scripts/playwright-simple.js <URL>\n```\n\n### Stealth Mode (Recommended)\n\nFor sites with Cloudflare or anti-bot protection:\n\n```bash\nnode scripts/playwright-stealth.js <URL>\n```\n\n**Anti-Bot Techniques:**\n- Hide `navigator.webdriver`\n- Realistic User-Agent (iPhone)\n- Human-like behavior simulation\n- Screenshot and HTML saving support\n\n---\n\n## 🎯 Customization\n\nAll scripts support environment variables:\n\n```bash\n# Show browser\nHEADLESS=false node scripts/playwright-stealth.js <URL>\n\n# Custom wait time (milliseconds)\nWAIT_TIME=10000 node scripts/playwright-stealth.js <URL>\n\n# Save screenshot\nSCREENSHOT_PATH=/tmp/page.png node scripts/playwright-stealth.js <URL>\n\n# Save HTML\nSAVE_HTML=true node scripts/playwright-stealth.js <URL>\n\n# Custom User-Agent\nUSER_AGENT=\"Mozilla/5.0 ...\" node scripts/playwright-stealth.js <URL>\n```\n\n---\n\n## 📊 Test Results\n\n| Website | Result | Time |\n|---------|--------|------|\n| **Discuss.com.hk** | ✅ 200 OK | 5-20s |\n| **Example.com** | ✅ 200 OK | 3-5s |\n| **Cloudflare Protected** | ✅ Mostly successful | 10-30s |\n\n---\n\n## 📁 File Structure\n\n```\nplaywright-scraper-skill/\n├── scripts/\n│   ├── playwright-simple.js       # Simple mode\n│   └── playwright-stealth.js      # Stealth mode ⭐\n├── examples/\n│   ├── discuss-hk.sh              # Discuss.com.hk example\n│   └── README.md                  # More examples\n├── SKILL.md                       # Full documentation\n├── INSTALL.md                     # Installation guide\n├── README.md                      # This file\n├── README_ZH.md                   # Chinese documentation\n├── CONTRIBUTING.md                # Contribution guide\n├── CHANGELOG.md                   # Version history\n└── package.json                   # npm config\n```\n\n---\n\n## 💡 Best Practices\n\n1. **Try web_fetch first** — OpenClaw's built-in tool is fastest\n2. **Use Simple for dynamic sites** — When no anti-bot protection\n3. **Use Stealth for protected sites** ⭐ — Main workhorse\n4. **Use specialized skills** — For YouTube, Reddit, etc.\n\n---\n\n## 🐛 Troubleshooting\n\n### Getting 403 blocked?\n\nUse Stealth mode:\n```bash\nnode scripts/playwright-stealth.js <URL>\n```\n\n### Cloudflare challenge?\n\nIncrease wait time + headful mode:\n```bash\nHEADLESS=false WAIT_TIME=30000 node scripts/playwright-stealth.js <URL>\n```\n\n### Playwright not found?\n\nReinstall:\n```bash\nnpm install\nnpx playwright install chromium\n```\n\nMore issues? See [INSTALL.md](INSTALL.md)\n\n---\n\n## 🤝 Contributing\n\nContributions welcome! See [CONTRIBUTING.md](CONTRIBUTING.md)\n\n---\n\n## 📄 License\n\nMIT License - See [LICENSE](LICENSE)\n\n---\n\n## 🔗 Links\n\n- [Playwright Official Docs](https://playwright.dev/)\n- [Full Documentation (SKILL.md)](SKILL.md)\n- [Installation Guide (INSTALL.md)](INSTALL.md)\n- [Examples (examples/)](examples/)\n\nFile v1.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn7ez4h284sw1q2f7zvmge91wx83jdbt\",\n  \"slug\": \"102-playwright-scraper-skill\",\n  \"version\": \"1.0.0\",\n  \"publishedAt\": 1776323380650\n}\n\nFile v1.0.0:CHANGELOG.md\n\n# Changelog\n\n## [1.2.0] - 2026-02-07\n\n### 🔄 Major Changes\n\n- **Project Renamed** — `web-scraper` → `playwright-scraper-skill`\n- Updated all documentation and links\n- Updated GitHub repo name\n- **Bilingual Documentation** — All docs now in English (with Chinese README available)\n\n---\n\n## [1.1.0] - 2026-02-07\n\n### ✅ Added\n\n- **LICENSE** — MIT License\n- **CONTRIBUTING.md** — Contribution guidelines\n- **examples/README.md** — Detailed usage examples\n- **test.sh** — Automated test script\n- **README.md** — Redesigned with badges\n\n### 🔧 Improvements\n\n- Clearer file structure\n- More detailed documentation\n- More practical examples\n\n---\n\n## [1.0.0] - 2026-02-07\n\n### ✅ Initial Release\n\n**Tools Created:**\n- ✅ `playwright-simple.js` — Fast simple scraper\n- ✅ `playwright-stealth.js` — Anti-bot protected version (primary) ⭐\n\n**Test Results:**\n- ✅ Discuss.com.hk success (200 OK, 19.6s)\n- ✅ Example.com success (3.4s)\n- ✅ Auto fallback to deep-scraper's Playwright\n\n**Documentation:**\n- ✅ SKILL.md (full documentation)\n- ✅ README.md (quick reference)\n- ✅ Example scripts (discuss-hk.sh)\n- ✅ package.json\n\n**Key Findings:**\n1. **Playwright Stealth is the best solution** (100% success on Discuss.com.hk)\n2. **Don't use Crawlee** (easily detected)\n3. **Chaser (Rust) doesn't work currently** (blocked by Cloudflare)\n4. **Hiding `navigator.webdriver` is key**\n\n---\n\n## Future Plans\n\n- [ ] Add proxy IP rotation\n- [ ] CAPTCHA handling integration\n- [ ] Cookie management (maintain login state)\n- [ ] Batch scraping (parallel processing)\n- [ ] Integration with OpenClaw browser tool\n\nFile v1.0.0:CONTRIBUTING.md\n\n# Contributing Guide\n\nThank you for considering contributing to playwright-scraper-skill!\n\n## 🐛 Reporting Issues\n\nIf you find a bug or have a feature suggestion:\n\n1. Check [Issues](https://github.com/waisimon/playwright-scraper-skill/issues) to see if it already exists\n2. If not, create a new Issue\n3. Provide the following information:\n   - Problem description\n   - Steps to reproduce\n   - Expected vs actual behavior\n   - Environment (Node.js version, OS)\n   - Error messages (if any)\n\n## 💡 Feature Requests\n\n1. Create an Issue with `[Feature Request]` in the title\n2. Explain:\n   - The desired feature\n   - Use cases\n   - Why this feature would be useful\n\n## 🔧 Submitting Code\n\n### Setting Up Development Environment\n\n```bash\n# Fork the repo and clone\ngit clone https://github.com/YOUR_USERNAME/playwright-scraper-skill.git\ncd playwright-scraper-skill\n\n# Install dependencies\nnpm install\nnpx playwright install chromium\n\n# Test\nnode scripts/playwright-simple.js https://example.com\n```\n\n### Contribution Workflow\n\n1. Create a new branch:\n   ```bash\n   git checkout -b feature/my-new-feature\n   ```\n\n2. Make your changes\n\n3. Test your changes:\n   ```bash\n   npm test\n   node scripts/playwright-stealth.js <test-URL>\n   ```\n\n4. Commit:\n   ```bash\n   git add .\n   git commit -m \"Add: brief description of changes\"\n   ```\n\n5. Push and create a Pull Request:\n   ```bash\n   git push origin feature/my-new-feature\n   ```\n\n### Commit Message Guidelines\n\nUse clear commit messages:\n\n- `Add: new feature`\n- `Fix: issue description`\n- `Update: existing feature`\n- `Refactor: code refactoring`\n- `Docs: documentation update`\n- `Test: add or modify tests`\n\nExample:\n```\nFix: playwright-stealth.js screenshot timeout issue\n\n- Increase timeout parameter to 10 seconds\n- Add try-catch error handling\n- Update documentation\n```\n\n## 📝 Documentation\n\nIf your changes affect usage:\n\n- Update `SKILL.md` (full documentation)\n- Update `README.md` (quick reference)\n- Update `examples/README.md` (if adding new examples)\n- Update `CHANGELOG.md` (record changes)\n\n## ✅ Checklist\n\nBefore submitting a PR, confirm:\n\n- [ ] Code runs properly\n- [ ] Doesn't break existing functionality\n- [ ] Updated relevant documentation\n- [ ] Clear commit messages\n- [ ] No sensitive information (API keys, personal paths, etc.)\n\n## 🎯 Priority Areas\n\nCurrently welcoming contributions in:\n\n1. **New anti-bot techniques** — Improve success rates\n2. **Support more websites** — Test and share success cases\n3. **Performance optimization** — Speed up scraping\n4. **Error handling** — Better error messages and recovery\n5. **Documentation improvements** — Clearer explanations and examples\n\n## 🚫 Unaccepted Contributions\n\n- Adding complex dependencies (keep it lightweight)\n- Features violating privacy or laws\n- Breaking existing API changes (unless well justified)\n\n## 📞 Contact\n\nHave questions? Feel free to:\n- Create an Issue for discussion\n- Ask in Pull Request comments\n\n---\n\nThank you for your contribution! 🙏\n\nFile v1.0.0:INSTALL.md\n\n# Installation Guide\n\n## 📦 Quick Installation\n\n### 1. Clone or Download the Skill\n\n```bash\n# Method 1: Using git clone (if public repo)\ngit clone https://github.com/waisimon/playwright-scraper-skill.git\ncd playwright-scraper-skill\n\n# Method 2: Download ZIP and extract\n# After downloading, enter the directory\ncd playwright-scraper-skill\n```\n\n### 2. Install Dependencies\n\n```bash\n# Install Playwright (recommended)\nnpm install\n\n# Install browser (Chromium)\nnpx playwright install chromium\n```\n\n### 3. Test\n\n```bash\n# Quick test\nnode scripts/playwright-simple.js https://example.com\n\n# Test Stealth version\nnode scripts/playwright-stealth.js https://example.com\n```\n\n---\n\n## 🔧 Advanced Installation\n\n### Using with OpenClaw\n\nIf you're using OpenClaw, you can place this skill in the skills directory:\n\n```bash\n# Assuming your OpenClaw workspace is at ~/.openclaw/workspace\ncp -r playwright-scraper-skill ~/.openclaw/workspace/skills/\n\n# Then you can invoke it in OpenClaw\n```\n\n---\n\n## ✅ Verify Installation\n\nRun the example script:\n\n```bash\n# Discuss.com.hk example (verified working)\nbash examples/discuss-hk.sh\n```\n\nIf you see output similar to this, installation is successful:\n\n```\n🕷️  Starting Playwright Stealth scraper...\n📱 Navigating to: https://m.discuss.com.hk/#hot\n📡 HTTP Status: 200\n✅ Scraping complete!\n```\n\n---\n\n## 🐛 Common Issues\n\n### Issue: Playwright not found\n\n**Error message:** `Error: Cannot find module 'playwright'`\n\n**Solution:**\n```bash\nnpm install\nnpx playwright install chromium\n```\n\n### Issue: Browser launch failed\n\n**Error message:** `browserType.launch: Executable doesn't exist`\n\n**Solution:**\n```bash\nnpx playwright install chromium\n```\n\n### Issue: Permission errors\n\n**Error message:** `Permission denied`\n\n**Solution:**\n```bash\nchmod +x scripts/*.js\nchmod +x examples/*.sh\n```\n\n---\n\n## 📝 System Requirements\n\n- **Node.js:** v18+ recommended\n- **OS:** macOS / Linux / Windows\n- **Disk Space:** ~500MB (including Chromium)\n- **RAM:** 2GB+ recommended\n\n---\n\n## 🚀 Next Steps\n\nAfter installation, check out:\n- [README.md](README.md) — Quick reference\n- [SKILL.md](SKILL.md) — Full documentation\n- [examples/](examples/) — Example scripts\n\nFile v1.0.0:README_ZH.md\n\n# Playwright Scraper Skill 🕷️\n\n[![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)\n[![Node.js](https://img.shields.io/badge/Node.js-18+-green.svg)](https://nodejs.org/)\n[![Playwright](https://img.shields.io/badge/Playwright-1.40+-blue.svg)](https://playwright.dev/)\n\n基於 Playwright 的網頁爬蟲 OpenClaw Skill。支援反爬保護，已驗證成功爬取 Discuss.com.hk 等複雜網站。\n\n> 📦 **安裝方法：** 查看 [INSTALL.md](INSTALL.md)  \n> 📚 **完整文件：** 查看 [SKILL.md](SKILL.md)  \n> 💡 **使用範例：** 查看 [examples/README.md](examples/README.md)\n\n---\n\n## ✨ 特色\n\n- ✅ **純 Playwright** — 現代、強大、易用\n- ✅ **反爬保護** — 隱藏自動化特徵、真實 UA\n- ✅ **已驗證** — Discuss.com.hk 100% 成功\n- ✅ **簡單易用** — 一行命令搞定\n- ✅ **可自訂** — 支援環境變數配置\n\n---\n\n## 🚀 快速開始\n\n### 安裝\n\n```bash\nnpm install\nnpx playwright install chromium\n```\n\n### 使用\n\n```bash\n# 快速爬取\nnode scripts/playwright-simple.js https://example.com\n\n# 反爬保護版（推薦）\nnode scripts/playwright-stealth.js \"https://m.discuss.com.hk/#hot\"\n```\n\n---\n\n## 📖 兩種模式\n\n| 模式 | 適用場景 | 速度 | 反爬能力 |\n|------|---------|------|----------|\n| **Simple** | 一般動態網站 | 快（3-5秒） | 無 |\n| **Stealth** ⭐ | 有反爬保護的網站 | 中（5-20秒） | 中高 |\n\n### Simple 模式\n\n適合沒有反爬保護的網站：\n\n```bash\nnode scripts/playwright-simple.js <URL>\n```\n\n### Stealth 模式（推薦）\n\n適合有 Cloudflare 或反爬保護的網站：\n\n```bash\nnode scripts/playwright-stealth.js <URL>\n```\n\n**反爬技巧：**\n- 隱藏 `navigator.webdriver`\n- 真實 User-Agent（iPhone）\n- 模擬真人行為\n- 支援截圖和 HTML 儲存\n\n---\n\n## 🎯 自訂參數\n\n所有腳本都支援環境變數：\n\n```bash\n# 顯示瀏覽器\nHEADLESS=false node scripts/playwright-stealth.js <URL>\n\n# 自訂等待時間（毫秒）\nWAIT_TIME=10000 node scripts/playwright-stealth.js <URL>\n\n# 儲存截圖\nSCREENSHOT_PATH=/tmp/page.png node scripts/playwright-stealth.js <URL>\n\n# 儲存 HTML\nSAVE_HTML=true node scripts/playwright-stealth.js <URL>\n\n# 自訂 User-Agent\nUSER_AGENT=\"Mozilla/5.0 ...\" node scripts/playwright-stealth.js <URL>\n```\n\n---\n\n## 📊 測試結果\n\n| 網站 | 結果 | 時間 |\n|------|------|------|\n| **Discuss.com.hk** | ✅ 200 OK | 5-20 秒 |\n| **Example.com** | ✅ 200 OK | 3-5 秒 |\n| **Cloudflare 保護網站** | ✅ 多數成功 | 10-30 秒 |\n\n---\n\n## 📁 檔案結構\n\n```\nplaywright-scraper-skill/\n├── scripts/\n│   ├── playwright-simple.js       # 簡單版\n│   └── playwright-stealth.js      # Stealth 版 ⭐\n├── examples/\n│   ├── discuss-hk.sh              # Discuss.com.hk 範例\n│   └── README.md                  # 更多範例\n├── SKILL.md                       # 完整文件\n├── INSTALL.md                     # 安裝指南\n├── README.md                      # 本檔案\n├── CONTRIBUTING.md                # 貢獻指南\n├── CHANGELOG.md                   # 版本記錄\n└── package.json                   # npm 配置\n```\n\n---\n\n## 💡 使用建議\n\n1. **先試 web_fetch** — OpenClaw 內建工具最快\n2. **動態網站用 Simple** — 沒有反爬保護時\n3. **反爬網站用 Stealth** ⭐ — 主力工具\n4. **特殊網站用專用 skill** — YouTube、Reddit 等\n\n---\n\n## 🐛 故障排除\n\n### 被 403 擋住？\n\n使用 Stealth 模式：\n```bash\nnode scripts/playwright-stealth.js <URL>\n```\n\n### Cloudflare 挑戰？\n\n增加等待時間 + 有頭模式：\n```bash\nHEADLESS=false WAIT_TIME=30000 node scripts/playwright-stealth.js <URL>\n```\n\n### 找不到 Playwright？\n\n重新安裝：\n```bash\nnpm install\nnpx playwright install chromium\n```\n\n更多問題查看 [INSTALL.md](INSTALL.md)\n\n---\n\n## 🤝 貢獻\n\n歡迎貢獻！查看 [CONTRIBUTING.md](CONTRIBUTING.md)\n\n---\n\n## 📄 授權\n\nMIT License - 查看 [LICENSE](LICENSE)\n\n---\n\n## 🔗 相關連結\n\n- [Playwright 官方文檔](https://playwright.dev/)\n- [完整文件 (SKILL.md)](SKILL.md)\n- [安裝指南 (INSTALL.md)](INSTALL.md)\n- [使用範例 (examples/)](examples/)\n\nFile v1.0.0:skill-card.md\n\n## Description:\n\nPlaywright-based web scraping OpenClaw Skill with anti-bot protection. Successfully tested on complex sites like Discuss.com.hk.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[smallkeyboy](https://clawhub.ai/user/smallkeyboy)\n\n### License/Terms of Use:\n\nMIT\n\n## Use Case:\n\nDevelopers and engineers use this skill to collect content from dynamic or anti-bot-protected web pages with Playwright-based simple and stealth scraping modes.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The stealth scraper can access arbitrary URLs and is designed for anti-bot-protected pages.\n\nMitigation: Use it only against sites the operator is authorized to access, and run it in a disposable, low-privilege environment with restricted network access.\n\nRisk: The scraper can write screenshots and optional HTML captures to local paths.\n\nMitigation: Choose non-sensitive output locations and avoid scraping authenticated or sensitive pages unless necessary.\n\nRisk: The stealth mode weakens browser safety controls.\n\nMitigation: Review the scripts before deployment and keep execution isolated from credentials and sensitive local files.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/smallkeyboy/skills/102-playwright-scraper-skill)\n- [Playwright official docs](https://playwright.dev/)\n- [puppeteer-extra-plugin-stealth](https://github.com/berstend/puppeteer-extra/tree/master/packages/puppeteer-extra-plugin-stealth)\n- [deep-scraper skill](https://clawhub.com/opsun/deep-scraper)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, code, shell commands, configuration, files, guidance]\n\n**Output Format:** [Markdown guidance with shell commands; scraper scripts emit JSON and may write screenshot or HTML files.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Output may include page title, URL, content previews, elapsed time, Cloudflare detection status, extracted links, screenshot paths, and optional HTML capture paths.]\n\n## Skill Version(s):\n\n1.2.0 (source: frontmatter, package.json, CHANGELOG released 2026-02-07)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.0:package-lock.json\n\n{\n  \"name\": \"web-scraper\",\n  \"version\": \"1.0.0\",\n  \"lockfileVersion\": 3,\n  \"requires\": true,\n  \"packages\": {\n    \"\": {\n      \"name\": \"web-scraper\",\n      \"version\": \"1.0.0\",\n      \"license\": \"MIT\",\n      \"dependencies\": {\n        \"playwright\": \"^1.40.0\"\n      }\n    },\n    \"node_modules/fsevents\": {\n      \"version\": \"2.3.2\",\n      \"resolved\": \"https://registry.npmjs.org/fsevents/-/fsevents-2.3.2.tgz\",\n      \"integrity\": \"sha512-xiqMQR4xAeHTuB9uWm+fFRcIOgKBMiOBP+eXiyT7jsgVCq1bkVygt00oASowB7EdtpOHaaPgKt812P9ab+DDKA==\",\n      \"hasInstallScript\": true,\n      \"license\": \"MIT\",\n      \"optional\": true,\n      \"os\": [\n        \"darwin\"\n      ],\n      \"engines\": {\n        \"node\": \"^8.16.0 || ^10.6.0 || >=11.0.0\"\n      }\n    },\n    \"node_modules/playwright\": {\n      \"version\": \"1.58.2\",\n      \"resolved\": \"https://registry.npmjs.org/playwright/-/playwright-1.58.2.tgz\",\n      \"integrity\": \"sha512-vA30H8Nvkq/cPBnNw4Q8TWz1EJyqgpuinBcHET0YVJVFldr8JDNiU9LaWAE1KqSkRYazuaBhTpB5ZzShOezQ6A==\",\n      \"license\": \"Apache-2.0\",\n      \"dependencies\": {\n        \"playwright-core\": \"1.58.2\"\n      },\n      \"bin\": {\n        \"playwright\": \"cli.js\"\n      },\n      \"engines\": {\n        \"node\": \">=18\"\n      },\n      \"optionalDependencies\": {\n        \"fsevents\": \"2.3.2\"\n      }\n    },\n    \"node_modules/playwright-core\": {\n      \"version\": \"1.58.2\",\n      \"resolved\": \"https://registry.npmjs.org/playwright-core/-/playwright-core-1.58.2.tgz\",\n      \"integrity\": \"sha512-yZkEtftgwS8CsfYo7nm0KE8jsvm6i/PTgVtB8DL726wNf6H2IMsDuxCpJj59KDaxCtSnrWan2AeDqM7JBaultg==\",\n      \"license\": \"Apache-2.0\",\n      \"bin\": {\n        \"playwright-core\": \"cli.js\"\n      },\n      \"engines\": {\n        \"node\": \">=18\"\n      }\n    }\n  }\n}\n\nFile v1.0.0:package.json\n\n{\n  \"name\": \"playwright-scraper-skill\",\n  \"version\": \"1.2.0\",\n  \"description\": \"基於 Playwright 的網頁爬蟲 OpenClaw Skill\",\n  \"main\": \"scripts/playwright-stealth.js\",\n  \"scripts\": {\n    \"simple\": \"node scripts/playwright-simple.js\",\n    \"stealth\": \"node scripts/playwright-stealth.js\",\n    \"test\": \"bash test.sh\"\n  },\n  \"keywords\": [\n    \"scraper\",\n    \"playwright\",\n    \"puppeteer\",\n    \"cloudflare\",\n    \"anti-detection\"\n  ],\n  \"author\": \"多米\",\n  \"license\": \"MIT\",\n  \"dependencies\": {\n    \"playwright\": \"^1.40.0\"\n  }\n}","readmeExcerpt":"Skill: 102 Playwright Scraper Skill Owner: smallkeyboy Summary: Playwright-based web scraping OpenClaw Skill with anti-bot protection. Successfully tested on complex sites like Discuss.com.hk. Tags: latest:1.0.0 Version history: v1.0.0 | 2026-04-16T07:09:40.650Z | auto **Playwright Scraper Skill v1.2.0:** - Added comprehensive usage guide and script descriptions, including detailed guidance for various website anti-b","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"cd playwright-scraper-skill\nnpm install\nnpx playwright install chromium"},{"language":"bash","snippet":"# Invoke directly in OpenClaw\nHey, fetch me the content from https://example.com"},{"language":"bash","snippet":"node scripts/playwright-simple.js \"https://example.com\""},{"language":"json","snippet":"{\n  \"url\": \"https://example.com\",\n  \"title\": \"Example Domain\",\n  \"content\": \"...\",\n  \"elapsedSeconds\": \"3.45\"\n}"},{"language":"bash","snippet":"node scripts/playwright-stealth.js \"https://m.discuss.com.hk/#hot\""},{"language":"bash","snippet":"# Install deep-scraper skill\nnpx clawhub install deep-scraper\n\n# Use it\ncd skills/deep-scraper\nnode assets/youtube_handler.js \"https://www.youtube.com/watch?v=VIDEO_ID\""}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: playwright-scraper-skill\ndescription: Playwright-based web scraping OpenClaw Skill with anti-bot protection. Successfully tested on complex sites like Discuss.com.hk.\nversion: 1.2.0\nauthor: Simon Chan\n---\n\n# Playwright Scraper Skill\n\nA Playwright-based web scraping OpenClaw Skill with anti-bot protection. Choose the best approach based on the target website's anti-bot level.\n\n---\n\n## 🎯 Use Case Matrix\n\n| Target Website | Anti-Bot Level | Recommended Method | Script |\n|---------------|----------------|-------------------|--------|\n| **Regular Sites** | Low | web_fetch tool | N/A (built-in) |\n| **Dynamic Sites** | Medium | Playwright Simple | `scripts/playwright-simple.js` |\n| **Cloudflare Protected** | High | **Playwright Stealth** ⭐ | `scripts/playwright-stealth.js` |\n| **YouTube** | Special | deep-scraper | Install separately |\n| **Reddit** | Special | reddit-scraper | Install separately |\n\n---\n\n## 📦 Installation\n\n```bash\ncd playwright-scraper-skill\nnpm install\nnpx playwright install chromium\n```\n\n---\n\n## 🚀 Quick Start\n\n### 1️⃣ Simple Sites (No Anti-Bot)\n\nUse OpenClaw's built-in `web_fetch` tool:\n\n```bash\n# Invoke directly in OpenClaw\nHey, fetch me the content from https://example.com\n```\n\n---\n\n### 2️⃣ Dynamic Sites (Requires JavaScript)\n\nUse **Playwright Simple**:\n\n```bash\nnode scripts/playwright-simple.js \"https://example.com\"\n```\n\n**Example output:**\n```json\n{\n  \"url\": \"https://example.com\",\n  \"title\": \"Example Domain\",\n  \"content\": \"...\",\n  \"elapsedSeconds\": \"3.45\"\n}\n```\n\n---\n\n### 3️⃣ Anti-Bot Protected Sites (Cloudflare etc.)\n\nUse **Playwright Stealth**:\n\n```bash\nnode scripts/playwright-stealth.js \"https://m.discuss.com.hk/#hot\"\n```\n\n**Features:**\n- Hide automation markers (`navigator.webdriver = false`)\n- Realistic User-Agent (iPhone, Android)\n- Random delays to mimic human behavior\n- Screenshot and HTML saving support\n\n---\n\n### 4️⃣ YouTube Video Transcripts\n\nUse **deep-scraper** (install separately):\n\n```bash\n# Install deep-scraper skill\nnpx clawhub install deep-scraper\n\n# Use it\ncd skills/deep-scraper\nnode assets/youtube_handler.js \"https://www.youtube.com/watch?v=VIDEO_ID\"\n```\n\n---\n\n## 📖 Script Descriptions\n\n### `scripts/playwright-simple.js`\n- **Use Case:** Regular dynamic websites\n- **Speed:** Fast (3-5 seconds)\n- **Anti-Bot:** None\n- **Output:** JSON (title, content, URL)\n\n### `scripts/playwright-stealth.js` ⭐\n- **Use Case:** Sites with Cloudflare or anti-bot protection\n- **Speed:** Medium (5-20 seconds)\n- **Anti-Bot:** Medium-High (hides automation, realistic UA)\n- **Output:** JSON + Screenshot + HTML file\n- **Verified:** 100% success on Discuss.com.hk\n\n---\n\n## 🎓 Best Practices\n\n### 1. Try web_fetch First\nIf the site doesn't have dynamic loading, use OpenClaw's `web_fetch` tool—it's fastest.\n\n### 2. Need JavaScript? Use Playwright Simple\nIf you need to wait for JavaScript rendering, use `playwright-simple.js`.\n\n### 3. Getting Blocked? Use Stealth\nIf you encounter 403 or Cloudflare challenges, use `playwright-stealth.j"},{"path":"examples/README.md","content":"# Usage Examples\n\n## Basic Usage\n\n### 1. Quick Scrape (Example.com)\n\n```bash\nnode scripts/playwright-simple.js https://example.com\n```\n\n**Output:**\n```json\n{\n  \"title\": \"Example Domain\",\n  \"url\": \"https://example.com/\",\n  \"content\": \"Example Domain\\n\\nThis domain is for use...\",\n  \"metaDescription\": \"\",\n  \"elapsedSeconds\": \"3.42\"\n}\n```\n\n---\n\n### 2. Anti-Bot Protected Site (Discuss.com.hk)\n\n```bash\nnode scripts/playwright-stealth.js \"https://m.discuss.com.hk/#hot\"\n```\n\n**Output:**\n```json\n{\n  \"title\": \"香港討論區 discuss.com.hk\",\n  \"url\": \"https://m.discuss.com.hk/#hot\",\n  \"htmlLength\": 186345,\n  \"contentPreview\": \"...\",\n  \"cloudflare\": false,\n  \"screenshot\": \"./screenshot-1770467444364.png\",\n  \"data\": {\n    \"links\": [\n      {\n        \"text\": \"區議員周潔瑩疑消防通道違泊 道歉稱急於搬貨\",\n        \"href\": \"https://m.discuss.com.hk/index.php?action=thread&tid=32148378...\"\n      }\n    ]\n  },\n  \"elapsedSeconds\": \"19.59\"\n}\n```\n\n---\n\n## Advanced Usage\n\n### 3. Custom Wait Time\n\n```bash\nWAIT_TIME=15000 node scripts/playwright-stealth.js <URL>\n```\n\n### 4. Show Browser (Debug Mode)\n\n```bash\nHEADLESS=false node scripts/playwright-stealth.js <URL>\n```\n\n### 5. Save Screenshot and HTML\n\n```bash\nSCREENSHOT_PATH=/tmp/my-page.png \\\nSAVE_HTML=true \\\nnode scripts/playwright-stealth.js <URL>\n```\n\n### 6. Custom User-Agent\n\n```bash\nUSER_AGENT=\"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36\" \\\nnode scripts/playwright-stealth.js <URL>\n```\n\n---\n\n## Integration Examples\n\n### Using in Shell Scripts\n\n```bash\n#!/bin/bash\n# Run from playwright-scraper-skill directory\n\nURL=\"https://example.com\"\nOUTPUT_FILE=\"result.json\"\n\necho \"🕷️  Starting scrape: $URL\"\n\nnode scripts/playwright-stealth.js \"$URL\" > \"$OUTPUT_FILE\"\n\nif [ $? -eq 0 ]; then\n  echo \"✅ Success! Results saved to: $OUTPUT_FILE\"\nelse\n  echo \"❌ Failed\"\n  exit 1\nfi\n```\n\n### Batch Scraping Multiple URLs\n\n```bash\n#!/bin/bash\n\nURLS=(\n  \"https://example.com\"\n  \"https://example.org\"\n  \"https://example.net\"\n)\n\nfor url in \"${URLS[@]}\"; do\n  echo \"Scraping: $url\"\n  node scripts/playwright-stealth.js \"$url\" > \"output_$(date +%s).json\"\n  sleep 5  # Avoid IP blocking\ndone\n```\n\n---\n\n## Calling from Node.js\n\n```javascript\nconst { spawn } = require('child_process');\n\nfunction scrape(url) {\n  return new Promise((resolve, reject) => {\n    const proc = spawn('node', [\n      'scripts/playwright-stealth.js',\n      url\n    ]);\n    \n    let output = '';\n    \n    proc.stdout.on('data', (data) => {\n      output += data.toString();\n    });\n    \n    proc.on('close', (code) => {\n      if (code === 0) {\n        try {\n          // Extract JSON (last line)\n          const lines = output.trim().split('\\n');\n          const json = JSON.parse(lines[lines.length - 1]);\n          resolve(json);\n        } catch (e) {\n          reject(e);\n        }\n      } else {\n        reject(new Error(`Exit code: ${code}`));\n      }\n    });\n  });\n}\n\n// Usage\n(async () => {\n  const result = await scrape('https://example.com');\n  console.log(result.title);\n})();\n```\n\n---\n\n## Common Scen"},{"path":"README.md","content":"# Playwright Scraper Skill 🕷️\n\n[![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)\n[![Node.js](https://img.shields.io/badge/Node.js-18+-green.svg)](https://nodejs.org/)\n[![Playwright](https://img.shields.io/badge/Playwright-1.40+-blue.svg)](https://playwright.dev/)\n\n**[中文文檔](README_ZH.md)** | English\n\nA Playwright-based web scraping OpenClaw Skill with anti-bot protection. Successfully tested on complex websites like Discuss.com.hk.\n\n> 📦 **Installation:** See [INSTALL.md](INSTALL.md)  \n> 📚 **Full Documentation:** See [SKILL.md](SKILL.md)  \n> 💡 **Examples:** See [examples/README.md](examples/README.md)\n\n---\n\n## ✨ Features\n\n- ✅ **Pure Playwright** — Modern, powerful, easy to use\n- ✅ **Anti-Bot Protection** — Hides automation, realistic UA\n- ✅ **Verified** — 100% success on Discuss.com.hk\n- ✅ **Simple to Use** — One-line commands\n- ✅ **Customizable** — Environment variable support\n\n---\n\n## 🚀 Quick Start\n\n### Installation\n\n```bash\nnpm install\nnpx playwright install chromium\n```\n\n### Usage\n\n```bash\n# Quick scraping\nnode scripts/playwright-simple.js https://example.com\n\n# Stealth mode (recommended)\nnode scripts/playwright-stealth.js \"https://m.discuss.com.hk/#hot\"\n```\n\n---\n\n## 📖 Two Modes\n\n| Mode | Use Case | Speed | Anti-Bot |\n|------|----------|-------|----------|\n| **Simple** | Regular dynamic sites | Fast (3-5s) | None |\n| **Stealth** ⭐ | Sites with anti-bot | Medium (5-20s) | Medium-High |\n\n### Simple Mode\n\nFor sites without anti-bot protection:\n\n```bash\nnode scripts/playwright-simple.js <URL>\n```\n\n### Stealth Mode (Recommended)\n\nFor sites with Cloudflare or anti-bot protection:\n\n```bash\nnode scripts/playwright-stealth.js <URL>\n```\n\n**Anti-Bot Techniques:**\n- Hide `navigator.webdriver`\n- Realistic User-Agent (iPhone)\n- Human-like behavior simulation\n- Screenshot and HTML saving support\n\n---\n\n## 🎯 Customization\n\nAll scripts support environment variables:\n\n```bash\n# Show browser\nHEADLESS=false node scripts/playwright-stealth.js <URL>\n\n# Custom wait time (milliseconds)\nWAIT_TIME=10000 node scripts/playwright-stealth.js <URL>\n\n# Save screenshot\nSCREENSHOT_PATH=/tmp/page.png node scripts/playwright-stealth.js <URL>\n\n# Save HTML\nSAVE_HTML=true node scripts/playwright-stealth.js <URL>\n\n# Custom User-Agent\nUSER_AGENT=\"Mozilla/5.0 ...\" node scripts/playwright-stealth.js <URL>\n```\n\n---\n\n## 📊 Test Results\n\n| Website | Result | Time |\n|---------|--------|------|\n| **Discuss.com.hk** | ✅ 200 OK | 5-20s |\n| **Example.com** | ✅ 200 OK | 3-5s |\n| **Cloudflare Protected** | ✅ Mostly successful | 10-30s |\n\n---\n\n## 📁 File Structure\n\n```\nplaywright-scraper-skill/\n├── scripts/\n│   ├── playwright-simple.js       # Simple mode\n│   └── playwright-stealth.js      # Stealth mode ⭐\n├── examples/\n│   ├── discuss-hk.sh              # Discuss.com.hk example\n│   └── README.md                  # More examples\n├── SKILL.md                       # Full documentation\n├── INSTALL.md                     # Installation g"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7ez4h284sw1q2f7zvmge91wx83jdbt\",\n  \"slug\": \"102-playwright-scraper-skill\",\n  \"version\": \"1.0.0\",\n  \"publishedAt\": 1776323380650\n}"},{"path":"CHANGELOG.md","content":"# Changelog\n\n## [1.2.0] - 2026-02-07\n\n### 🔄 Major Changes\n\n- **Project Renamed** — `web-scraper` → `playwright-scraper-skill`\n- Updated all documentation and links\n- Updated GitHub repo name\n- **Bilingual Documentation** — All docs now in English (with Chinese README available)\n\n---\n\n## [1.1.0] - 2026-02-07\n\n### ✅ Added\n\n- **LICENSE** — MIT License\n- **CONTRIBUTING.md** — Contribution guidelines\n- **examples/README.md** — Detailed usage examples\n- **test.sh** — Automated test script\n- **README.md** — Redesigned with badges\n\n### 🔧 Improvements\n\n- Clearer file structure\n- More detailed documentation\n- More practical examples\n\n---\n\n## [1.0.0] - 2026-02-07\n\n### ✅ Initial Release\n\n**Tools Created:**\n- ✅ `playwright-simple.js` — Fast simple scraper\n- ✅ `playwright-stealth.js` — Anti-bot protected version (primary) ⭐\n\n**Test Results:**\n- ✅ Discuss.com.hk success (200 OK, 19.6s)\n- ✅ Example.com success (3.4s)\n- ✅ Auto fallback to deep-scraper's Playwright\n\n**Documentation:**\n- ✅ SKILL.md (full documentation)\n- ✅ README.md (quick reference)\n- ✅ Example scripts (discuss-hk.sh)\n- ✅ package.json\n\n**Key Findings:**\n1. **Playwright Stealth is the best solution** (100% success on Discuss.com.hk)\n2. **Don't use Crawlee** (easily detected)\n3. **Chaser (Rust) doesn't work currently** (blocked by Cloudflare)\n4. **Hiding `navigator.webdriver` is key**\n\n---\n\n## Future Plans\n\n- [ ] Add proxy IP rotation\n- [ ] CAPTCHA handling integration\n- [ ] Cookie management (maintain login state)\n- [ ] Batch scraping (parallel processing)\n- [ ] Integration with OpenClaw browser tool"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Playwright-based web scraping OpenClaw Skill with anti-bot protection. Successfully tested on complex sites like Discuss.com.hk. Skill: 102 Playwright Scraper Skill Owner: smallkeyboy Summary: Playwright-based web scraping OpenClaw Skill with anti-bot protection. Successfully tested on complex sites like Discuss.com.hk. Tags: latest:1.0.0 Version history: v1.0.0 | 2026-04-16T07:09:40.650Z | auto **Playwright Scraper Skill v1.2.0:** - Added comprehensive usage guide and script descriptions, including detailed guidance for various website anti-b","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1183,"uniquenessScore":45,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T14:54:26.941Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T14:54:26.941Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T17:42:16.059Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}