{"id":"a3e951bb-12d4-41c2-bc4b-48f99e021579","entityType":"agent","slug":"clawhub-iichaner-boss-resume-crawler","name":"boss-resume-crawler","canonicalUrl":"https://www.xpersona.co/agent/clawhub-iichaner-boss-resume-crawler","canonicalPath":"/agent/clawhub-iichaner-boss-resume-crawler","generatedAt":"2026-10-10T17:37:45.987Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-10T14:22:16.495Z","emptyReason":null},"description":"从 Boss 直聘批量爬取职位详情（含 security_id、职位描述），支持 PUA 薪资解码和增量去重。 Skill: boss-resume-crawler Owner: iichaner Summary: 从 Boss 直聘批量爬取职位详情（含 security_id、职位描述），支持 PUA 薪资解码和增量去重。 Tags: latest:0.1.0 Version history: v0.1.0 | 2026-08-18T10:45:58.143Z | auto Initial release of boss-resume-crawler. - Supports batch crawling of Boss直聘 job details, including security_id and job description - Implements PUA salary decoding and incremental deduplication - Provides strict dependency checks (Pyth","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.4K downloads reported by the source. Last updated 10/10/2026.","installCommand":"clawhub skill install s17evhyc3z1a82bvpm10c418p9849pg7:boss-resume-crawler","sourceUrl":"https://clawhub.ai/iichaner/boss-resume-crawler","homepage":"https://clawhub.ai/iichaner/skills/boss-resume-crawler","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/iichaner/boss-resume-crawler","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/iichaner/skills/boss-resume-crawler","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":63,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"从 Boss 直聘批量爬取职位详情（含 security_id、职位描述），支持 PUA 薪资解码和增量去重。 Skill: boss-resume-crawler Owner: iichaner Summary: 从 Boss 直聘批量爬取职位详情（含 security_id、职位描述），支持 PUA 薪资解码和增量"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-10T14:22:16.495Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T14:22:16.495Z","emptyReason":null},"stars":null,"forks":null,"downloads":1389,"packageName":null,"latestVersion":"0.1.0","tractionLabel":"1.4K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T14:22:16.495Z","emptyReason":null},"lastUpdatedAt":"2026-10-10T14:22:16.495Z","lastCrawledAt":"2026-10-10T14:22:16.495Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-11T14:22:16.495Z","lastVerifiedAt":null,"highlights":[{"version":"0.1.0","createdAt":"2026-08-18T10:45:58.143Z","changelog":"Initial release of boss-resume-crawler. - Supports batch crawling of Boss直聘 job details, including security_id and job description - Implements PUA salary decoding and incremental deduplication - Provides strict dependency checks (Python3, websocket-client, CDP connection via CloakBrowser) - Enforces manual login verification before crawling begins - Features resilient crawling logic: randomized delays, per-job CSV append, error logging and automatic retries - Includes detailed usage instructions, performance benchmarks, error handling, and anti-scraping precautions","fileCount":13,"zipByteSize":34418}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17evhyc3z1a82bvpm10c418p9849pg7:boss-resume-crawler","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-iichaner-boss-resume-crawler/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-iichaner-boss-resume-crawler/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-iichaner-boss-resume-crawler/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-iichaner-boss-resume-crawler/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-iichaner-boss-resume-crawler/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-iichaner-boss-resume-crawler/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T17:37:45.986Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-iichaner-boss-resume-crawler/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-iichaner-boss-resume-crawler/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-iichaner-boss-resume-crawler/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-iichaner-boss-resume-crawler/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-10T14:22:16.495Z","emptyReason":null},"readme":"Skill: boss-resume-crawler\n\nOwner: iichaner\n\nSummary: 从 Boss 直聘批量爬取职位详情（含 security_id、职位描述），支持 PUA 薪资解码和增量去重。\n\nTags: latest:0.1.0\n\nVersion history:\n\nv0.1.0 | 2026-08-18T10:45:58.143Z | auto\n\nInitial release of boss-resume-crawler.\n\n- Supports batch crawling of Boss直聘 job details, including security_id and job description\n- Implements PUA salary decoding and incremental deduplication\n- Provides strict dependency checks (Python3, websocket-client, CDP connection via CloakBrowser)\n- Enforces manual login verification before crawling begins\n- Features resilient crawling logic: randomized delays, per-job CSV append, error logging and automatic retries\n- Includes detailed usage instructions, performance benchmarks, error handling, and anti-scraping precautions\n\nArchive index:\n\nArchive v0.1.0: 13 files, 34418 bytes\n\nFiles: _meta.json (138b), LICENSE (1056b), README.md (7189b), references (0b), references/data-spec.md (2201b), references/error-handling.md (1855b), references/sop.md (9652b), scripts (0b), scripts/boss_extract_cdp.py (17822b), scripts/boss_extract_final.py (18224b), scripts/boss_extract_pure.py (15587b), skill-card.md (2531b), SKILL.md (9509b)\n\nFile v0.1.0:SKILL.md\n\n---\nname: boss-resume-crawler\ndescription: \"从 Boss 直聘批量爬取职位详情（含 security_id、职位描述），支持 PUA 薪资解码和增量去重。\"\nmetadata:\n  {\n    \"openclaw\":\n      {\n        \"emoji\": \"🕷️\",\n        \"requires\": { \"bins\": [\"curl\", \"python3\"] },\n      },\n  }\nread_when:\n  - 用户要求爬取 Boss 直聘职位\n  - 用户提到\"爬取\"、\"抓取\"、\"JD 数据\"、\"Boss 直聘\"\n  - 用户要求批量获取职位详情或 security_id\nallowed-tools: Bash,Read,Write,exec\n---\n\n# Boss直聘职位爬取\n\n## 快速验证（环境 OK 后立即跑通）\n\n```bash\n# 一条命令验证：爬取 5 条职位，输出到 /tmp\npython3 scripts/boss_extract_cdp.py --max-scroll 3 --max-jobs 5 --output /tmp/boss_test\n```\n\n验证通过 → 正式爬取。失败 → 检下方依赖和 CDP 连接。\n\n---\n\n## 性能基准（实测）\n\n| 指标 | 数值 |\n|------|------|\n| 单页滚动加载 | ~2 秒/次 |\n| 详情页提取 | ~22 秒/条（含 20 秒等待 + 随机波动） |\n| 100 条职位总耗时 | ~40 分钟 |\n| 500 条职位总耗时 | ~3 小时 |\n\n> 详情页耗时主要由等待时间决定（20 秒/条），这是 Boss 直聘客户端渲染的硬限制。\n\n---\n\n## 首次使用：依赖检查（必须）\n\n**在执行任何爬取操作之前，必须先检查以下依赖是否就绪。缺失时提示用户安装。**\n\n### 检查脚本\n\n```bash\n# 1. Python3\npython3 --version 2>/dev/null || echo \"❌ 未安装 Python3 → https://www.python.org/downloads/\"\n\n# 2. websocket-client（Python 库）\npython3 -c \"import websocket\" 2>/dev/null || echo \"❌ 缺少 websocket-client → pip3 install websocket-client\"\n\n# 3. CDP 连接（CloakBrowser 是否启动）\ncurl -s http://localhost:9222/json >/dev/null 2>&1 || echo \"❌ CDP 未连接，请先启动 CloakBrowser（见下方说明）\"\n```\n\n### CloakBrowser 启动方法\n\nCloakBrowser 是一个反检测 Chromium 浏览器，用于绕过 Boss 直聘的自动化检测。\n\n**安装：**\n```bash\n# 安装依赖\nnpm install cloakbrowser playwright-core\n\n# 下载 Chromium（需要代理）\nexport https_proxy=http://127.0.0.1:7890  # 根据你的代理配置\ncurl -L --max-time 600 -o /tmp/cloakbrowser-darwin-x64.tar.gz <下载链接>\ntar -xzf /tmp/cloakbrowser-darwin-x64.tar.gz -C ~/.cache/cloakbrowser/\n\n# 设置环境变量\nexport CLOAKBROWSER_BINARY_PATH=~/.cache/cloakbrowser/Chromium.app/Contents/MacOS/Chromium\n```\n\n> CloakBrowser 项目地址：https://github.com/nickspaargaren/cloakbrowser\n\n**启动（有头模式，必须）：**\n```bash\nopen ~/.cache/cloakbrowser/Chromium.app --args \\\n  --remote-debugging-port=9222 \\\n  \"--remote-allow-origins=*\" \\\n  --user-data-dir=<你的浏览器数据目录> \\\n  \"<Boss直聘列表页URL>\"\n```\n\n> ⚠️ Boss 直聘会检测 headless 模式，**必须使用有头模式**（能看到浏览器窗口）。\n\n### 依赖就绪标志\n\n所有 ✅ 后方可执行爬取：\n- [ ] Python3 可用\n- [ ] websocket-client 已安装\n- [ ] CDP 连接正常（`curl -s http://localhost:9222/json` 返回页面列表）\n- [ ] 用户已登录 Boss 直聘（见下方登录检查）\n\n---\n\n## 输入要求\n\n- **必须提供** Boss 直聘列表页 URL（含 `zhipin.com/web/geek/jobs`）\n- 未提供 URL 时必须主动询问，不要自行构造\n\n## 登录状态检查（必须在 Phase 1 之前执行）\n\n打开列表页后，**首先检查登录状态**，未登录则暂停等待人类操作：\n\n```bash\nsnapshot=$(agent-browser --cdp 9222 snapshot -i --timeout 8000 2>/dev/null)\nif echo \"$snapshot\" | grep -qE \"登录/注册|立即登录|登录\"; then\n  echo \"⚠️ 未登录状态，请手动扫码登录\"\n  echo \"登录完成后告知我，我再继续\"\nfi\n```\n\n**判断逻辑：**\n- ❌ 出现「登录/注册」「立即登录」「我要找工作」等链接 → 未登录\n- ✅ 出现用户名或用户头像链接 → 已登录\n\n**未登录时的处理：**\n1. 暂停所有爬取工作\n2. 提示用户：「页面显示未登录，请在浏览器中扫码登录，完成后告知我」\n3. 等待用户明确说「已登录」或「继续」后，再执行后续 Phase\n\n---\n\n## 执行流程\n\n### Phase 1：列表页滚动加载\n使用 CDP `Input.dispatchMouseEvent mouseWheel` 模拟真实鼠标滚轮（agent-browser scroll 无效）。最多 100 次滚动，随机等待 1.5-3.5 秒，连续 3 次数量不变则停止。启动浏览器需添加 `\"--remote-allow-origins=*\"` 参数。\n详见 [references/sop.md](references/sop.md) SOP-1。\n\n### Phase 2：职位列表提取\n通过 CDP 执行 JavaScript 从 `.job-card-wrap` 提取职位基础信息，薪资需 PUA 解码（0xe031→0, ..., 0xe03a→9）。选择器和字段规格见 [references/data-spec.md](references/data-spec.md)。\n\n### Phase 3：详情页逐条爬取\nCDP 开新 tab → **等待 20 秒 + 随机波动（0-3 秒）** → 检查 `readyState` → 提取 security_id + 职位描述 → **立即关闭 tab**。\n\n> **为什么必须等 20 秒？** Boss 直聘详情页使用客户端渲染（React/Vue），CDP 打开新 tab 后需要等待\n> JS 框架完成 hydration + API 请求返回数据。实测 <15 秒约 40% 概率拿到空数据，<10 秒几乎必定为空。\n\n**为什么每次都要关闭 tab？** 不关闭会导致 tab 堆积，占用 CDP 连接资源，后续操作超时。\n\n详见 [references/sop.md](references/sop.md) SOP-2、SOP-4。\n\n### Phase 4：CSV 增量存储\n**每提取 1 条立即追加写入 CSV**（不缓存到内存）。追加模式（`'a'`）+ 跨文件 job_id 去重。**禁止使用 `'w'` 覆盖模式。**\n详见 [references/sop.md](references/sop.md) SOP-3。\n\n### Phase 5：质量报告\n输出本次新增/去重跳过/累计总量/字段完整率/错误详情。格式见下方「输出规范」。\n\n## 数据校验标准\n\n| 字段 | 校验 |\n|------|------|\n| job_id | 非空，>=20 字符 |\n| security_id | 非空，>=30 字符 |\n| 薪资 | 包含 \"K\"（实习岗日薪除外） |\n| 职位描述 | 非空，>=100 字符 |\n| 公司名称 | 非空，>=2 字符 |\n\n## 输出规范\n\n- CSV 路径：`<工作目录>/jobs_data_{YYYYMMDD}_{HHMM}.csv`\n- 错误日志：同目录下 `temp/error_log.csv`\n- 质量报告格式：\n  - 本次新增：N 条\n  - 去重跳过：N 条\n  - 累计总量：N 条\n  - 字段完整率：job_id X% | security_id X% | 职位描述 X%\n  - 错误：N 条（详情见 error_log.csv）\n\n## 错误处理\n\n两层 Fallback：\n1. **第一层**：增加等待时间重试（+5 秒/次），同一位置最多 3 次\n2. **第二层**：跳过错误职位，记录到 `temp/error_log.csv`，继续下一个\n\n### 常见错误速查\n\n| 现象 | 原因 | 解决方案 |\n|------|------|---------|\n| WebSocket 连接超时 | CDP 长连接被断开 | 新脚本已修复（每次独立连接） |\n| security_id 为空 | 详情页等待时间不足 | 增加 `--base-wait 25` |\n| 职位描述全部为空 | 等待 <15 秒 | 确保等待 ≥20 秒 |\n| ACCOUNT_RISK | 被风控拦截 | 切换 CloakBrowser profile（用新 --user-data-dir） |\n| CDP 连接被拒 | 缺少 `--remote-allow-origins=*` | 启动时添加该参数 |\n| 滚动后只有 15 条 | agent-browser scroll 无效 | 脚本已使用 CDP `Input.dispatchMouseEvent mouseWheel` |\n| CSV 数据被覆盖 | 用了 `'w'` 模式 | 脚本已使用 `'a'` 追加模式 |\n| PUA 薪资乱码 | 未解码 | 脚本已内置 decode_pua 函数 |\n| 进程中断后数据丢失 | 旧脚本无批次缓存 | 新脚本每条立即写入 CSV |\n\n已知问题和修复方案详见 [references/error-handling.md](references/error-handling.md)。\n\n## 脚本\n\n| 脚本 | 用途 | 运行说明 |\n|------|------|---------|\n| `scripts/boss_extract_cdp.py` | **默认脚本**：纯 CDP 模式，反爬优化 | 依赖：websocket-client。推荐使用 |\n| `scripts/boss_extract_pure.py` | 旧版备用：纯 CDP 模式（无反爬优化） | 依赖：websocket-client。不推荐 |\n| `scripts/boss_extract_final.py` | agent-browser 模式 | 依赖：agent-browser, websocket-client |\n\n**默认脚本参数：**\n```bash\n# 指定输出目录（默认当前目录）\npython3 scripts/boss_extract_cdp.py --output ~/Desktop/jobs\n\n# 限制滚动次数（默认100）\npython3 scripts/boss_extract_cdp.py --max-scroll 30\n\n# 限制爬取条数（默认全部）\npython3 scripts/boss_extract_cdp.py --max-jobs 10\n\n# 调整详情页等待秒数（默认20）\npython3 scripts/boss_extract_cdp.py --base-wait 25\n\n# 组合使用\npython3 scripts/boss_extract_cdp.py --output ~/Desktop/jobs --max-scroll 50 --max-jobs 20\n```\n\n**脚本设计原则（反爬）：**\n- 每次 CDP 操作新建 WebSocket 连接，用完即关（防超时）\n- 每提取 1 条立即追加写入 CSV（防数据丢失）\n- 详情页提取后立即关闭 tab（防 tab 堆积）\n- 所有等待时间加随机波动（反爬）\n- 失败自动重试，指数退避（容错）\n\n优先使用 Agent 自主执行（实时处理异常更灵活），职位数量大时运行脚本批量处理。\n\n## 反爬注意事项\n\n- **必须使用有头模式**（headless 会被拦截）\n- **必须使用 CloakBrowser**（反检测 Chromium）\n- **滚动必须使用 CDP `Input.dispatchMouseEvent mouseWheel`**（agent-browser scroll 无效）\n- 启动浏览器需添加 `\"--remote-allow-origins=*\"` 参数\n- 滚动间隔 1.5-3.5 秒（随机化），模拟人类阅读节奏\n- 详情页等待 20 秒 + 随机波动（0-3 秒）+ 重试退避（+5 秒/次）\n- 每次 CDP 操作独立连接，避免长连接被检测\n- 详情页提取后立即关闭 tab，避免大量 tab 堆积\n\nFile v0.1.0:README.md\n\n# Boss直聘职位爬取 Skill\n\n> 🕷️ 从 Boss 直聘批量提取职位详情（含 security_id、职位描述），支持 PUA 薪资解码和增量去重。\n\n一个 [OpenClaw](https://github.com/openclaw/openclaw) Skill，帮助 AI Agent 或用户自动化爬取 Boss 直聘的职位数据。\n\n---\n\n## 功能特性\n\n- 🔄 **智能滚动加载** — 使用 CDP 模拟真实鼠标滚轮，绕过 Boss 直聘反爬检测\n- 🔐 **PUA 薪资解码** — 自动将 Boss 直聘的特殊 Unicode 字符还原为真实薪资数字\n- 📋 **详情页深度提取** — 逐条打开详情页，提取 security_id 和完整职位描述\n- 💾 **即时写入存储** — 每提取 1 条立即写入 CSV，中断不丢数据\n- ✅ **质量报告** — 每次爬取后输出字段完整率和错误统计\n- 🛡️ **两层容错** — 失败自动重试（指数退避），仍失败则跳过并记录错误日志\n- 🕶️ **反爬优化** — 独立连接、随机等待、tab 即关，降低被检测风险\n\n## 前置依赖\n\n| 依赖 | 必需 | 说明 |\n|------|------|------|\n| Python 3.6+ | ✅ | 运行脚本 |\n| websocket-client | ✅ | Python CDP 通信库 |\n| CloakBrowser | ✅ | 反检测 Chromium 浏览器 |\n\n## 安装\n\n### 1. 克隆仓库\n\n```bash\ngit clone https://github.com/iichaner/boss-resume-crawler.git\ncd boss-resume-crawler\n```\n\n### 2. 安装 Python 依赖\n\n```bash\npip3 install websocket-client\n```\n\n### 3. 安装 CloakBrowser\n\n```bash\nnpm install cloakbrowser playwright-core\n\n# 下载 Chromium（需要代理访问 GitHub）\nexport https_proxy=http://127.0.0.1:7890\ncurl -L --max-time 600 -o /tmp/cloakbrowser-darwin-x64.tar.gz <下载链接>\ntar -xzf /tmp/cloakbrowser-darwin-x64.tar.gz -C ~/.cache/cloakbrowser/\n```\n\n> 📖 [CloakBrowser 文档](https://github.com/nickspaargaren/cloakbrowser)\n\n## 快速开始\n\n```bash\n# 1. 启动 CloakBrowser（有头模式，必须）\nopen ~/.cache/cloakbrowser/Chromium.app --args \\\n  --remote-debugging-port=9222 \\\n  \"--remote-allow-origins=*\" \\\n  --user-data-dir=/tmp/chrome-cdp-profile \\\n  \"https://www.zhipin.com/web/geek/jobs?query=总经理助理&city=101010100\"\n\n# 2. 在浏览器中扫码登录 Boss 直聘\n\n# 3. 快速验证（爬取 5 条）\npython3 scripts/boss_extract_cdp.py --max-scroll 3 --max-jobs 5 --output /tmp/boss_test\n\n# 4. 正式爬取\npython3 scripts/boss_extract_cdp.py --output ~/Desktop/jobs --max-scroll 50\n```\n\n## 脚本说明\n\n| 脚本 | 用途 | 推荐度 |\n|------|------|--------|\n| `scripts/boss_extract_cdp.py` | **默认脚本**：纯 CDP 模式，反爬优化 | ⭐⭐⭐ 推荐 |\n| `scripts/boss_extract_pure.py` | 旧版：纯 CDP 模式（无反爬优化） | ⭐ 备用 |\n| `scripts/boss_extract_final.py` | agent-browser 模式 | ⭐⭐ 可选 |\n\n### 默认脚本参数\n\n```bash\npython3 scripts/boss_extract_cdp.py [OPTIONS]\n\n--output DIR       输出目录（默认当前目录）\n--max-scroll N     最大滚动次数（默认 100）\n--max-jobs N       最大爬取条数（默认全部）\n--base-wait N      详情页基础等待秒数（默认 20）\n```\n\n### 示例\n\n```bash\n# 爬取全部职位，输出到桌面\npython3 scripts/boss_extract_cdp.py --output ~/Desktop/jobs\n\n# 只爬 20 条，滚动 30 次\npython3 scripts/boss_extract_cdp.py --max-jobs 20 --max-scroll 30\n\n# 网络较慢时增加等待\npython3 scripts/boss_extract_cdp.py --base-wait 25\n```\n\n## 反爬设计\n\n| 措施 | 实现 | 说明 |\n|------|------|------|\n| 有头模式 | CloakBrowser | headless 会被拦截 |\n| 反检测浏览器 | CloakBrowser | 绕过自动化检测 |\n| 真实滚轮模拟 | CDP `Input.dispatchMouseEvent` | 普通 scroll 不触发加载 |\n| 随机滚动间隔 | 1.5-3.5 秒随机 | 模拟人类阅读节奏 |\n| 随机详情页等待 | 20 秒 + 0-3 秒波动 | 降低请求规律性 |\n| 独立 WebSocket 连接 | 每次操作新建，用完即关 | 防长连接被检测 |\n| Tab 即关 | 提取后立即关闭 | 防 tab 堆积 |\n| 指数退避重试 | +5 秒/次，最多 3 次 | 避免频繁重试触发风控 |\n\n## 输出格式\n\n### CSV 字段\n\n| 字段名 | 说明 | 示例 |\n|--------|------|------|\n| 职位名称 | 职位标题 | 总经理助理 |\n| 薪资 | PUA 解码后的真实薪资 | 15-20K·13薪 |\n| 经验要求 | 工作经验 | 5-10年 |\n| 学历要求 | 最低学历 | 本科 |\n| 公司名称 | 招聘公司 | 某某科技有限公司 |\n| 城市 | 工作城市 | 上海 |\n| 区域 | 具体区域 | 浦东新区·张江 |\n| job_id | Boss 直聘职位 ID | abc123def456 |\n| security_id | 安全标识 | a1b2c3d4... |\n| 职位描述 | 完整岗位职责和任职要求 | 负责协助总经理... |\n| 创建日期 | 爬取时间 | 2026-06-01 20:50 |\n\n### 质量报告示例\n\n```\n==================================================\n爬取完成\n==================================================\n本次新增:     45 条\n本次失败:     2 条\n去重跳过:     3 条\n累计总量:     128 条\n输出文件:     /Users/ii/Desktop/jobs/jobs_data_20260601_2050.csv\n错误日志:     /Users/ii/Desktop/jobs/temp/error_log.csv\n```\n\n## 常见问题\n\n### Q: CDP 连接失败？\n\n```bash\ncurl -s http://localhost:9222/json\n```\n返回页面列表即正常。如果为空，检查 CloakBrowser 是否启动。\n\n### Q: 职位描述全部为空？\n\n等待时间不足。使用 `--base-wait 25` 增加等待。\n\n### Q: 滚动后职位数量不变？\n\n脚本已使用 CDP `Input.dispatchMouseEvent mouseWheel`，如果仍无效，可能是网络问题或页面未登录。\n\n### Q: 被风控拦截（ACCOUNT_RISK）？\n\n- 使用新的 `--user-data-dir` 启动 CloakBrowser\n- 减少 `--max-scroll` 次数\n- 不要同时开多个爬取进程\n\n### Q: CSV 打开乱码？\n\nCSV 使用 `utf-8-sig` 编码。Excel 打开时选择 UTF-8，或用 WPS/文本编辑器打开。\n\n## 项目结构\n\n```\nboss-resume-crawler/\n├── SKILL.md                          # OpenClaw Skill 入口\n├── README.md                         # 本文档\n├── LICENSE                           # MIT 开源协议\n├── references/\n│   ├── sop.md                        # 详细操作流程（SOP）\n│   ├── data-spec.md                  # 数据字段规格和选择器\n│   └── error-handling.md             # 错误处理方案\n└── scripts/\n    ├── boss_extract_cdp.py           # 默认脚本（反爬优化）\n    ├── boss_extract_pure.py          # 旧版备用脚本\n    └── boss_extract_final.py         # agent-browser 模式脚本\n```\n\n## 更新日志\n\n### v2.0.0 (2026-06-01)\n\n**反爬优化重写：**\n- ✨ 新增默认脚本 `boss_extract_cdp.py`（反爬优化版）\n- 🔧 每次 CDP 操作新建独立 WebSocket 连接（修复超时问题）\n- 🔧 每提取 1 条立即写入 CSV（修复中断丢数据问题）\n- 🔧 详情页提取后立即关闭 tab（修复 tab 堆积问题）\n- 🔧 所有等待时间加随机波动（增强反爬能力）\n- 🔧 失败重试改为指数退避（+5 秒/次）\n- 📝 更新 SKILL.md、SOP、错误处理文档\n\n### v1.0.0 (2026-05-28)\n\n- 🎉 初始版本\n- 支持列表页滚动加载、详情页提取、PUA 薪资解码\n- 支持 CSV 增量存储和跨文件去重\n\n## License\n\n[MIT](LICENSE)\n\nFile v0.1.0:_meta.json\n\n{\n  \"ownerId\": \"kn76dgfmrcmcfw91f0m92tj4rd84805k\",\n  \"slug\": \"boss-resume-crawler\",\n  \"version\": \"0.1.0\",\n  \"publishedAt\": 1787049958143\n}\n\nFile v0.1.0:references/data-spec.md\n\n# 数据字段规格\n\n## 必要字段（缺失即失败）\n\n| 字段名 | 类型 | 校验标准 | 提取来源 |\n|--------|------|---------|---------|\n| job_id | Text | 非空，>=20 字符 | 列表页 href 正则 `/job_detail/(.+?)\\.html` |\n| security_id | Text | 非空，>=30 字符 | 详情页 script 标签 |\n| 薪资 | Text | 包含 \"K\" | 列表页 `.job-salary`，需 PUA 解码 |\n| 职位描述 | Text | 非空，>=100 字符 | 详情页 body.innerText |\n| 公司名称 | Text | 非空，>=2 字符 | 列表页 `.boss-name` link text |\n\n## 普通字段\n\n| 字段名 | 提取来源 |\n|--------|---------|\n| 职位名称 | 列表页 `.job-name` link text |\n| 经验要求 | 列表页 `.tag-list li`，匹配 `\\d+-\\d+年` |\n| 学历要求 | 列表页 `.tag-list li`，匹配 `本科\\|大专\\|硕士\\|博士\\|学历不限` |\n| 城市 | 列表页 `.company-location`，按 `·` 拆分取第一段 |\n| 区域 | 列表页 `.company-location`，按 `·` 拆分取剩余 |\n| 招聘者 | 详情页（如有） |\n| 招聘者职位 | 详情页（如有） |\n| 创建日期 | 爬取时间，格式 `YYYY-MM-DD HH:MM` |\n\n## CSV 字段顺序\n\n```\n职位名称,薪资,经验要求,学历要求,公司名称,城市,区域,job_id,security_id,职位描述,创建日期\n```\n\n## PUA 薪资解码\n\nBoss 直聘使用 PUA Unicode 字符隐藏真实薪资数字：\n\n```\n0xe031 → 0    0xe036 → 5\n0xe032 → 1    0xe037 → 6\n0xe033 → 2    0xe038 → 7\n0xe034 → 3    0xe039 → 8\n0xe035 → 4    0xe03a → 9\n```\n\nPython 解码函数（已内嵌于脚本）：\n\n```python\nPUA_MAP = {\n    0xe031: '0', 0xe032: '1', 0xe033: '2', 0xe034: '3', 0xe035: '4',\n    0xe036: '5', 0xe037: '6', 0xe038: '7', 0xe039: '8', 0xe03a: '9'\n}\ndef decode_pua(text):\n    if not text: return text\n    return ''.join(PUA_MAP.get(ord(c), c) for c in text)\n```\n\n## 页面选择器（当前有效）\n\n| 元素 | 选择器 | 备注 |\n|------|--------|------|\n| 职位名称 | `.job-name` | ✅ |\n| 薪资 | `.job-salary` | ✅ 需 PUA 解码 |\n| 公司名称 | `.boss-name` | ✅ |\n| 地区 | `.company-location` | ✅ |\n| 经验/学历 | `.tag-list li` | ✅ |\n\n**已失效选择器**：`.salary`, `.job-title`, `.company-name a`, `.area`\n\nFile v0.1.0:references/error-handling.md\n\n# 错误处理\n\n## 两层 Fallback 机制\n\n### 第一层:重试\n\n- **触发**:页面加载慢、网络波动、反爬延迟\n- **策略**:增加 50% 等待时间,重试当前步骤\n- **上限**:同一位置最多重试 3 次,超过则升级到第二层\n\n### 第二层:修复并继续\n\n- **触发**:一层重试仍失败\n- **策略**:\n  1. 跳过当前错误职位\n  2. 记录错误到 `temp/error_log.csv`\n  3. 继续处理下一个职位\n\n### 错误日志格式\n\n```csv\ntitle,job_id,error,timestamp\n```\n\n---\n\n## 当前需注意的陷阱\n\n- **详情页等待必须 20 秒以上**:不足会导致职位描述全部为空(0 字节)\n- **滚动必须模拟人类节奏**:快速连续滚动会导致列表只加载部分(如 50 条只加载 15 条)\n- **CSV 必须用 `'a'` 追加模式**:`'w'` 模式会覆盖历史数据\n\n---\n\n## 已修复问题归档\n\n| # | 问题 | 修复方案 | 状态 |\n|---|------|---------|------|\n| 0 | CSV 覆盖历史数据 | 改用 `'a'` 追加 + job_id 去重 | ✅ |\n| 1 | 列表页上限 49 条 | 连续 5 次滚动不变则停止 | ✅ |\n| 2 | 职位描述提取不完整 | 从“职位描述”标题开始，多结束标记 | ✅ |\n| 3 | job_id 提取失败 | 改用 `.+?\\.html` 正则 | ✅ |\n| 4 | 公司名称解析错误 | 直接从 link 元素获取 | ✅ |\n| 5 | security_id 提取为空 | 从详情页 script 标签正则提取 | ✅ |\n| 6 | 职位描述全部为空 | 等待增至 20 秒 + readyState 检测 | ✅ |\n| 7 | 快速滚动加载不全 | 人类滚动策略：随机等待 + 停留 | ✅ |\n| 8 | WebSocket 长连接超时 | 每次 CDP 操作新建独立连接，用完即关 | ✅ |\n| 9 | 进程中断数据全丢 | 每提取 1 条立即追加写入 CSV | ✅ |\n| 10 | 详情页 tab 堆积 | 提取后立即 close_tab | ✅ |\n| 11 | 反爬检测（固定间隔） | 所有等待时间加随机波动 | ✅ |\n\nFile v0.1.0:references/sop.md\n\n# 详细 SOP\n\n## SOP-0：登录状态检查\n\n> 用于所有 Phase 之前。BOSS 直聘未登录时只能看到有限职位，且无法获取完整数据。\n\n### 执行步骤\n\n```bash\n# 1. 获取页面快照\nsnapshot=$(agent-browser --cdp 9222 snapshot -i --timeout 8000 2>/dev/null)\n\n# 2. 检查登录状态\nif echo \"$snapshot\" | grep -qE \"登录/注册|立即登录|我要找工作\"; then\n  echo \"⚠️ 未登录状态\"\n  echo \"请在 CloakBrowser 浏览器中手动扫码登录 BOSS 直聘\"\n  echo \"登录完成后请告知我，我再继续执行\"\n  # 暂停，等待用户确认\n  exit 0\nfi\n\n# 3. 确认已登录\nusername=$(echo \"$snapshot\" | grep -oE 'link \"[^\"]+\"' | head -20 | grep -vE \"首页|职位|公司|校园|海归|APP|消息|简历|推荐|搜索|地图\" | head -1)\\necho \"✅ 已登录: $username\"\n```\n\n### 判断标准\n\n| 状态 | 特征 | 处理 |\n|------|------|------|\n| ❌ 未登录 | snapshot 中出现「登录/注册」「立即登录」「我要找工作」 | 暂停，等待用户登录 |\n| ✅ 已登录 | snapshot 中出现用户名（如「陈新彦」）或「简历」「消息」链接 | 继续执行 |\n\n### 注意事项\n\n- 登录态保存在浏览器 profile 中，通常无需每次登录\n- 如果页面跳转到登录页，说明 session 过期，需重新登录\n- **绝对不要尝试自动登录**（扫码需人工操作）\n\n---\n\n## SOP-1：列表页滚动策略\n\n> 用于 Phase 1。BOSS 直聘反爬机制检测快速滚动，必须模拟人类鼠标滚轮。\n\n### 关键发现\n\n- ❌ `agent-browser scroll bottom` 无效（仅触发 scroll 事件，不加载新内容）\n- ✅ `CDP Input.dispatchMouseEvent mouseWheel` 有效（模拟真实鼠标滚轮）\n\n### 执行步骤（Python 脚本）\n\n```python\nimport json\nimport websocket\nimport time\nimport subprocess\n\ndef get_ws_url():\n    result = subprocess.run(['curl', '-s', 'http://localhost:9222/json'], capture_output=True, text=True, timeout=5)\n    pages = json.loads(result.stdout)\n    for p in pages:\n        if p.get('type') == 'page' and 'zhipin' in p.get('url', ''):\n            return p['webSocketDebuggerUrl']\n    return None\n\ndef cdp_scroll(ws_url, delta_y=800):\n    \"\"\"使用 CDP Input.dispatchMouseEvent 模拟鼠标滚轮\"\"\"\n    ws = websocket.create_connection(ws_url, timeout=10)\n    command = {\n        \"id\": 1,\n        \"method\": \"Input.dispatchMouseEvent\",\n        \"params\": {\n            \"type\": \"mouseWheel\",\n            \"x\": 500,\n            \"y\": 400,\n            \"deltaX\": 0,\n            \"deltaY\": delta_y\n        }\n    }\n    ws.send(json.dumps(command))\n    ws.recv()\n    ws.close()\n\ndef count_jobs(ws_url):\n    \"\"\"统计当前页面职位数\"\"\"\n    ws = websocket.create_connection(ws_url, timeout=10)\n    command = {\n        \"id\": 1,\n        \"method\": \"Runtime.evaluate\",\n        \"params\": {\n            \"expression\": \"document.querySelectorAll('[class*=job-card]').length\",\n            \"returnByValue\": True\n        }\n    }\n    ws.send(json.dumps(command))\n    response = ws.recv()\n    result = json.loads(response)\n    ws.close()\n    if 'result' in result and 'result' in result['result']:\n        return result['result']['result'].get('value', 0)\n    return 0\n\n# 主流程\nws_url = get_ws_url()\n\n# 回到顶部\nws = websocket.create_connection(ws_url, timeout=10)\nws.send(json.dumps({\"id\": 1, \"method\": \"Runtime.evaluate\", \"params\": {\"expression\": \"window.scrollTo(0, 0)\", \"returnByValue\": True}}))\nws.recv()\nws.close()\ntime.sleep(1)\n\n# 滚动加载\nprev_count = 0\nsame_count = 0\n\nfor i in range(100):  # 最多滚动100次\n    cdp_scroll(ws_url, 800)\n    time.sleep(1.5 + (i % 3) * 0.5)  # 随机等待1.5-3秒\n    \n    if (i + 1) % 5 == 0:\n        count = count_jobs(ws_url)\n        print(f\"第 {i+1} 次滚动: {count} 条职位\")\n        \n        if count == prev_count:\n            same_count += 1\n            if same_count >= 3:\n                print(\"连续3次数量不变，停止\")\n                break\n        else:\n            same_count = 0\n        prev_count = count\n```\n\n### 启动浏览器要求\n\n必须添加 `--remote-allow-origins=*` 参数，否则 CDP WebSocket 连接会被拒绝：\n\n```bash\nopen ~/.cache/cloakbrowser/Chromium.app --args \\\n  --remote-debugging-port=9222 \\\n  \"--remote-allow-origins=*\" \\\n  --user-data-dir=<你的浏览器数据目录> \\\n  \"<列表页URL>\"\n```\n\n### 验证加载\n\n```bash\n# 使用 agent-browser snapshot 统计\nagent-browser --cdp 9222 snapshot -i --timeout 8000 2>/dev/null | grep -cE \"listitem.*K.*(年|经验)\"\n```\n\n也可以用纯 JS 统计：\n```bash\ncurl -s http://localhost:9222/json | python3 -c \"import json,sys; pages=json.load(sys.stdin); [print(p['webSocketDebuggerUrl']) for p in pages if 'jobs' in p.get('url','')]\"\n```\n\n### 上限检测\n\n- 连续 3 次滚动后数量不变 → 判定到达上限\n- 记录最后一条职位（公司名 + 岗位名），供用户人工校验\n\n### 测试结果\n\n| 滚动次数 | 职位数 |\n|----------|--------|\n| 0（初始） | 15 |\n| 10 | 90 |\n| 20 | 405 |\n| 30 | 540 |\n| 50 | 810 |\n\n---\n\n## SOP-2：详情页数据提取\n\n> 用于 Phase 3。逐条打开详情页，提取 security_id 和职位描述。\n\n### 打开详情页\n\n```python\nimport urllib.request\n\ndef open_new_tab(url):\n    req = urllib.request.Request(f'http://localhost:9222/json/new?{url}', method='PUT')\n    with urllib.request.urlopen(req, timeout=10) as response:\n        return json.loads(response.read().decode('utf-8')).get('id')\n```\n\n### 等待与校验\n\n```python\ntime.sleep(20)  # 强制等待 20 秒（关键！）\n\n# 检测页面加载完成\nready_state = cdp_execute(ws_url, \"document.readyState\")\nif ready_state != 'complete':\n    close_tab(page_id)\n    continue  # 跳过当前职位\n```\n\n### 提取 security_id\n\n```javascript\n(function() {\n    var scripts = document.querySelectorAll('script');\n    for (var i = 0; i < scripts.length; i++) {\n        var text = scripts[i].innerText || scripts[i].textContent || '';\n        // 匹配 securityId:'xxx' 或 securityId:\"xxx\" 或 securityId: 'xxx'\n        var match = text.match(/securityId['\":\\s]+['\"]([^'\"]+)['\"]/);\n        if (match) return match[1];\n    }\n    return '';\n})()\n```\n\n### 提取职位描述\n\n```javascript\n(function() {\n    var text = document.body.innerText;\n    var start = text.indexOf('职位描述');\n    if (start == -1) start = text.indexOf('岗位职责');\n    if (start == -1) return '';\n\n    var descStart = start + '职位描述'.length;\n    var endMarkers = ['刚刚活跃', '工作地址', '查看更多信息', '在线状态', '投诉举报', '相似职位'];\n    var end = text.length;\n    for (var i = 0; i < endMarkers.length; i++) {\n        var pos = text.indexOf(endMarkers[i], start);\n        if (pos != -1 && pos < end) end = pos;\n    }\n    return text.substring(descStart, end).trim();\n})()\n```\n\n### 关闭详情页（必须）\n\n**提取完成后立即关闭 tab，不关闭会导致 tab 堆积，占用 CDP 连接资源。**\n\n```python\nsubprocess.run(['curl', '-s', f'http://localhost:9222/json/close/{page_id}'],\n              capture_output=True, timeout=5)\ntime.sleep(1)  # 等待 tab 完全关闭\n```\n\n> 新脚本 `boss_extract_cdp.py` 在 `finally` 块中确保关闭，即使提取失败也会关闭。\n\n### 数据校验\n\n提取后立即校验，不合格则跳过：\n- `security_id` 长度 < 30 → 跳过\n- 职位描述长度 < 100 → 警告但保留\n\n---\n\n## SOP-3：CSV 增量存储\n\n> 用于 Phase 4。每次爬取生成新文件，增量追加，绝不覆盖。\n\n### 文件命名\n\n```\njobs_data_{YYYYMMDD}_{HHMM}.csv\n```\n\n路径：`<输出目录>/`（由 `--output` 参数指定，默认当前目录）\n\n### 去重逻辑\n\n```python\nimport glob\n\nexisting_job_ids = set()\nfor f in glob.glob('<输出目录>/jobs_data*.csv'):  # 替换为实际输出目录\n    with open(f, 'r', encoding='utf-8-sig') as fh:\n        for row in csv.DictReader(fh):\n            if row.get('job_id'):\n                existing_job_ids.add(row['job_id'])\n\nnew_jobs = [j for j in all_jobs if j['job_id'] not in existing_job_ids]\n```\n\n### 追加写入\n\n```python\nwith open(output_file, 'a', newline='', encoding='utf-8-sig') as f:\n    writer = csv.DictWriter(f, fieldnames=fieldnames)\n    if not file_exists:\n        writer.writeheader()\n    for job in new_jobs:\n        writer.writerow(job)\n```\n\n**⚠️ 禁止使用 `'w'` 模式！必须用 `'a'` 追加模式。**\n\n### 即时写入（新策略）\n\n**每提取 1 条立即追加写入主 CSV 文件**，不再使用批次缓存。\n\n优势：\n- 进程中断最多丢失 1 条数据（当前正在提取的那条）\n- 无需手动合并批次文件\n- 实时可见爬取进度（CSV 行数 = 已完成条数）\n\n---\n\n## SOP-4：CDP 工具函数\n\n> 供详情页提取使用的核心工具函数。\n\n### cdp_execute\n\n```python\nimport websocket, json\n\ndef cdp_execute(ws_url, js_code, timeout=15):\n    ws = websocket.create_connection(ws_url, timeout=timeout)\n    command = {\n        \"id\": 1,\n        \"method\": \"Runtime.evaluate\",\n        \"params\": {\"expression\": js_code, \"returnByValue\": True}\n    }\n    ws.send(json.dumps(command))\n    response = ws.recv()\n    result = json.loads(response)\n    ws.close()\n    if 'result' in result and 'result' in result['result']:\n        return result['result']['result'].get('value', '')\n    return None\n```\n\n### 获取页面 ID\n\n```python\ndef get_list_page_id():\n    result = subprocess.run(['curl', '-s', 'http://localhost:9222/json'],\n                          capture_output=True, text=True, timeout=5)\n    pages = json.loads(result.stdout)\n    for p in pages:\n        if 'jobs?' in p.get('url', ''):\n            return p['id'], p['webSocketDebuggerUrl']\n    return None, None\n```\n\nFile v0.1.0:skill-card.md\n\n## Description:\n\nBatch crawls Boss Zhipin job details, including security_id and job descriptions, with PUA salary decoding and incremental deduplication.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[iichaner](https://clawhub.ai/user/iichaner)\n\n### License/Terms of Use:\n\nMIT\n\n## Use Case:\n\nDevelopers and agents use this skill to collect Boss Zhipin job listing data through a browser CDP session, producing job-detail CSV records with security IDs, descriptions, decoded salaries, deduplication, and run-quality reporting.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill controls an authenticated browser through CDP and can interact with logged-in Boss Zhipin pages.\n\nMitigation: Run it only with a dedicated temporary browser profile and close the CDP browser when finished.\n\nRisk: Scraping Boss Zhipin data may violate platform terms or local rules for a given use case.\n\nMitigation: Confirm that collection and downstream use of Boss Zhipin data is permitted before running the crawler.\n\nRisk: Browser binaries and Python dependencies are installed outside the skill package.\n\nMitigation: Verify browser binaries and dependency versions before execution, preferably in an isolated environment.\n\n## Reference(s):\n\n- [Server-resolved source repository](https://github.com/iichaner/boss-resume-crawler)\n- [ClawHub skill page](https://clawhub.ai/iichaner/skills/boss-resume-crawler)\n- [Data Field Specification](references/data-spec.md)\n- [Error Handling](references/error-handling.md)\n- [Detailed SOP](references/sop.md)\n- [OpenClaw](https://github.com/openclaw/openclaw)\n- [CloakBrowser](https://github.com/nickspaargaren/cloakbrowser)\n- [Python downloads](https://www.python.org/downloads/)\n\n## Skill Output:\n\n**Output Type(s):** [Guidance, Shell commands, Code, Configuration, Files]\n\n**Output Format:** [Markdown guidance with bash commands; runtime scripts produce UTF-8 CSV files, error-log CSV files, and console quality reports.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Requires Python 3, curl, websocket-client, and an authenticated headed CloakBrowser CDP session on localhost:9222.]\n\n## Skill Version(s):\n\n0.1.0 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v0.1.0:LICENSE\n\nMIT License\n\nCopyright (c) 2026\n\nPermission is hereby granted, free of charge, to any person obtaining a copy\nof this software and associated documentation files (the \"Software\"), to deal\nin the Software without restriction, including without limitation the rights\nto use, copy, modify, merge, publish, distribute, sublicense, and/or sell\ncopies of the Software, and to permit persons to whom the Software is\nfurnished to do so, subject to the following conditions:\n\nThe above copyright notice and this permission notice shall be included in all\ncopies or substantial portions of the Software.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR\nIMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,\nFITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE\nAUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER\nLIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,\nOUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE\nSOFTWARE.","readmeExcerpt":"Skill: boss-resume-crawler Owner: iichaner Summary: 从 Boss 直聘批量爬取职位详情（含 security_id、职位描述），支持 PUA 薪资解码和增量去重。 Tags: latest:0.1.0 Version history: v0.1.0 | 2026-08-18T10:45:58.143Z | auto Initial release of boss-resume-crawler. - Supports batch crawling of Boss直聘 job details, including security_id and job description - Implements PUA salary decoding and incremental deduplication - Provides strict dependency checks (Pyth","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"# 一条命令验证：爬取 5 条职位，输出到 /tmp\npython3 scripts/boss_extract_cdp.py --max-scroll 3 --max-jobs 5 --output /tmp/boss_test"},{"language":"bash","snippet":"curl -s http://localhost:9222/json >/dev/null 2>&1 || echo \"❌ CDP 未连接，请先启动 CloakBrowser（见下方说明）\""},{"language":"bash","snippet":"# 1. Python3\npython3 --version 2>/dev/null || echo \"❌ 未安装 Python3 → https://www.python.org/downloads/\"\n\n# 2. websocket-client（Python 库）\npython3 -c \"import websocket\" 2>/dev/null || echo \"❌ 缺少 websocket-client → pip3 install websocket-client\"\n\n# 3. CDP 连接（CloakBrowser 是否启动）\ncurl -s http://localhost:9222/json >/dev/null 2>&1 || echo \"❌ CDP 未连接，请先启动 CloakBrowser（见下方说明）\""},{"language":"bash","snippet":"curl -L --max-time 600 -o /tmp/cloakbrowser-darwin-x64.tar.gz <下载链接>"},{"language":"bash","snippet":"# 安装依赖\nnpm install cloakbrowser playwright-core\n\n# 下载 Chromium（需要代理）\nexport https_proxy=http://127.0.0.1:7890  # 根据你的代理配置\ncurl -L --max-time 600 -o /tmp/cloakbrowser-darwin-x64.tar.gz <下载链接>\ntar -xzf /tmp/cloakbrowser-darwin-x64.tar.gz -C ~/.cache/cloakbrowser/\n\n# 设置环境变量\nexport CLOAKBROWSER_BINARY_PATH=~/.cache/cloakbrowser/Chromium.app/Contents/MacOS/Chromium"},{"language":"bash","snippet":"open ~/.cache/cloakbrowser/Chromium.app --args \\\n  --remote-debugging-port=9222 \\\n  \"--remote-allow-origins=*\" \\\n  --user-data-dir=<你的浏览器数据目录> \\\n  \"<Boss直聘列表页URL>\""}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: boss-resume-crawler\ndescription: \"从 Boss 直聘批量爬取职位详情（含 security_id、职位描述），支持 PUA 薪资解码和增量去重。\"\nmetadata:\n  {\n    \"openclaw\":\n      {\n        \"emoji\": \"🕷️\",\n        \"requires\": { \"bins\": [\"curl\", \"python3\"] },\n      },\n  }\nread_when:\n  - 用户要求爬取 Boss 直聘职位\n  - 用户提到\"爬取\"、\"抓取\"、\"JD 数据\"、\"Boss 直聘\"\n  - 用户要求批量获取职位详情或 security_id\nallowed-tools: Bash,Read,Write,exec\n---\n\n# Boss直聘职位爬取\n\n## 快速验证（环境 OK 后立即跑通）\n\n```bash\n# 一条命令验证：爬取 5 条职位，输出到 /tmp\npython3 scripts/boss_extract_cdp.py --max-scroll 3 --max-jobs 5 --output /tmp/boss_test\n```\n\n验证通过 → 正式爬取。失败 → 检下方依赖和 CDP 连接。\n\n---\n\n## 性能基准（实测）\n\n| 指标 | 数值 |\n|------|------|\n| 单页滚动加载 | ~2 秒/次 |\n| 详情页提取 | ~22 秒/条（含 20 秒等待 + 随机波动） |\n| 100 条职位总耗时 | ~40 分钟 |\n| 500 条职位总耗时 | ~3 小时 |\n\n> 详情页耗时主要由等待时间决定（20 秒/条），这是 Boss 直聘客户端渲染的硬限制。\n\n---\n\n## 首次使用：依赖检查（必须）\n\n**在执行任何爬取操作之前，必须先检查以下依赖是否就绪。缺失时提示用户安装。**\n\n### 检查脚本\n\n```bash\n# 1. Python3\npython3 --version 2>/dev/null || echo \"❌ 未安装 Python3 → https://www.python.org/downloads/\"\n\n# 2. websocket-client（Python 库）\npython3 -c \"import websocket\" 2>/dev/null || echo \"❌ 缺少 websocket-client → pip3 install websocket-client\"\n\n# 3. CDP 连接（CloakBrowser 是否启动）\ncurl -s http://localhost:9222/json >/dev/null 2>&1 || echo \"❌ CDP 未连接，请先启动 CloakBrowser（见下方说明）\"\n```\n\n### CloakBrowser 启动方法\n\nCloakBrowser 是一个反检测 Chromium 浏览器，用于绕过 Boss 直聘的自动化检测。\n\n**安装：**\n```bash\n# 安装依赖\nnpm install cloakbrowser playwright-core\n\n# 下载 Chromium（需要代理）\nexport https_proxy=http://127.0.0.1:7890  # 根据你的代理配置\ncurl -L --max-time 600 -o /tmp/cloakbrowser-darwin-x64.tar.gz <下载链接>\ntar -xzf /tmp/cloakbrowser-darwin-x64.tar.gz -C ~/.cache/cloakbrowser/\n\n# 设置环境变量\nexport CLOAKBROWSER_BINARY_PATH=~/.cache/cloakbrowser/Chromium.app/Contents/MacOS/Chromium\n```\n\n> CloakBrowser 项目地址：https://github.com/nickspaargaren/cloakbrowser\n\n**启动（有头模式，必须）：**\n```bash\nopen ~/.cache/cloakbrowser/Chromium.app --args \\\n  --remote-debugging-port=9222 \\\n  \"--remote-allow-origins=*\" \\\n  --user-data-dir=<你的浏览器数据目录> \\\n  \"<Boss直聘列表页URL>\"\n```\n\n> ⚠️ Boss 直聘会检测 headless 模式，**必须使用有头模式**（能看到浏览器窗口）。\n\n### 依赖就绪标志\n\n所有 ✅ 后方可执行爬取：\n- [ ] Python3 可用\n- [ ] websocket-client 已安装\n- [ ] CDP 连接正常（`curl -s http://localhost:9222/json` 返回页面列表）\n- [ ] 用户已登录 Boss 直聘（见下方登录检查）\n\n---\n\n## 输入要求\n\n- **必须提供** Boss 直聘列表页 URL（含 `zhipin.com/web/geek/jobs`）\n- 未提供 URL 时必须主动询问，不要自行构造\n\n## 登录状态检查（必须在 Phase 1 之前执行）\n\n打开列表页后，**首先检查登录状态**，未登录则暂停等待人类操作：\n\n```bash\nsnapshot=$(agent-browser --cdp 9222 snapshot -i --timeout 8000 2>/dev/null)\nif echo \"$snapshot\" | grep -qE \"登录/注册|立即登录|登录\"; then\n  echo \"⚠️ 未登录状态，请手动扫码登录\"\n  echo \"登录完成后告知我，我再继续\"\nfi\n```\n\n**判断逻辑：**\n- ❌ 出现「登录/注册」「立即登录」「我要找工作」等链接 → 未登录\n- ✅ 出现用户名或用户头像链接 → 已登录\n\n**未登录时的处理：**\n1. 暂停所有爬取工作\n2. 提示用户：「页面显示未登录，请在浏览器中扫码登录，完成后告知我」\n3. 等待用户明确说「已登录」或「继续」后，再执行后续 Phase\n\n---\n\n## 执行流程\n\n### Phase 1：列表页滚动加载\n使用 CDP `Input.dispatchMouseEvent mouseWheel` 模拟真实鼠标滚轮（agent-browser scroll 无效）。最多 100 次滚动，随机等待 1.5-3.5 秒，连续 3 次数量不变则停止。启动浏览器需添加 `\"--remote-allow-origins=*\"` 参数。\n详见 [references/sop.md](references/sop.md) SOP-1。\n\n### Phase 2：职位列表提取\n通过 CDP 执行 JavaScript 从 `.job-card-wrap` 提取职位基础信息，薪资需 PUA 解码"},{"path":"README.md","content":"# Boss直聘职位爬取 Skill\n\n> 🕷️ 从 Boss 直聘批量提取职位详情（含 security_id、职位描述），支持 PUA 薪资解码和增量去重。\n\n一个 [OpenClaw](https://github.com/openclaw/openclaw) Skill，帮助 AI Agent 或用户自动化爬取 Boss 直聘的职位数据。\n\n---\n\n## 功能特性\n\n- 🔄 **智能滚动加载** — 使用 CDP 模拟真实鼠标滚轮，绕过 Boss 直聘反爬检测\n- 🔐 **PUA 薪资解码** — 自动将 Boss 直聘的特殊 Unicode 字符还原为真实薪资数字\n- 📋 **详情页深度提取** — 逐条打开详情页，提取 security_id 和完整职位描述\n- 💾 **即时写入存储** — 每提取 1 条立即写入 CSV，中断不丢数据\n- ✅ **质量报告** — 每次爬取后输出字段完整率和错误统计\n- 🛡️ **两层容错** — 失败自动重试（指数退避），仍失败则跳过并记录错误日志\n- 🕶️ **反爬优化** — 独立连接、随机等待、tab 即关，降低被检测风险\n\n## 前置依赖\n\n| 依赖 | 必需 | 说明 |\n|------|------|------|\n| Python 3.6+ | ✅ | 运行脚本 |\n| websocket-client | ✅ | Python CDP 通信库 |\n| CloakBrowser | ✅ | 反检测 Chromium 浏览器 |\n\n## 安装\n\n### 1. 克隆仓库\n\n```bash\ngit clone https://github.com/iichaner/boss-resume-crawler.git\ncd boss-resume-crawler\n```\n\n### 2. 安装 Python 依赖\n\n```bash\npip3 install websocket-client\n```\n\n### 3. 安装 CloakBrowser\n\n```bash\nnpm install cloakbrowser playwright-core\n\n# 下载 Chromium（需要代理访问 GitHub）\nexport https_proxy=http://127.0.0.1:7890\ncurl -L --max-time 600 -o /tmp/cloakbrowser-darwin-x64.tar.gz <下载链接>\ntar -xzf /tmp/cloakbrowser-darwin-x64.tar.gz -C ~/.cache/cloakbrowser/\n```\n\n> 📖 [CloakBrowser 文档](https://github.com/nickspaargaren/cloakbrowser)\n\n## 快速开始\n\n```bash\n# 1. 启动 CloakBrowser（有头模式，必须）\nopen ~/.cache/cloakbrowser/Chromium.app --args \\\n  --remote-debugging-port=9222 \\\n  \"--remote-allow-origins=*\" \\\n  --user-data-dir=/tmp/chrome-cdp-profile \\\n  \"https://www.zhipin.com/web/geek/jobs?query=总经理助理&city=101010100\"\n\n# 2. 在浏览器中扫码登录 Boss 直聘\n\n# 3. 快速验证（爬取 5 条）\npython3 scripts/boss_extract_cdp.py --max-scroll 3 --max-jobs 5 --output /tmp/boss_test\n\n# 4. 正式爬取\npython3 scripts/boss_extract_cdp.py --output ~/Desktop/jobs --max-scroll 50\n```\n\n## 脚本说明\n\n| 脚本 | 用途 | 推荐度 |\n|------|------|--------|\n| `scripts/boss_extract_cdp.py` | **默认脚本**：纯 CDP 模式，反爬优化 | ⭐⭐⭐ 推荐 |\n| `scripts/boss_extract_pure.py` | 旧版：纯 CDP 模式（无反爬优化） | ⭐ 备用 |\n| `scripts/boss_extract_final.py` | agent-browser 模式 | ⭐⭐ 可选 |\n\n### 默认脚本参数\n\n```bash\npython3 scripts/boss_extract_cdp.py [OPTIONS]\n\n--output DIR       输出目录（默认当前目录）\n--max-scroll N     最大滚动次数（默认 100）\n--max-jobs N       最大爬取条数（默认全部）\n--base-wait N      详情页基础等待秒数（默认 20）\n```\n\n### 示例\n\n```bash\n# 爬取全部职位，输出到桌面\npython3 scripts/boss_extract_cdp.py --output ~/Desktop/jobs\n\n# 只爬 20 条，滚动 30 次\npython3 scripts/boss_extract_cdp.py --max-jobs 20 --max-scroll 30\n\n# 网络较慢时增加等待\npython3 scripts/boss_extract_cdp.py --base-wait 25\n```\n\n## 反爬设计\n\n| 措施 | 实现 | 说明 |\n|------|------|------|\n| 有头模式 | CloakBrowser | headless 会被拦截 |\n| 反检测浏览器 | CloakBrowser | 绕过自动化检测 |\n| 真实滚轮模拟 | CDP `Input.dispatchMouseEvent` | 普通 scroll 不触发加载 |\n| 随机滚动间隔 | 1.5-3.5 秒随机 | 模拟人类阅读节奏 |\n| 随机详情页等待 | 20 秒 + 0-3 秒波动 | 降低请求规律性 |\n| 独立 WebSocket 连接 | 每次操作新建，用完即关 | 防长连接被检测 |\n| Tab 即关 | 提取后立即关闭 | 防 tab 堆积 |\n| 指数退避重试 | +5 秒/次，最多 3 次 | 避免频繁重试触发风控 |\n\n## 输出格式\n\n### CSV 字段\n\n| 字段名 | 说明 | 示例 |\n|--------|------|------|\n| 职位名称 | 职位标题 | 总经理助理 |\n| 薪资 | PUA 解码后的真实薪资 | 15-20K·13薪 |\n| 经验要求 | 工作经验 | 5-10年 |\n| 学历要求 | 最低学历 | 本科 |\n| 公司名称 | 招聘公司 | 某某科技有限公司 |\n| 城市 | 工作城市 | 上海 |\n| 区域 |"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn76dgfmrcmcfw91f0m92tj4rd84805k\",\n  \"slug\": \"boss-resume-crawler\",\n  \"version\": \"0.1.0\",\n  \"publishedAt\": 1787049958143\n}"},{"path":"references/data-spec.md","content":"# 数据字段规格\n\n## 必要字段（缺失即失败）\n\n| 字段名 | 类型 | 校验标准 | 提取来源 |\n|--------|------|---------|---------|\n| job_id | Text | 非空，>=20 字符 | 列表页 href 正则 `/job_detail/(.+?)\\.html` |\n| security_id | Text | 非空，>=30 字符 | 详情页 script 标签 |\n| 薪资 | Text | 包含 \"K\" | 列表页 `.job-salary`，需 PUA 解码 |\n| 职位描述 | Text | 非空，>=100 字符 | 详情页 body.innerText |\n| 公司名称 | Text | 非空，>=2 字符 | 列表页 `.boss-name` link text |\n\n## 普通字段\n\n| 字段名 | 提取来源 |\n|--------|---------|\n| 职位名称 | 列表页 `.job-name` link text |\n| 经验要求 | 列表页 `.tag-list li`，匹配 `\\d+-\\d+年` |\n| 学历要求 | 列表页 `.tag-list li`，匹配 `本科\\|大专\\|硕士\\|博士\\|学历不限` |\n| 城市 | 列表页 `.company-location`，按 `·` 拆分取第一段 |\n| 区域 | 列表页 `.company-location`，按 `·` 拆分取剩余 |\n| 招聘者 | 详情页（如有） |\n| 招聘者职位 | 详情页（如有） |\n| 创建日期 | 爬取时间，格式 `YYYY-MM-DD HH:MM` |\n\n## CSV 字段顺序\n\n```\n职位名称,薪资,经验要求,学历要求,公司名称,城市,区域,job_id,security_id,职位描述,创建日期\n```\n\n## PUA 薪资解码\n\nBoss 直聘使用 PUA Unicode 字符隐藏真实薪资数字：\n\n```\n0xe031 → 0    0xe036 → 5\n0xe032 → 1    0xe037 → 6\n0xe033 → 2    0xe038 → 7\n0xe034 → 3    0xe039 → 8\n0xe035 → 4    0xe03a → 9\n```\n\nPython 解码函数（已内嵌于脚本）：\n\n```python\nPUA_MAP = {\n    0xe031: '0', 0xe032: '1', 0xe033: '2', 0xe034: '3', 0xe035: '4',\n    0xe036: '5', 0xe037: '6', 0xe038: '7', 0xe039: '8', 0xe03a: '9'\n}\ndef decode_pua(text):\n    if not text: return text\n    return ''.join(PUA_MAP.get(ord(c), c) for c in text)\n```\n\n## 页面选择器（当前有效）\n\n| 元素 | 选择器 | 备注 |\n|------|--------|------|\n| 职位名称 | `.job-name` | ✅ |\n| 薪资 | `.job-salary` | ✅ 需 PUA 解码 |\n| 公司名称 | `.boss-name` | ✅ |\n| 地区 | `.company-location` | ✅ |\n| 经验/学历 | `.tag-list li` | ✅ |\n\n**已失效选择器**：`.salary`, `.job-title`, `.company-name a`, `.area`"},{"path":"references/error-handling.md","content":"# 错误处理\n\n## 两层 Fallback 机制\n\n### 第一层:重试\n\n- **触发**:页面加载慢、网络波动、反爬延迟\n- **策略**:增加 50% 等待时间,重试当前步骤\n- **上限**:同一位置最多重试 3 次,超过则升级到第二层\n\n### 第二层:修复并继续\n\n- **触发**:一层重试仍失败\n- **策略**:\n  1. 跳过当前错误职位\n  2. 记录错误到 `temp/error_log.csv`\n  3. 继续处理下一个职位\n\n### 错误日志格式\n\n```csv\ntitle,job_id,error,timestamp\n```\n\n---\n\n## 当前需注意的陷阱\n\n- **详情页等待必须 20 秒以上**:不足会导致职位描述全部为空(0 字节)\n- **滚动必须模拟人类节奏**:快速连续滚动会导致列表只加载部分(如 50 条只加载 15 条)\n- **CSV 必须用 `'a'` 追加模式**:`'w'` 模式会覆盖历史数据\n\n---\n\n## 已修复问题归档\n\n| # | 问题 | 修复方案 | 状态 |\n|---|------|---------|------|\n| 0 | CSV 覆盖历史数据 | 改用 `'a'` 追加 + job_id 去重 | ✅ |\n| 1 | 列表页上限 49 条 | 连续 5 次滚动不变则停止 | ✅ |\n| 2 | 职位描述提取不完整 | 从“职位描述”标题开始，多结束标记 | ✅ |\n| 3 | job_id 提取失败 | 改用 `.+?\\.html` 正则 | ✅ |\n| 4 | 公司名称解析错误 | 直接从 link 元素获取 | ✅ |\n| 5 | security_id 提取为空 | 从详情页 script 标签正则提取 | ✅ |\n| 6 | 职位描述全部为空 | 等待增至 20 秒 + readyState 检测 | ✅ |\n| 7 | 快速滚动加载不全 | 人类滚动策略：随机等待 + 停留 | ✅ |\n| 8 | WebSocket 长连接超时 | 每次 CDP 操作新建独立连接，用完即关 | ✅ |\n| 9 | 进程中断数据全丢 | 每提取 1 条立即追加写入 CSV | ✅ |\n| 10 | 详情页 tab 堆积 | 提取后立即 close_tab | ✅ |\n| 11 | 反爬检测（固定间隔） | 所有等待时间加随机波动 | ✅ |"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"从 Boss 直聘批量爬取职位详情（含 security_id、职位描述），支持 PUA 薪资解码和增量去重。 Skill: boss-resume-crawler Owner: iichaner Summary: 从 Boss 直聘批量爬取职位详情（含 security_id、职位描述），支持 PUA 薪资解码和增量去重。 Tags: latest:0.1.0 Version history: v0.1.0 | 2026-08-18T10:45:58.143Z | auto Initial release of boss-resume-crawler. - Supports batch crawling of Boss直聘 job details, including security_id and job description - Implements PUA salary decoding and incremental deduplication - Provides strict dependency checks (Pyth","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":713,"uniquenessScore":57,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-10T14:22:16.495Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-10T14:22:16.495Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T17:37:45.987Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}