{"id":"8254012c-fd40-4b59-bdca-1ec73cb7dae8","entityType":"agent","slug":"clawhub-cat-xierluo-paddle-ocr","name":"PaddleOCR","canonicalUrl":"https://www.xpersona.co/agent/clawhub-cat-xierluo-paddle-ocr","canonicalPath":"/agent/clawhub-cat-xierluo-paddle-ocr","generatedAt":"2026-10-10T10:43:47.322Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-10T08:16:05.771Z","emptyReason":null},"description":"面向法律 PDF 与扫描件的 PaddleOCR 结构化解析技能。默认将本地 PDF 或图片转换为 Markdown，并在技能内部保留可追溯 archive 归档。本技能应在用户需要法律 PDF OCR、卷宗 OCR、病历 OCR、证据扫描件转 Markdown、表格识别、公式识别、版面分析、PDF 转 Mark... Skill: PaddleOCR Owner: cat-xierluo Summary: 面向法律 PDF 与扫描件的 PaddleOCR 结构化解析技能。默认将本地 PDF 或图片转换为 Markdown，并在技能内部保留可追溯 archive 归档。本技能应在用户需要法律 PDF OCR、卷宗 OCR、病历 OCR、证据扫描件转 Markdown、表格识别、公式识别、版面分析、PDF 转 Mark... Tags: latest:1.1.1 Version history: v1.1.1 | 2026-04-15T10:02:54.231Z | user 面向法律 PDF 与扫描件的 PaddleOCR 结构化解析，支持表格识别、公式识别、版面分析，保留 archive 归档 Archive index: Archive v1.1.1: 13 files, 23861 bytes Files: CHANGELOG.md (2","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.6K downloads reported by the source. Last updated 10/10/2026.","installCommand":"clawhub skill install s179ghx665dhgcsrdzyss2am0x83g7nf:paddle-ocr","sourceUrl":"https://clawhub.ai/cat-xierluo/paddle-ocr","homepage":"https://clawhub.ai/cat-xierluo/skills/paddle-ocr","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/cat-xierluo/paddle-ocr","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/cat-xierluo/skills/paddle-ocr","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":64,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"面向法律 PDF 与扫描件的 PaddleOCR 结构化解析技能。默认将本地 PDF 或图片转换为 Markdown，并在技能内部保留可追溯 archive 归档。本技能应在用户需要法律 PDF OCR、卷宗 OCR、病历 OCR、证据扫描件转 Markdown、表格识别、公式识别、版面分析、PDF 转 Mark..."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-10T08:16:05.771Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T08:16:05.771Z","emptyReason":null},"stars":null,"forks":null,"downloads":1564,"packageName":null,"latestVersion":"1.1.1","tractionLabel":"1.6K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T08:16:05.771Z","emptyReason":null},"lastUpdatedAt":"2026-10-10T08:16:05.771Z","lastCrawledAt":"2026-10-10T08:16:05.771Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-11T08:16:05.771Z","lastVerifiedAt":null,"highlights":[{"version":"1.1.1","createdAt":"2026-04-15T10:02:54.231Z","changelog":"面向法律 PDF 与扫描件的 PaddleOCR 结构化解析，支持表格识别、公式识别、版面分析，保留 archive 归档","fileCount":13,"zipByteSize":23861}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s179ghx665dhgcsrdzyss2am0x83g7nf:paddle-ocr","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-cat-xierluo-paddle-ocr/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-cat-xierluo-paddle-ocr/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-cat-xierluo-paddle-ocr/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-cat-xierluo-paddle-ocr/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-cat-xierluo-paddle-ocr/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-cat-xierluo-paddle-ocr/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T10:43:47.322Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-cat-xierluo-paddle-ocr/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-cat-xierluo-paddle-ocr/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-cat-xierluo-paddle-ocr/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-cat-xierluo-paddle-ocr/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-10T08:16:05.771Z","emptyReason":null},"readme":"Skill: PaddleOCR\n\nOwner: cat-xierluo\n\nSummary: 面向法律 PDF 与扫描件的 PaddleOCR 结构化解析技能。默认将本地 PDF 或图片转换为 Markdown，并在技能内部保留可追溯 archive 归档。本技能应在用户需要法律 PDF OCR、卷宗 OCR、病历 OCR、证据扫描件转 Markdown、表格识别、公式识别、版面分析、PDF 转 Mark...\n\nTags: latest:1.1.1\n\nVersion history:\n\nv1.1.1 | 2026-04-15T10:02:54.231Z | user\n\n面向法律 PDF 与扫描件的 PaddleOCR 结构化解析，支持表格识别、公式识别、版面分析，保留 archive 归档\n\nArchive index:\n\nArchive v1.1.1: 13 files, 23861 bytes\n\nFiles: CHANGELOG.md (2835b), LICENSE.txt (1090b), references/output_schema.md (2881b), scripts/convert.js (1618b), scripts/convert.py (14169b), scripts/layout_caller.py (2376b), scripts/lib.py (10886b), scripts/optimize_file.py (3374b), scripts/smoke_test.py (2069b), scripts/split_pdf.py (4992b), skill-card.md (2294b), SKILL.md (6604b), _meta.json (129b)\n\nFile v1.1.1:SKILL.md\n\n---\nname: paddle-ocr\nhomepage: https://github.com/cat-xierluo/legal-skills\nauthor: 杨卫薪律师（微信ywxlaw）\nversion: \"1.1.1\"\nlicense: MIT\ndescription: 面向法律 PDF 与扫描件的 PaddleOCR 结构化解析技能。默认将本地 PDF 或图片转换为 Markdown，并在技能内部保留可追溯 archive 归档。本技能应在用户需要法律 PDF OCR、卷宗 OCR、病历 OCR、证据扫描件转 Markdown、表格识别、公式识别、版面分析、PDF 转 Markdown、复杂 PDF 解析时使用。\n---\n# PaddleOCR 法律 PDF 转 Markdown\n\n本技能服务于**法律材料 OCR**。默认目标不是返回一段临时文本，而是：\n\n1. 将本地 PDF / 图片转换为可继续编辑和分析的 Markdown。\n2. 在 `archive/` 下保留完整归档，便于复核、追溯和二次处理。\n\n## 何时使用\n\n在以下场景使用本技能：\n\n- 需要把卷宗、病历、证据材料、法院通知、财报、票据等扫描 PDF 转成 Markdown。\n- 文档包含表格、印章、页眉页脚、多栏排版、公式或复杂版面。\n- 希望保留一个技能内的 archive，沉淀原文件、Markdown、结构化 JSON 和批次结果。\n- 后续还要继续做法律分析、证据摘录、知识入库或 RAG 切片。\n\n在以下场景不要优先使用本技能：\n\n- 只是快速读取一小段清晰文本，且不需要 Markdown 和归档。\n- 只是截图抄字，速度比结构化质量更重要。\n- 输入不是 PDF / 常见图片格式。\n\n## 主产出\n\n默认主产出只有两类：\n\n- **Markdown 文件**：保存在源文件同目录，默认与原文件同名、扩展名为 `.md`\n- **archive 归档目录**：保存在 `paddle-ocr/archive/时间戳_文件名/`\n\narchive 默认包含：\n\n- 原始输入文件副本\n- 最终 `result.md`\n- 最终 `result.json`\n- 批次级 `batches/*.json`\n- 提取出的图片资源\n- `metadata.json`\n\n## 依赖\n\n### 系统依赖\n\n| 依赖 | 安装方式 |\n|------|----------|\n| `python3` | macOS 通常已内置 |\n| `uv` | macOS: `brew install uv` |\n\n### Python 包\n\n脚本使用 `uv run` 执行，依赖写在脚本头部，无需单独维护 `requirements.txt`。\n\n## 首次配置\n\n### 获取 API 信息\n\n1. 打开 [PaddleOCR 官网](https://www.paddleocr.com)\n2. 进入对应模型的 API 页面\n3. 在示例代码中复制：\n   - `API_URL`\n   - `Access Token`\n\n### 配置方式\n\n优先编辑 `config/.env`：\n\n```bash\ncd paddle-ocr/config\ncp .env.example .env\nnano .env\n```\n\n必填项：\n\n- `PADDLEOCR_DOC_PARSING_API_URL`\n- `PADDLEOCR_ACCESS_TOKEN`\n\n## 常用命令\n\n### 主工作流：生成 Markdown + archive\n\n在技能根目录运行：\n\n```bash\nuv run scripts/convert.py \"/path/to/legal-document.pdf\"\n```\n\n或继续兼容旧入口：\n\n```bash\n/usr/bin/osascript -l JavaScript scripts/convert.js \"/path/to/legal-document.pdf\"\n```\n\n可选参数：\n\n```bash\nuv run scripts/convert.py \"/path/to/legal-document.pdf\" --pages \"1-20\"\nuv run scripts/convert.py \"/path/to/legal-document.pdf\" --output \"/tmp/output.md\"\nuv run scripts/convert.py \"/path/to/legal-document.pdf\" --archive-name \"某案卷宗-证据一\"\n```\n\n### 底层调试：只调用解析接口，输出结构化 JSON\n\n```bash\nuv run scripts/layout_caller.py --file-path \"/path/to/legal-document.pdf\" --pretty\nuv run scripts/layout_caller.py --file-url \"https://example.com/document.pdf\" --stdout --pretty\n```\n\n当你只想检查原始接口结果，或后续要自己解析表格/坐标信息时，使用这个底层脚本。\n\n### 自检\n\n```bash\nuv run scripts/smoke_test.py --skip-api-test\nuv run scripts/smoke_test.py\n```\n\n### 拆分页码\n\n```bash\nuv run scripts/split_pdf.py input.pdf output.pdf --pages \"1-5,8,10-12\"\n```\n\n## 法律 PDF 工作流\n\n按以下顺序工作：\n\n1. 优先使用 `scripts/convert.py`。\n2. 如只需部分页码，先传 `--pages`，避免整卷上传。\n3. 对大体量卷宗，脚本会按配置自动分批请求，再合并为一个 Markdown。\n4. 需要复核时，到 `archive/` 查看：\n   - `output/result.md`\n   - `output/result.json`\n   - `metadata.json`\n   - `batches/*.json`\n\n## 大文件策略\n\n本技能为了法律材料的稳定性，默认采用**保守批次策略**：\n\n- PDF 页数超过 `PADDLEOCR_BATCH_PAGES` 时自动分批\n- 预估 Base64 大小超过 `PADDLEOCR_MAX_BASE64_MB` 时自动分批\n\n这意味着它可能比官方上限更早拆分，但通常能降低长卷宗、病历合并件和扫描质量不稳定文档的失败率。\n\n## 输出说明\n\n### Markdown\n\n- 默认保存到源文件同目录\n- 如果传 `--output` 且是 `.md` 文件路径，则保存到指定路径\n- 如果 `--output` 是目录，则在该目录下生成同名 `.md`\n\n### archive\n\n默认归档目录结构：\n\n```text\narchive/\n└── 20260405_153000_文件名/\n    ├── input/\n    │   └── 原文件.pdf\n    ├── output/\n    │   ├── result.md\n    │   ├── result.json\n    │   └── images/\n    ├── batches/\n    │   ├── batch_001_1-40.json\n    │   └── batch_002_41-67.json\n    └── metadata.json\n```\n\n## 配置项\n\n编辑 `config/.env`：\n\n| 选项 | 默认值 | 说明 |\n|------|--------|------|\n| `PADDLEOCR_DOC_PARSING_API_URL` | 空 | 官方要求的完整 `layout-parsing` 端点 |\n| `PADDLEOCR_ACCESS_TOKEN` | 空 | 官方 Access Token |\n| `PADDLEOCR_DOC_ORIENTATION` | `false` | 是否启用方向分类 |\n| `PADDLEOCR_DOC_UNWARP` | `false` | 是否启用去扭曲 |\n| `PADDLEOCR_CHART_RECOG` | `false` | 是否启用图表识别 |\n| `PADDLEOCR_DOC_PARSING_TIMEOUT` | `600` | 单次请求超时秒数 |\n| `PADDLEOCR_BATCH_PAGES` | `40` | PDF 自动分批页数阈值兼批次大小 |\n| `PADDLEOCR_MAX_BASE64_MB` | `20` | 触发分批的保守大小阈值 |\n| `PADDLEOCR_LOG_LEVEL` | `medium` | `low` / `medium` / `high` |\n\n## 结果结构\n\n如果需要理解底层 JSON 包装格式，读取：\n\n- `references/output_schema.md`\n\n## 故障排除\n\n| 问题 | 解决方式 |\n|------|----------|\n| 未配置 API | 先补 `config/.env`，再执行 `uv run scripts/smoke_test.py --skip-api-test` |\n| 403 / Token 错误 | 更新 `PADDLEOCR_ACCESS_TOKEN` |\n| 请求超时 | 调大 `PADDLEOCR_DOC_PARSING_TIMEOUT`，或减少页码范围 |\n| 大 PDF 失败 | 使用 `--pages` 缩小范围，或让脚本自动分批 |\n| Markdown 为空 | 到 `archive/` 查看 `batches/*.json` 和 `metadata.json`，确认是否原文件质量过差 |\n| 需要看原始坐标和表格结构 | 使用 `scripts/layout_caller.py`，并读取 `result.result.layoutParsingResults[*].prunedResult` |\n\n## 维护建议\n\n修改本技能后，同步更新：\n\n- `CHANGELOG.md`\n\nFile v1.1.1:_meta.json\n\n{\n  \"ownerId\": \"kn7bn9h1qxa9ja48qkmaxtfjgx81ksex\",\n  \"slug\": \"paddle-ocr\",\n  \"version\": \"1.1.1\",\n  \"publishedAt\": 1776247374231\n}\n\nFile v1.1.1:references/output_schema.md\n\n# 输出结构说明\n\n本技能有两层输出：\n\n1. **底层接口层**：`scripts/layout_caller.py` 输出稳定 JSON envelope\n2. **高层法律工作流**：`scripts/convert.py` 生成 Markdown，并把结构化结果写入 archive\n\n## 一、`layout_caller.py` 输出结构\n\n`layout_caller.py` 用于直接调用 PaddleOCR 接口，返回统一包装：\n\n```json\n{\n  \"ok\": true,\n  \"text\": \"从所有页面拼接出的 Markdown 文本\",\n  \"result\": { \"errorCode\": 0, \"result\": { \"...\": \"原始接口结果\" } },\n  \"error\": null\n}\n```\n\n失败时：\n\n```json\n{\n  \"ok\": false,\n  \"text\": \"\",\n  \"result\": null,\n  \"error\": {\n    \"code\": \"CONFIG_ERROR | INPUT_ERROR | API_ERROR\",\n    \"message\": \"可直接展示给用户的错误信息\"\n  }\n}\n```\n\n重点字段：\n\n- `text`：由 `result.result.layoutParsingResults[*].markdown.text` 拼接而成\n- `result.result.layoutParsingResults[*].markdown.images`：页面内图片资源\n- `result.result.layoutParsingResults[*].prunedResult`：坐标、分类、置信度等结构化版面信息\n\n## 二、`convert.py` 的 archive 结构\n\n`convert.py` 是高层入口，默认生成 Markdown 并写入 `archive/`。\n\n归档目录示例：\n\n```text\narchive/\n└── 20260405_153000_某案卷宗/\n    ├── input/\n    │   └── 某案卷宗.pdf\n    ├── output/\n    │   ├── result.md\n    │   ├── result.json\n    │   └── images/\n    ├── batches/\n    │   ├── batch_001_1-40.json\n    │   └── batch_002_41-67.json\n    └── metadata.json\n```\n\n### `output/result.json`\n\n这是高层工作流的汇总文件，包含：\n\n- 输入文件基本信息\n- 处理模式（单次 / 自动分批）\n- 提取出的全文 Markdown\n- 输出图片列表\n- 各批次摘要\n\n示例：\n\n```json\n{\n  \"ok\": true,\n  \"source\": {\n    \"path\": \"/path/to/file.pdf\",\n    \"name\": \"file.pdf\",\n    \"sha256\": \"...\"\n  },\n  \"processing\": {\n    \"mode\": \"batched\",\n    \"batch_count\": 2,\n    \"total_pages\": 67,\n    \"processed_pages\": 67,\n    \"selected_pages\": \"1-67\"\n  },\n  \"text\": \"最终合并后的 Markdown\",\n  \"images\": [],\n  \"batches\": [\n    {\n      \"index\": 1,\n      \"label\": \"1-40\",\n      \"text_length\": 12345,\n      \"image_count\": 2\n    }\n  ]\n}\n```\n\n### `batches/*.json`\n\n每个批次对应一个底层 envelope，便于排查：\n\n- 哪一批 OCR 异常\n- 哪一批版面错乱\n- 某页的 `prunedResult` 是否需要单独读取\n\n### `metadata.json`\n\n记录：\n\n- 处理时间\n- provider 名称\n- 关键配置\n- Markdown 输出路径\n- 图片目录路径\n\n## 三、建议读取顺序\n\n如果只是要最终文本：\n\n1. 读取 `output/result.md`\n\n如果需要排查 OCR 质量：\n\n1. 读取 `output/result.json`\n2. 再按需读取 `batches/*.json`\n\n如果需要提取表格、坐标、阅读顺序：\n\n1. 直接运行 `scripts/layout_caller.py`\n2. 读取 `result.result.layoutParsingResults[*].prunedResult`\n\nFile v1.1.1:CHANGELOG.md\n\n# 变更记录\n\n## [1.1.1] - 2026-04-05\n\n### 改进\n- 将对外配置字段统一收敛为官方命名：`PADDLEOCR_DOC_PARSING_API_URL` 与 `PADDLEOCR_ACCESS_TOKEN`。\n- `SKILL.md` 与 `.env.example` 删除旧别名说明，避免用户在配置时产生歧义。\n\n### 技术优化\n- `scripts/lib.py` 不再从旧别名字段读取 API 地址和 Token，配置接口与官方保持一致。\n\n### 文档完善\n- 更新配置章节，明确只按官方字段填写 `.env`。\n\n## [1.1.0] - 2026-04-05\n\n### 新增\n- 新增 `scripts/lib.py`，统一配置读取、接口调用、稳定 JSON envelope 与错误包装。\n- 新增 `scripts/layout_caller.py`，支持直接调试底层 JSON 结果。\n- 新增 `scripts/split_pdf.py`，支持 PDF 页码提取与自动分批。\n- 新增 `scripts/smoke_test.py`，支持配置检查与 API 连通性自检。\n- 新增 `scripts/optimize_file.py`，支持对扫描图片做压缩优化。\n- 新增 `references/output_schema.md`，说明底层 JSON envelope 与 archive 结构。\n- 新增 `TASKS.md` 与 `DECISIONS.md`，补齐技能级协作文档。\n- 新增 `LICENSE.txt`，统一许可证文件名与版权信息。\n\n### 改进\n- 将技能定位收敛为“面向法律 PDF / 扫描件的 Markdown + archive 工作流”。\n- 默认输出保持为 Markdown 文件，并在技能内部保留可追溯 archive。\n- 为卷宗、病历、证据材料等长文档增加自动分批逻辑，优先保障稳定性。\n- `convert.js` 改为兼容层，转而调用 Python 主链路，不再内置核心 OCR 逻辑。\n- `SKILL.md` 重写为以法律文档场景为中心的说明文档，并补充适用/不适用场景。\n\n### 技术优化\n- 移除旧的固定 `test/paddle-ocr` 路径依赖，改为基于脚本位置动态推导 skill 根目录。\n- 用 `pypdfium2` 替代 Ghostscript 方案，降低系统依赖。\n- 统一支持新旧环境变量字段，兼容已有 `.env` 配置。\n- 归档目录新增 `metadata.json`、批次 JSON 和输出结构说明，增强复核与追溯能力。\n\n### 文档完善\n- 将配置说明更新为 `PADDLEOCR_DOC_PARSING_API_URL` / `PADDLEOCR_ACCESS_TOKEN` 主字段。\n- 补充大文件策略、页码范围、主入口与底层入口的分工说明。\n\n### 待办事项\n- 增加真实法律 PDF 样本的回归测试集。\n- 评估页眉页脚、印章、批注的后处理去噪规则。\n\n## [1.0.0] - 2026-01-15\n\n### 新增\n- 初始版本发布。\n- 支持将 PDF 和图片转换为 Markdown。\n- 集成 PaddleOCR 文档解析接口。\n- 支持 OCR、表格识别、公式识别与图片提取。\n- 增加基础 archive 归档能力。\n\n### 技术优化\n- 使用 JXA 与 Python 组合实现基础转换流程。\n- 支持 Base64 上传与 `fileType` 自动检测。\n\n### 文档完善\n- 提供基础配置说明、故障排除和与 MinerU 的差异说明。\n\nFile v1.1.1:skill-card.md\n\n## Description:\n\nPaddleOCR converts legal PDFs and scanned document images into Markdown with structured OCR outputs and a traceable local archive.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[cat-xierluo](https://clawhub.ai/user/cat-xierluo)\n\n### License/Terms of Use:\n\nMIT\n\n## Use Case:\n\nDevelopers, legal operations teams, and agents use this skill to convert legal PDFs, case files, medical records, evidence scans, tables, formulas, and complex layouts into editable Markdown and archived JSON outputs for review or downstream analysis.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Sensitive legal, medical, financial, or evidence files may be uploaded to the configured OCR service.\n\nMitigation: Use only a trusted PaddleOCR endpoint and prefer limited page ranges for sensitive documents.\n\nRisk: The skill keeps local archives by default, which may retain sensitive source files, OCR text, JSON, and extracted images.\n\nMitigation: Use --no-archive when retention is not appropriate and review archive storage permissions before processing sensitive files.\n\nRisk: Output image directories can be replaced during conversion.\n\nMitigation: Avoid choosing output paths where an existing *_images directory contains important files.\n\n## Reference(s):\n\n- [PaddleOCR official site](https://www.paddleocr.com)\n- [Output schema](references/output_schema.md)\n- [ClawHub skill page](https://clawhub.ai/cat-xierluo/skills/paddle-ocr)\n- [Project homepage](https://github.com/cat-xierluo/legal-skills)\n\n## Skill Output:\n\n**Output Type(s):** [Markdown, Text, JSON, Files, Shell commands, Configuration guidance]\n\n**Output Format:** [Markdown files, JSON archive files, local file paths, and shell command guidance]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Creates a local archive by default; optional page ranges and output paths can narrow processing.]\n\n## Skill Version(s):\n\n1.1.1 (source: frontmatter and changelog, released 2026-04-05)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.1.1:LICENSE.txt\n\nMIT License\n\nCopyright (c) 2025 杨卫薪律师（微信ywxlaw）\n\nPermission is hereby granted, free of charge, to any person obtaining a copy\nof this software and associated documentation files (the \"Software\"), to deal\nin the Software without restriction, including without limitation the rights\nto use, copy, modify, merge, publish, distribute, sublicense, and/or sell\ncopies of the Software, and to permit persons to whom the Software is\nfurnished to do so, subject to the following conditions:\n\nThe above copyright notice and this permission notice shall be included in all\ncopies or substantial portions of the Software.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR\nIMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,\nFITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE\nAUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER\nLIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,\nOUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE\nSOFTWARE.","readmeExcerpt":"Skill: PaddleOCR Owner: cat-xierluo Summary: 面向法律 PDF 与扫描件的 PaddleOCR 结构化解析技能。默认将本地 PDF 或图片转换为 Markdown，并在技能内部保留可追溯 archive 归档。本技能应在用户需要法律 PDF OCR、卷宗 OCR、病历 OCR、证据扫描件转 Markdown、表格识别、公式识别、版面分析、PDF 转 Mark... Tags: latest:1.1.1 Version history: v1.1.1 | 2026-04-15T10:02:54.231Z | user 面向法律 PDF 与扫描件的 PaddleOCR 结构化解析，支持表格识别、公式识别、版面分析，保留 archive 归档 Archive index: Archive v1.1.1: 13 files, 23861 bytes Files: CHANGELOG.md (2","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"cd paddle-ocr/config\ncp .env.example .env\nnano .env"},{"language":"bash","snippet":"uv run scripts/convert.py \"/path/to/legal-document.pdf\""},{"language":"bash","snippet":"/usr/bin/osascript -l JavaScript scripts/convert.js \"/path/to/legal-document.pdf\""},{"language":"bash","snippet":"uv run scripts/convert.py \"/path/to/legal-document.pdf\" --pages \"1-20\"\nuv run scripts/convert.py \"/path/to/legal-document.pdf\" --output \"/tmp/output.md\"\nuv run scripts/convert.py \"/path/to/legal-document.pdf\" --archive-name \"某案卷宗-证据一\""},{"language":"bash","snippet":"uv run scripts/layout_caller.py --file-path \"/path/to/legal-document.pdf\" --pretty\nuv run scripts/layout_caller.py --file-url \"https://example.com/document.pdf\" --stdout --pretty"},{"language":"bash","snippet":"uv run scripts/smoke_test.py --skip-api-test\nuv run scripts/smoke_test.py"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: paddle-ocr\nhomepage: https://github.com/cat-xierluo/legal-skills\nauthor: 杨卫薪律师（微信ywxlaw）\nversion: \"1.1.1\"\nlicense: MIT\ndescription: 面向法律 PDF 与扫描件的 PaddleOCR 结构化解析技能。默认将本地 PDF 或图片转换为 Markdown，并在技能内部保留可追溯 archive 归档。本技能应在用户需要法律 PDF OCR、卷宗 OCR、病历 OCR、证据扫描件转 Markdown、表格识别、公式识别、版面分析、PDF 转 Markdown、复杂 PDF 解析时使用。\n---\n# PaddleOCR 法律 PDF 转 Markdown\n\n本技能服务于**法律材料 OCR**。默认目标不是返回一段临时文本，而是：\n\n1. 将本地 PDF / 图片转换为可继续编辑和分析的 Markdown。\n2. 在 `archive/` 下保留完整归档，便于复核、追溯和二次处理。\n\n## 何时使用\n\n在以下场景使用本技能：\n\n- 需要把卷宗、病历、证据材料、法院通知、财报、票据等扫描 PDF 转成 Markdown。\n- 文档包含表格、印章、页眉页脚、多栏排版、公式或复杂版面。\n- 希望保留一个技能内的 archive，沉淀原文件、Markdown、结构化 JSON 和批次结果。\n- 后续还要继续做法律分析、证据摘录、知识入库或 RAG 切片。\n\n在以下场景不要优先使用本技能：\n\n- 只是快速读取一小段清晰文本，且不需要 Markdown 和归档。\n- 只是截图抄字，速度比结构化质量更重要。\n- 输入不是 PDF / 常见图片格式。\n\n## 主产出\n\n默认主产出只有两类：\n\n- **Markdown 文件**：保存在源文件同目录，默认与原文件同名、扩展名为 `.md`\n- **archive 归档目录**：保存在 `paddle-ocr/archive/时间戳_文件名/`\n\narchive 默认包含：\n\n- 原始输入文件副本\n- 最终 `result.md`\n- 最终 `result.json`\n- 批次级 `batches/*.json`\n- 提取出的图片资源\n- `metadata.json`\n\n## 依赖\n\n### 系统依赖\n\n| 依赖 | 安装方式 |\n|------|----------|\n| `python3` | macOS 通常已内置 |\n| `uv` | macOS: `brew install uv` |\n\n### Python 包\n\n脚本使用 `uv run` 执行，依赖写在脚本头部，无需单独维护 `requirements.txt`。\n\n## 首次配置\n\n### 获取 API 信息\n\n1. 打开 [PaddleOCR 官网](https://www.paddleocr.com)\n2. 进入对应模型的 API 页面\n3. 在示例代码中复制：\n   - `API_URL`\n   - `Access Token`\n\n### 配置方式\n\n优先编辑 `config/.env`：\n\n```bash\ncd paddle-ocr/config\ncp .env.example .env\nnano .env\n```\n\n必填项：\n\n- `PADDLEOCR_DOC_PARSING_API_URL`\n- `PADDLEOCR_ACCESS_TOKEN`\n\n## 常用命令\n\n### 主工作流：生成 Markdown + archive\n\n在技能根目录运行：\n\n```bash\nuv run scripts/convert.py \"/path/to/legal-document.pdf\"\n```\n\n或继续兼容旧入口：\n\n```bash\n/usr/bin/osascript -l JavaScript scripts/convert.js \"/path/to/legal-document.pdf\"\n```\n\n可选参数：\n\n```bash\nuv run scripts/convert.py \"/path/to/legal-document.pdf\" --pages \"1-20\"\nuv run scripts/convert.py \"/path/to/legal-document.pdf\" --output \"/tmp/output.md\"\nuv run scripts/convert.py \"/path/to/legal-document.pdf\" --archive-name \"某案卷宗-证据一\"\n```\n\n### 底层调试：只调用解析接口，输出结构化 JSON\n\n```bash\nuv run scripts/layout_caller.py --file-path \"/path/to/legal-document.pdf\" --pretty\nuv run scripts/layout_caller.py --file-url \"https://example.com/document.pdf\" --stdout --pretty\n```\n\n当你只想检查原始接口结果，或后续要自己解析表格/坐标信息时，使用这个底层脚本。\n\n### 自检\n\n```bash\nuv run scripts/smoke_test.py --skip-api-test\nuv run scripts/smoke_test.py\n```\n\n### 拆分页码\n\n```bash\nuv run scripts/split_pdf.py input.pdf output.pdf --pages \"1-5,8,10-12\"\n```\n\n## 法律 PDF 工作流\n\n按以下顺序工作：\n\n1. 优先使用 `scripts/convert.py`。\n2. 如只需部分页码，先传 `--pages`，避免整卷上传。\n3. 对大体量卷宗，脚本会按配置自动分批请求，再合并为一个 Markdown。\n4. 需要复核时，到 `archive/` 查看：\n   - `output/result.md`\n   - `output/result.json`\n   - `metadata.json`\n   - `batches/*.json`\n\n## 大文件策略\n\n本技能为了法律材料的稳定性，默认采用**保守批次策略**：\n\n- PDF 页数超过 `PADDLEOCR_BATCH_PAGES` 时自动分批\n- 预估 Base64 大小超过 `PADDLEOCR_MAX_BASE64_MB` 时自动分批\n\n这意味着它可能比官方上限更早拆分，但通常能降低长卷宗、病历合并件和扫描质量不稳定文档的失败率。\n\n## 输出说明\n\n### Markdown\n\n- 默认保存到源文件同目录\n- 如果传 `--output` 且是 `.md` 文件路径，则保存到指定路径\n- 如果 `--output` 是目录，则在该目录下生成同名 `.md`\n\n### archive\n\n默认归档目录结构：\n\n```text\narchive/\n└── 2026"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7bn9h1qxa9ja48qkmaxtfjgx81ksex\",\n  \"slug\": \"paddle-ocr\",\n  \"version\": \"1.1.1\",\n  \"publishedAt\": 1776247374231\n}"},{"path":"references/output_schema.md","content":"# 输出结构说明\n\n本技能有两层输出：\n\n1. **底层接口层**：`scripts/layout_caller.py` 输出稳定 JSON envelope\n2. **高层法律工作流**：`scripts/convert.py` 生成 Markdown，并把结构化结果写入 archive\n\n## 一、`layout_caller.py` 输出结构\n\n`layout_caller.py` 用于直接调用 PaddleOCR 接口，返回统一包装：\n\n```json\n{\n  \"ok\": true,\n  \"text\": \"从所有页面拼接出的 Markdown 文本\",\n  \"result\": { \"errorCode\": 0, \"result\": { \"...\": \"原始接口结果\" } },\n  \"error\": null\n}\n```\n\n失败时：\n\n```json\n{\n  \"ok\": false,\n  \"text\": \"\",\n  \"result\": null,\n  \"error\": {\n    \"code\": \"CONFIG_ERROR | INPUT_ERROR | API_ERROR\",\n    \"message\": \"可直接展示给用户的错误信息\"\n  }\n}\n```\n\n重点字段：\n\n- `text`：由 `result.result.layoutParsingResults[*].markdown.text` 拼接而成\n- `result.result.layoutParsingResults[*].markdown.images`：页面内图片资源\n- `result.result.layoutParsingResults[*].prunedResult`：坐标、分类、置信度等结构化版面信息\n\n## 二、`convert.py` 的 archive 结构\n\n`convert.py` 是高层入口，默认生成 Markdown 并写入 `archive/`。\n\n归档目录示例：\n\n```text\narchive/\n└── 20260405_153000_某案卷宗/\n    ├── input/\n    │   └── 某案卷宗.pdf\n    ├── output/\n    │   ├── result.md\n    │   ├── result.json\n    │   └── images/\n    ├── batches/\n    │   ├── batch_001_1-40.json\n    │   └── batch_002_41-67.json\n    └── metadata.json\n```\n\n### `output/result.json`\n\n这是高层工作流的汇总文件，包含：\n\n- 输入文件基本信息\n- 处理模式（单次 / 自动分批）\n- 提取出的全文 Markdown\n- 输出图片列表\n- 各批次摘要\n\n示例：\n\n```json\n{\n  \"ok\": true,\n  \"source\": {\n    \"path\": \"/path/to/file.pdf\",\n    \"name\": \"file.pdf\",\n    \"sha256\": \"...\"\n  },\n  \"processing\": {\n    \"mode\": \"batched\",\n    \"batch_count\": 2,\n    \"total_pages\": 67,\n    \"processed_pages\": 67,\n    \"selected_pages\": \"1-67\"\n  },\n  \"text\": \"最终合并后的 Markdown\",\n  \"images\": [],\n  \"batches\": [\n    {\n      \"index\": 1,\n      \"label\": \"1-40\",\n      \"text_length\": 12345,\n      \"image_count\": 2\n    }\n  ]\n}\n```\n\n### `batches/*.json`\n\n每个批次对应一个底层 envelope，便于排查：\n\n- 哪一批 OCR 异常\n- 哪一批版面错乱\n- 某页的 `prunedResult` 是否需要单独读取\n\n### `metadata.json`\n\n记录：\n\n- 处理时间\n- provider 名称\n- 关键配置\n- Markdown 输出路径\n- 图片目录路径\n\n## 三、建议读取顺序\n\n如果只是要最终文本：\n\n1. 读取 `output/result.md`\n\n如果需要排查 OCR 质量：\n\n1. 读取 `output/result.json`\n2. 再按需读取 `batches/*.json`\n\n如果需要提取表格、坐标、阅读顺序：\n\n1. 直接运行 `scripts/layout_caller.py`\n2. 读取 `result.result.layoutParsingResults[*].prunedResult`"},{"path":"CHANGELOG.md","content":"# 变更记录\n\n## [1.1.1] - 2026-04-05\n\n### 改进\n- 将对外配置字段统一收敛为官方命名：`PADDLEOCR_DOC_PARSING_API_URL` 与 `PADDLEOCR_ACCESS_TOKEN`。\n- `SKILL.md` 与 `.env.example` 删除旧别名说明，避免用户在配置时产生歧义。\n\n### 技术优化\n- `scripts/lib.py` 不再从旧别名字段读取 API 地址和 Token，配置接口与官方保持一致。\n\n### 文档完善\n- 更新配置章节，明确只按官方字段填写 `.env`。\n\n## [1.1.0] - 2026-04-05\n\n### 新增\n- 新增 `scripts/lib.py`，统一配置读取、接口调用、稳定 JSON envelope 与错误包装。\n- 新增 `scripts/layout_caller.py`，支持直接调试底层 JSON 结果。\n- 新增 `scripts/split_pdf.py`，支持 PDF 页码提取与自动分批。\n- 新增 `scripts/smoke_test.py`，支持配置检查与 API 连通性自检。\n- 新增 `scripts/optimize_file.py`，支持对扫描图片做压缩优化。\n- 新增 `references/output_schema.md`，说明底层 JSON envelope 与 archive 结构。\n- 新增 `TASKS.md` 与 `DECISIONS.md`，补齐技能级协作文档。\n- 新增 `LICENSE.txt`，统一许可证文件名与版权信息。\n\n### 改进\n- 将技能定位收敛为“面向法律 PDF / 扫描件的 Markdown + archive 工作流”。\n- 默认输出保持为 Markdown 文件，并在技能内部保留可追溯 archive。\n- 为卷宗、病历、证据材料等长文档增加自动分批逻辑，优先保障稳定性。\n- `convert.js` 改为兼容层，转而调用 Python 主链路，不再内置核心 OCR 逻辑。\n- `SKILL.md` 重写为以法律文档场景为中心的说明文档，并补充适用/不适用场景。\n\n### 技术优化\n- 移除旧的固定 `test/paddle-ocr` 路径依赖，改为基于脚本位置动态推导 skill 根目录。\n- 用 `pypdfium2` 替代 Ghostscript 方案，降低系统依赖。\n- 统一支持新旧环境变量字段，兼容已有 `.env` 配置。\n- 归档目录新增 `metadata.json`、批次 JSON 和输出结构说明，增强复核与追溯能力。\n\n### 文档完善\n- 将配置说明更新为 `PADDLEOCR_DOC_PARSING_API_URL` / `PADDLEOCR_ACCESS_TOKEN` 主字段。\n- 补充大文件策略、页码范围、主入口与底层入口的分工说明。\n\n### 待办事项\n- 增加真实法律 PDF 样本的回归测试集。\n- 评估页眉页脚、印章、批注的后处理去噪规则。\n\n## [1.0.0] - 2026-01-15\n\n### 新增\n- 初始版本发布。\n- 支持将 PDF 和图片转换为 Markdown。\n- 集成 PaddleOCR 文档解析接口。\n- 支持 OCR、表格识别、公式识别与图片提取。\n- 增加基础 archive 归档能力。\n\n### 技术优化\n- 使用 JXA 与 Python 组合实现基础转换流程。\n- 支持 Base64 上传与 `fileType` 自动检测。\n\n### 文档完善\n- 提供基础配置说明、故障排除和与 MinerU 的差异说明。"},{"path":"skill-card.md","content":"## Description:\n\nPaddleOCR converts legal PDFs and scanned document images into Markdown with structured OCR outputs and a traceable local archive.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[cat-xierluo](https://clawhub.ai/user/cat-xierluo)\n\n### License/Terms of Use:\n\nMIT\n\n## Use Case:\n\nDevelopers, legal operations teams, and agents use this skill to convert legal PDFs, case files, medical records, evidence scans, tables, formulas, and complex layouts into editable Markdown and archived JSON outputs for review or downstream analysis.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Sensitive legal, medical, financial, or evidence files may be uploaded to the configured OCR service.\n\nMitigation: Use only a trusted PaddleOCR endpoint and prefer limited page ranges for sensitive documents.\n\nRisk: The skill keeps local archives by default, which may retain sensitive source files, OCR text, JSON, and extracted images.\n\nMitigation: Use --no-archive when retention is not appropriate and review archive storage permissions before processing sensitive files.\n\nRisk: Output image directories can be replaced during conversion.\n\nMitigation: Avoid choosing output paths where an existing *_images directory contains important files.\n\n## Reference(s):\n\n- [PaddleOCR official site](https://www.paddleocr.com)\n- [Output schema](references/output_schema.md)\n- [ClawHub skill page](https://clawhub.ai/cat-xierluo/skills/paddle-ocr)\n- [Project homepage](https://github.com/cat-xierluo/legal-skills)\n\n## Skill Output:\n\n**Output Type(s):** [Markdown, Text, JSON, Files, Shell commands, Configuration guidance]\n\n**Output Format:** [Markdown files, JSON archive files, local file paths, and shell command guidance]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Creates a local archive by default; optional page ranges and output paths can narrow processing.]\n\n## Skill Version(s):\n\n1.1.1 (source: frontmatter and changelog, released 2026-04-05)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"面向法律 PDF 与扫描件的 PaddleOCR 结构化解析技能。默认将本地 PDF 或图片转换为 Markdown，并在技能内部保留可追溯 archive 归档。本技能应在用户需要法律 PDF OCR、卷宗 OCR、病历 OCR、证据扫描件转 Markdown、表格识别、公式识别、版面分析、PDF 转 Mark... Skill: PaddleOCR Owner: cat-xierluo Summary: 面向法律 PDF 与扫描件的 PaddleOCR 结构化解析技能。默认将本地 PDF 或图片转换为 Markdown，并在技能内部保留可追溯 archive 归档。本技能应在用户需要法律 PDF OCR、卷宗 OCR、病历 OCR、证据扫描件转 Markdown、表格识别、公式识别、版面分析、PDF 转 Mark... Tags: latest:1.1.1 Version history: v1.1.1 | 2026-04-15T10:02:54.231Z | user 面向法律 PDF 与扫描件的 PaddleOCR 结构化解析，支持表格识别、公式识别、版面分析，保留 archive 归档 Archive index: Archive v1.1.1: 13 files, 23861 bytes Files: CHANGELOG.md (2","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":970,"uniquenessScore":50,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-10T08:16:05.771Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-10T08:16:05.771Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T10:43:47.322Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}