{"id":"b79792b7-bebc-4d08-8c20-ae51a91ad8ad","entityType":"agent","slug":"clawhub-tobewin-china-doc-ocr","name":"china-doc-ocr","canonicalUrl":"https://www.xpersona.co/agent/clawhub-tobewin-china-doc-ocr","canonicalPath":"/agent/clawhub-tobewin-china-doc-ocr","generatedAt":"2026-10-09T18:22:05.938Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-09T16:13:35.034Z","emptyReason":null},"description":"智能文档OCR识别与结构化提取。Use when the user has a complex document, PDF, scanned image, photo, invoice, receipt, ID card, table, or chart that needs to be recognized a... Skill: china-doc-ocr Owner: tobewin Summary: 智能文档OCR识别与结构化提取。Use when the user has a complex document, PDF, scanned image, photo, invoice, receipt, ID card, table, or chart that needs to be recognized a... Tags: china:1.2.0, document:1.2.0, image:1.2.0, invoice:1.2.0, latest:1.2.0, ocr:1.2.0, paddleocr:1.2.0, pdf:1.2.0, recognition:1.2.0, siliconflow:1.2.0 Version history: v1.2.0 | 2026-04-29T01:19:35.463Z | user v1.","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 2.3K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s174rz3f0862tcfw7pzfh5w8kn83hv2z:china-doc-ocr","sourceUrl":"https://clawhub.ai/tobewin/china-doc-ocr","homepage":"https://clawhub.ai/tobewin/skills/china-doc-ocr","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/tobewin/china-doc-ocr","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/tobewin/skills/china-doc-ocr","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":59,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"智能文档OCR识别与结构化提取。Use when the user has a complex document, PDF, scanned image, photo, invoice, receipt, ID card, table, or chart that needs to be recognized a..."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T16:13:35.034Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T16:13:35.034Z","emptyReason":null},"stars":null,"forks":null,"downloads":2337,"packageName":null,"latestVersion":"1.2.0","tractionLabel":"2.3K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T16:13:35.034Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T16:13:35.034Z","lastCrawledAt":"2026-10-09T16:13:35.034Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T16:13:35.034Z","lastVerifiedAt":null,"highlights":[{"version":"1.2.0","createdAt":"2026-04-29T01:19:35.463Z","changelog":"v1.2.0: Security hardening - removed curl dependency, replaced with Python script (ocr.py). Uses urllib for API calls, no external network tools required.","fileCount":7,"zipByteSize":9317},{"version":"1.1.0","createdAt":"2026-03-26T00:57:48.370Z","changelog":"优化模型选择：优先使用PaddleOCR-VL-1.5和DeepSeek-OCR，明确模型降级策略","fileCount":5,"zipByteSize":8262},{"version":"1.0.1","createdAt":"2026-03-23T05:47:23.681Z","changelog":"修复元数据格式：将多行YAML改为单行JSON，解决requires.env声明不被识别的安全扫描问题","fileCount":5,"zipByteSize":8113},{"version":"1.0.0","createdAt":"2026-03-21T15:47:37.366Z","changelog":"china-doc-ocr 1.0.0 – Initial Release - Adds intelligent OCR and structured content extraction for complex documents: PDFs, images, scans, receipts, IDs, tables, and charts. - Supports advanced layouts, multi-column text, and mixed image-text that default tools cannot process. - Integrates DeepSeek-OCR and PaddleOCR-VL models for domestic, VPN-free usage (uses same API key as china-image-gen and china-tts). - Provides fully documented environment checks, file type handling (including PDF and multipage support), and output formatting to Markdown. - Suitable for extracting structured information from complex, scanned, or graphical documents.","fileCount":5,"zipByteSize":8115}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s174rz3f0862tcfw7pzfh5w8kn83hv2z:china-doc-ocr","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tobewin-china-doc-ocr/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tobewin-china-doc-ocr/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tobewin-china-doc-ocr/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-tobewin-china-doc-ocr/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-tobewin-china-doc-ocr/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-tobewin-china-doc-ocr/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-09T18:22:05.937Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tobewin-china-doc-ocr/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tobewin-china-doc-ocr/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tobewin-china-doc-ocr/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tobewin-china-doc-ocr/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-09T16:13:35.034Z","emptyReason":null},"readme":"Skill: china-doc-ocr\n\nOwner: tobewin\n\nSummary: 智能文档OCR识别与结构化提取。Use when the user has a complex document, PDF, scanned image, photo, invoice, receipt, ID card, table, or chart that needs to be recognized a...\n\nTags: china:1.2.0, document:1.2.0, image:1.2.0, invoice:1.2.0, latest:1.2.0, ocr:1.2.0, paddleocr:1.2.0, pdf:1.2.0, recognition:1.2.0, siliconflow:1.2.0\n\nVersion history:\n\nv1.2.0 | 2026-04-29T01:19:35.463Z | user\n\nv1.2.0: Security hardening - removed curl dependency, replaced with Python script (ocr.py). Uses urllib for API calls, no external network tools required.\n\nv1.1.0 | 2026-03-26T00:57:48.370Z | user\n\n优化模型选择：优先使用PaddleOCR-VL-1.5和DeepSeek-OCR，明确模型降级策略\n\nv1.0.1 | 2026-03-23T05:47:23.681Z | user\n\n修复元数据格式：将多行YAML改为单行JSON，解决requires.env声明不被识别的安全扫描问题\n\nv1.0.0 | 2026-03-21T15:47:37.366Z | auto\n\nchina-doc-ocr 1.0.0 – Initial Release\n\n- Adds intelligent OCR and structured content extraction for complex documents: PDFs, images, scans, receipts, IDs, tables, and charts.\n- Supports advanced layouts, multi-column text, and mixed image-text that default tools cannot process.\n- Integrates DeepSeek-OCR and PaddleOCR-VL models for domestic, VPN-free usage (uses same API key as china-image-gen and china-tts).\n- Provides fully documented environment checks, file type handling (including PDF and multipage support), and output formatting to Markdown.\n- Suitable for extracting structured information from complex, scanned, or graphical documents.\n\nArchive index:\n\nArchive v1.2.0: 7 files, 9317 bytes\n\nFiles: LICENSE.txt (533b), references/models.md (1601b), references/prompts.md (3242b), scripts/ocr.py (4284b), skill-card.md (2448b), SKILL.md (5421b), _meta.json (132b)\n\nFile v1.2.0:SKILL.md\n\n---\nname: china-doc-ocr\ndescription: 智能文档OCR识别与结构化提取。Use when the user has a complex document, PDF, scanned image, photo, invoice, receipt, ID card, table, or chart that needs to be recognized and converted to text or Markdown. Uses PaddleOCR-VL-1.5 and DeepSeek-OCR. 文档OCR、发票识别、证件识别。\nversion: 1.2.0\nlicense: MIT-0\nmetadata: {\"openclaw\": {\"emoji\": \"📄\", \"requires\": {\"bins\": [\"python3\"], \"env\": [\"SILICONFLOW_API_KEY\"]}, \"primaryEnv\": \"SILICONFLOW_API_KEY\"}}\n---\n\n# 智能文档 OCR China Doc OCR\n\n识别并提取复杂文档内容：PDF、图片、扫描件、发票、表格、证件等。\n使用硅基流动 DeepSeek-OCR / PaddleOCR-VL，国内直连，无需翻墙。\n\n模型选择与参数说明 → `references/models.md`\n各场景提示词模板 → `references/prompts.md`\n\n## 触发时机\n\n- \"帮我识别这个PDF/图片里的内容\"\n- \"把这张发票/收据的信息提取出来\"\n- \"将这份扫描合同转成可编辑文字\"\n- \"这个表格里的数据帮我提取一下\"\n- \"帮我把这张截图的文字识别出来\"\n- \"这份报告转成 Markdown 格式\"\n- \"识别这张身份证/营业执照的信息\"\n\n---\n\n## 模型选择策略（优先OCR）\n\n```\nOCR优先级：\n1. PaddleOCR-VL-1.5 (免费、快速、专业OCR)\n2. DeepSeek-OCR (免费、效果好)\n3. Qwen2.5-VL-72B (视觉语言模型，OCR效果一般但可补充)\n\n默认使用 PaddleOCR-VL-1.5\n如果识别效果不好，降级到 DeepSeek-OCR\n如果仍然不好，降级到 Qwen2.5-VL-72B\n```\n\n---\n\n## Step 0：环境检查\n\n```bash\n# 检查 API Key\nif [ -z \"$SILICONFLOW_API_KEY\" ]; then\n  echo \"缺少 SILICONFLOW_API_KEY\"\n  echo \"配置方法：\"\n  echo \"  1. 访问 cloud.siliconflow.cn 注册（国内直连）\"\n  echo \"  2. 进入「API密钥」页面创建 Key\"\n  echo \"  3. export SILICONFLOW_API_KEY='sk-xxxxxxxx'\"\n  exit 1\nfi\n```\n\n---\n\n## Step 1：识别内容类型，选择处理模式\n\n```\n用户提供文件路径或 URL → 判断类型：\n\n文件扩展名/用户描述 → 处理模式：\n\n.pdf                    → PDF 模式\n.jpg/.jpeg/.png/.webp   → 图片模式\n.bmp/.tiff/.gif         → 图片模式（先转换格式）\nURL（http/https开头）   → URL 直接模式\n用户粘贴了 base64       → 直接使用\n\n用户意图 → 选择 Prompt 模式：\n\n\"转成文字/提取文字\"     → 通用OCR\n\"转成Markdown/保留格式\" → 文档转Markdown\n\"提取表格/表格数据\"     → 图表解析\n\"发票/收据/单据\"        → 发票识别\n\"身份证/证件/执照\"      → 证件识别\n\"图表/图形/柱状图\"      → 图表解析\n未指定                  → 默认文档转Markdown\n```\n\n---\n\n## Step 2：图片 OCR\n\n### 本地图片文件\n\n```bash\npython3 scripts/ocr.py \\\n  --image \"/path/to/image.jpg\" \\\n  --prompt \"Convert the document to markdown.\" \\\n  --model paddleocr\n```\n\n### 图片 URL\n\n```bash\npython3 scripts/ocr.py \\\n  --url \"https://example.com/document.jpg\" \\\n  --prompt \"Convert the document to markdown.\" \\\n  --model deepseek\n```\n\n### 指定模型\n\n```bash\n# 使用 PaddleOCR（默认，推荐）\npython3 scripts/ocr.py --image photo.jpg --model paddleocr\n\n# 使用 DeepSeek-OCR\npython3 scripts/ocr.py --image photo.jpg --model deepseek\n\n# 使用 Qwen2.5-VL\npython3 scripts/ocr.py --image photo.jpg --model qwen\n```\n\n---\n\n## Step 3：PDF OCR\n\n### 单页或少页 PDF\n\n```bash\npython3 scripts/ocr.py \\\n  --pdf \"/path/to/document.pdf\" \\\n  --prompt \"Convert the document to markdown.\" \\\n  --model deepseek\n```\n\n### 多页 PDF\n\n多页 PDF 需要分页处理。使用 Python 脚本：\n\n1. 使用 pypdf 分页\n2. 对每页分别调用 OCR\n3. 合并结果\n\n---\n\n## Step 4：格式化输出\n\n识别完成后根据用户需求输出：\n\n### 文档转 Markdown（保留结构）\n\n```\n直接输出 Markdown 内容，保留：\n  - 标题层级（# ## ###）\n  - 列表（- * 1.）\n  - 表格（| 列1 | 列2 |）\n  - 代码块（```）\n  - 加粗、斜体等格式\n```\n\n### 发票/证件识别（结构化输出）\n\n```\n发票识别结果\n━━━━━━━━━━━━━━━━━━━━\n发票类型：增值税专用发票\n发票号码：XXXXXXXXXXXXXXXX\n开票日期：2026年03月21日\n购买方：[公司名称]\n销售方：[公司名称]\n商品/服务：[明细]\n不含税金额：¥X,XXX.XX\n税率：13%\n税额：¥XXX.XX\n价税合计：¥X,XXX.XX\n```\n\n### 表格数据（CSV 友好格式）\n\n```\n识别结果同时输出：\n1. Markdown 表格（可读）\n2. 询问用户是否需要 CSV 格式（方便导入 Excel）\n```\n\n---\n\n## 输出文件保存\n\n识别结果保存到工作区，长期保留。\n\n---\n\n## 错误处理\n\n```\n文件不存在           → 提示用户确认路径\n文件过大（>10MB）    → 建议压缩或分页处理\n图片分辨率过低       → 提示识别效果可能较差，建议重新拍摄\nPDF 加密            → 提示需要先解密\n识别结果为空         → 可能是纯图片型PDF，尝试截图后重新识别\n401 错误            → API Key 失效，重新获取\n429 错误            → 请求频率超限，等待后重试\n```\n\n---\n\n## 注意事项\n\n- 图片最小 56×56，最大 3584×3584 像素，超出会自动压缩\n- PDF 支持 base64 编码输入\n- 多页 PDF 需要安装 pypdf（用户需手动安装）\n- detail=high 时按实际像素计费，detail=low 统一约256 token\n- 发票/证件等隐私文件处理后请及时删除工作区临时文件\n\nFile v1.2.0:_meta.json\n\n{\n  \"ownerId\": \"kn75z6gevjsyrznm7dg2ez6sen82h8sz\",\n  \"slug\": \"china-doc-ocr\",\n  \"version\": \"1.2.0\",\n  \"publishedAt\": 1777425575463\n}\n\nFile v1.2.0:references/models.md\n\n# 模型选择说明\n\n来源：硅基流动官方文档\n\n## 主力模型：Pro/deepseek-ai/DeepSeek-V3\n\n```\n模型名：Pro/deepseek-ai/DeepSeek-V3\n特点：\n  - 支持图片和 PDF 输入（base64 或 URL）\n  - 中文文档识别准确率极高\n  - 支持专用 OCR prompt 格式（<image>\\n<|grounding|>...）\n  - 支持多图对比分析\n适用：所有文档类型的首选模型\n```\n\n## 备用模型：Qwen2.5-VL-72B\n\n```\n模型名：Qwen/Qwen2.5-VL-72B-Instruct\n特点：\n  - 超大参数量，复杂文档理解力强\n  - 支持图片输入（不支持 PDF base64）\n  - 中英文双语文档效果好\n适用：DeepSeek-OCR 效果不佳时的备选\n```\n\n## PaddleOCR-VL（专业 OCR 场景）\n\n```\n模型名：PaddlePaddle/PaddleOCR-VL\n特点：\n  - 专为 OCR 任务优化\n  - 支持 CLI 和 API 两种调用方式\n  - 对复杂版面（多列、表格）效果好\n适用：需要精确版面还原的场景\n```\n\n## 图像输入计费（detail 参数影响）\n\n```\ndetail=low：\n  统一压缩为 448×448，约 256 token\n  适合：文字较大、布局简单的文档\n\ndetail=high（推荐）：\n  按实际像素计费：ceil(h/28) × ceil(w/28) token\n  适合：字体较小、布局复杂、表格密集的文档\n\n建议默认使用 detail=high，确保识别准确率\n```\n\n## 模型选择决策\n\n```\n发票/收据/证件        → DeepSeek-V3（中文场景最佳）\n普通文档/报告         → DeepSeek-V3\n复杂表格/多列版面     → DeepSeek-V3 或 PaddleOCR-VL\n英文为主的文档        → Qwen2.5-VL-72B\nDeepSeek 结果不满意  → 改用 Qwen2.5-VL-72B 重试\n```\n\nFile v1.2.0:references/prompts.md\n\n# OCR 提示词模板\n\n来源：硅基流动官方文档 DeepSeek-OCR 专用 prompt 格式\n\n## 使用方式\n\n所有 prompt 在 text 字段中传入，格式固定为：\n`<image>\\n<|grounding|>{具体指令}`\n\n---\n\n## 1. 文档转 Markdown（默认推荐）\n\n```\n<image>\n<|grounding|>Convert the document to markdown.\n```\n\n输出：保留标题层级、列表、表格、加粗等所有格式\n适用：报告、合同、论文、书籍页面、网页截图\n\n---\n\n## 2. 通用 OCR（纯文字提取）\n\n```\n<image>\n<|grounding|>OCR this image.\n```\n\n输出：按阅读顺序提取所有文字，保留基本换行\n适用：简单图片、截图文字、无复杂格式的文档\n\n---\n\n## 3. 无布局 OCR（去除所有格式）\n\n```\n<image>\nFree OCR.\n```\n\n注意：这个格式不加 `<|grounding|>` 前缀\n输出：纯文字流，不保留任何格式信息\n适用：只需要文字内容、不关心排版的场景\n\n---\n\n## 4. 图表/表格解析\n\n```\n<image>\n<|grounding|>Parse the figure.\n```\n\n输出：结构化描述图表内容，表格转为 Markdown 格式\n适用：柱状图、折线图、饼图、数据表格、流程图\n\n---\n\n## 5. 图片详细描述\n\n```\n<image>\n<|grounding|>Describe this image in detail.\n```\n\n输出：详细描述图片内容，包括文字、图形、布局\n适用：混合图文内容，需要全面理解图片的场景\n\n---\n\n## 6. 文字定位\n\n```\n<image>\n<|grounding|>Locate <|ref|>要定位的文字<|/ref|> in the image.\n```\n\n输出：指出特定文字在图片中的位置\n适用：需要找到特定字段位置的场景\n\n---\n\n## 场景专用 Prompt\n\n### 发票识别\n\n```\n请识别这张发票的所有信息，以结构化格式输出：\n发票类型、发票号码、开票日期、\n购买方信息（名称/税号/地址）、\n销售方信息（名称/税号/地址）、\n商品明细（名称/数量/单价/金额）、\n税率、税额、价税合计、备注。\n如有字段无法识别，标注\"不清晰\"。\n```\n\n### 身份证识别\n\n```\n请识别这张身份证的信息：\n姓名、性别、民族、出生日期、住址、公民身份号码。\n输出结构化格式，隐私字段用*号部分隐藏。\n```\n\n### 营业执照识别\n\n```\n请识别这份营业执照的信息：\n公司名称、统一社会信用代码、类型、法定代表人、\n注册资本、成立日期、营业期限、经营范围、注册地址。\n```\n\n### 银行流水/对账单\n\n```\n请识别这份银行流水的所有交易记录，\n以表格格式输出：日期、摘要、支出、收入、余额。\n如有多页请按时间顺序排列。\n```\n\n### 合同关键信息提取\n\n```\n请识别这份合同的关键信息：\n合同编号、签订日期、甲方、乙方、\n合同金额、付款方式、合同期限、\n主要条款摘要（不超过200字）、\n双方签字/盖章情况。\n```\n\n### 学术论文结构化\n\n```\n<image>\n<|grounding|>Convert the document to markdown.\n请保留所有标题层级、公式（用LaTeX格式）、\n参考文献编号和表格结构。\n```\n\n### 表格数据提取（输出 CSV）\n\n```\n请识别这个表格中的所有数据，\n以 CSV 格式输出（逗号分隔），\n第一行为表头，之后每行为一条数据。\n如有合并单元格，请拆分填充。\n```\n\nFile v1.2.0:skill-card.md\n\n## Description:\n\nChina Doc OCR helps agents recognize and structure content from documents, PDFs, scanned images, photos, invoices, receipts, ID cards, tables, and charts using SiliconFlow-hosted OCR models.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[tobewin](https://clawhub.ai/user/tobewin)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers, operators, and external users use this skill to convert Chinese and mixed-language document images, PDFs, invoices, IDs, tables, and charts into editable text, Markdown, or structured extraction results.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill sends documents to a third-party OCR API, which can expose sensitive personal, financial, legal, or regulated business information.\n\nMitigation: Use it only for documents the user is permitted to upload, and redact or avoid IDs, bank statements, contracts, invoices, and regulated records unless upload is approved.\n\nRisk: OCR results and temporary outputs may remain in the workspace after processing.\n\nMitigation: Delete retained workspace outputs and temporary files when the task is finished, especially after handling private documents.\n\nRisk: The skill requires a SiliconFlow API key for network calls.\n\nMitigation: Keep the API key in the environment, avoid pasting it into prompts or files, and rotate it if it may have been exposed.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/tobewin/skills/china-doc-ocr)\n- [Model selection reference](references/models.md)\n- [OCR prompt templates](references/prompts.md)\n- [SiliconFlow API endpoint](https://api.siliconflow.cn/v1/chat/completions)\n- [SiliconFlow console](https://cloud.siliconflow.cn)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, Shell commands, Configuration, Guidance]\n\n**Output Format:** [Markdown, structured text, and shell command snippets]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May save OCR results in the workspace; the helper supports image, PDF, or URL input and defaults to 4096 response tokens.]\n\n## Skill Version(s):\n\n1.2.0 (source: frontmatter, release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.2.0:LICENSE.txt\n\nMIT No Attribution (MIT-0)\n\nCopyright 2026 ToBeWin\n\nPermission is hereby granted, free of charge, to any person obtaining a copy\nof this software and associated documentation files (the \"Software\"), to deal\nin the Software without restriction, including without limitation the rights\nto use, copy, modify, merge, publish, distribute, sublicense, and/or sell\ncopies of the Software, and to permit persons to whom the Software is\nfurnished to do so.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR\nIMPLIED.\n\nArchive v1.1.0: 5 files, 8262 bytes\n\nFiles: LICENSE.txt (533b), references/models.md (1601b), references/prompts.md (3242b), SKILL.md (11457b), _meta.json (132b)\n\nFile v1.1.0:SKILL.md\n\n---\nname: china-doc-ocr\ndescription: 智能文档OCR识别与结构化提取。Use when the user has a complex document, PDF, scanned image, photo, invoice, receipt, ID card, table, or chart that needs to be recognized and converted to text or Markdown. Uses PaddleOCR-VL-1.5 and DeepSeek-OCR (free, fast). Falls back to Qwen2.5-VL if OCR fails. 国内直连，无需翻墙。\nversion: 1.1.0\nlicense: MIT-0\nmetadata: {\"openclaw\": {\"emoji\": \"📄\", \"requires\": {\"bins\": [\"curl\", \"python3\"], \"env\": [\"SILICONFLOW_API_KEY\"]}, \"primaryEnv\": \"SILICONFLOW_API_KEY\"}}\n---\n\n# 智能文档 OCR China Doc OCR\n\n识别并提取复杂文档内容：PDF、图片、扫描件、发票、表格、证件等。\n使用硅基流动 DeepSeek-OCR / PaddleOCR-VL，国内直连，无需翻墙。\n与 china-image-gen、china-tts 共用同一个 SILICONFLOW_API_KEY。\n\n模型选择与参数说明 → `references/models.md`\n各场景提示词模板 → `references/prompts.md`\n\n## 触发时机\n\n- \"帮我识别这个PDF/图片里的内容\"\n- \"把这张发票/收据的信息提取出来\"\n- \"将这份扫描合同转成可编辑文字\"\n- \"这个表格里的数据帮我提取一下\"\n- \"帮我把这张截图的文字识别出来\"\n- \"这份报告转成 Markdown 格式\"\n- \"识别这张身份证/营业执照的信息\"\n\n---\n\n## 模型选择策略（优先OCR）\n\n```\nOCR优先级：\n1. PaddleOCR-VL-1.5 (免费、快速、专业OCR)\n2. DeepSeek-OCR (免费、效果好)\n3. Qwen2.5-VL-72B (视觉语言模型，OCR效果一般但可补充)\n\n默认使用 PaddleOCR-VL-1.5\n如果识别效果不好，降级到 DeepSeek-OCR\n如果仍然不好，降级到 Qwen2.5-VL-72B\n```\n\n---\n\n## Step 0：环境检查\n\n```bash\n# 检查 API Key\nif [ -z \"$SILICONFLOW_API_KEY\" ]; then\n  echo \"❌ 缺少 SILICONFLOW_API_KEY\"\n  echo \"配置方法：\"\n  echo \"  1. 访问 cloud.siliconflow.cn 注册（国内直连）\"\n  echo \"  2. 进入「API密钥」页面创建 Key\"\n  echo \"  3. export SILICONFLOW_API_KEY='sk-xxxxxxxx'\"\n  echo \"  或写入 ~/.openclaw/.env\"\n  exit 1\nfi\n\n# 检查 python3（用于 base64 编码）\nif ! command -v python3 &> /dev/null; then\n  echo \"❌ 需要 python3（用于文件 base64 编码）\"\n  echo \"  macOS:  brew install python3\"\n  echo \"  Ubuntu: sudo apt install python3\"\n  exit 1\nfi\n\necho \"✅ 环境检查通过\"\n```\n\n---\n\n## Step 1：识别内容类型，选择处理模式\n\n```\n用户提供文件路径或 URL → 判断类型：\n\n文件扩展名/用户描述 → 处理模式：\n\n.pdf                    → PDF 模式（见 Step 3）\n.jpg/.jpeg/.png/.webp   → 图片模式（见 Step 2）\n.bmp/.tiff/.gif         → 图片模式（先转换格式）\nURL（http/https开头）   → URL 直接模式（见 Step 2B）\n用户粘贴了 base64       → 直接使用\n\n用户意图 → 选择 Prompt 模式（见 references/prompts.md）：\n\n\"转成文字/提取文字\"     → 通用OCR\n\"转成Markdown/保留格式\" → 文档转Markdown\n\"提取表格/表格数据\"     → 图表解析\n\"发票/收据/单据\"        → 发票识别\n\"身份证/证件/执照\"      → 证件识别\n\"图表/图形/柱状图\"      → 图表解析\n未指定                  → 默认文档转Markdown\n```\n\n---\n\n## Step 2：图片 OCR（本地文件）\n\n### Step 2A：本地图片文件\n\n```bash\nIMAGE_PATH=\"/path/to/image.jpg\"  # 用户提供的图片路径\nPROMPT=\"<image>\\n<|grounding|>Convert the document to markdown.\"  # 见 references/prompts.md\n\n# 将图片编码为 base64\nBASE64_DATA=$(python3 -c \"\nimport base64, sys\nwith open('$IMAGE_PATH', 'rb') as f:\n    data = base64.b64encode(f.read()).decode('utf-8')\nprint(data)\n\")\n\n# 判断图片格式（用于 data URL）\nEXT=\"${IMAGE_PATH##*.}\"\ncase \"$EXT\" in\n  jpg|jpeg) MIME=\"image/jpeg\" ;;\n  png)      MIME=\"image/png\" ;;\n  webp)     MIME=\"image/webp\" ;;\n  bmp)      MIME=\"image/bmp\" ;;\n  *)        MIME=\"image/jpeg\" ;;\nesac\n\n# 调用 OCR 模型（优先 PaddleOCR-VL-1.5）\n# 模型选择：PaddleOCR-VL-1.5 > DeepSeek-OCR > Qwen2.5-VL\nMODEL=\"PaddlePaddle/PaddleOCR-VL-1.5\"  # 默认使用PaddleOCR\n\ncurl -s -X POST \"https://api.siliconflow.cn/v1/chat/completions\" \\\n  -H \"Authorization: Bearer $SILICONFLOW_API_KEY\" \\\n  -H \"Content-Type: application/json\" \\\n  -d \"{\n    \\\"model\\\": \\\"$MODEL\\\",\n    \\\"messages\\\": [\n      {\n        \\\"role\\\": \\\"user\\\",\n        \\\"content\\\": [\n          {\n            \\\"type\\\": \\\"image_url\\\",\n            \\\"image_url\\\": {\n              \\\"url\\\": \\\"data:${MIME};base64,${BASE64_DATA}\\\",\n              \\\"detail\\\": \\\"high\\\"\n            }\n          },\n          {\n            \\\"type\\\": \\\"text\\\",\n            \\\"text\\\": \\\"$PROMPT\\\"\n          }\n        ]\n      }\n    ],\n    \\\"max_tokens\\\": 4096,\n    \\\"stream\\\": false\n  }\" | python3 -c \"\nimport sys, json\ndata = json.load(sys.stdin)\nif 'choices' in data:\n    print(data['choices'][0]['message']['content'])\nelse:\n    print('错误：', json.dumps(data, ensure_ascii=False))\n\"\n```\n\n### Step 2B：图片 URL（无需下载，直接传 URL）\n\n```bash\nIMAGE_URL=\"https://example.com/document.jpg\"\n\ncurl -s -X POST \"https://api.siliconflow.cn/v1/chat/completions\" \\\n  -H \"Authorization: Bearer $SILICONFLOW_API_KEY\" \\\n  -H \"Content-Type: application/json\" \\\n  -d \"{\n    \\\"model\\\": \\\"Pro/deepseek-ai/DeepSeek-V3\\\",\n    \\\"messages\\\": [\n      {\n        \\\"role\\\": \\\"user\\\",\n        \\\"content\\\": [\n          {\n            \\\"type\\\": \\\"image_url\\\",\n            \\\"image_url\\\": {\n              \\\"url\\\": \\\"$IMAGE_URL\\\",\n              \\\"detail\\\": \\\"high\\\"\n            }\n          },\n          {\n            \\\"type\\\": \\\"text\\\",\n            \\\"text\\\": \\\"<image>\\\\n<|grounding|>Convert the document to markdown.\\\"\n          }\n        ]\n      }\n    ],\n    \\\"max_tokens\\\": 4096\n  }\" | python3 -c \"\nimport sys, json\ndata = json.load(sys.stdin)\nprint(data['choices'][0]['message']['content'])\n\"\n```\n\n---\n\n## Step 3：PDF OCR\n\n### 单页或少页 PDF（直接整体处理）\n\n```bash\nPDF_PATH=\"/path/to/document.pdf\"\n\n# PDF 转 base64\nBASE64_PDF=$(python3 -c \"\nimport base64\nwith open('$PDF_PATH', 'rb') as f:\n    print(base64.b64encode(f.read()).decode('utf-8'))\n\")\n\ncurl -s -X POST \"https://api.siliconflow.cn/v1/chat/completions\" \\\n  -H \"Authorization: Bearer $SILICONFLOW_API_KEY\" \\\n  -H \"Content-Type: application/json\" \\\n  -d \"{\n    \\\"model\\\": \\\"Pro/deepseek-ai/DeepSeek-V3\\\",\n    \\\"messages\\\": [\n      {\n        \\\"role\\\": \\\"user\\\",\n        \\\"content\\\": [\n          {\n            \\\"type\\\": \\\"image_url\\\",\n            \\\"image_url\\\": {\n              \\\"url\\\": \\\"data:application/pdf;base64,${BASE64_PDF}\\\",\n              \\\"detail\\\": \\\"high\\\"\n            }\n          },\n          {\n            \\\"type\\\": \\\"text\\\",\n            \\\"text\\\": \\\"<image>\\\\n<|grounding|>Convert the document to markdown.\\\"\n          }\n        ]\n      }\n    ],\n    \\\"max_tokens\\\": 8192\n  }\" | python3 -c \"\nimport sys, json\ndata = json.load(sys.stdin)\nprint(data['choices'][0]['message']['content'])\n\"\n```\n\n### 多页 PDF 分页处理\n\n```bash\nPDF_PATH=\"/path/to/multipage.pdf\"\nOUTPUT_DIR=\"${OPENCLAW_WORKSPACE:-$PWD}/ocr_$(date +%Y%m%d_%H%M%S)\"\nmkdir -p \"$OUTPUT_DIR\"\n\n# 用 python3 分页提取并分别识别\npython3 << PYEOF\nimport base64, json, urllib.request, os, sys\n\npdf_path = \"$PDF_PATH\"\noutput_dir = \"$OUTPUT_DIR\"\napi_key = \"$SILICONFLOW_API_KEY\"\n\n# 尝试用 pypdf 分页\ntry:\n    import pypdf\n    reader = pypdf.PdfReader(pdf_path)\n    total_pages = len(reader.pages)\n    print(f\"PDF 共 {total_pages} 页，开始逐页识别...\")\n\n    all_results = []\n    for i, page in enumerate(reader.pages):\n        # 单页写成临时 PDF\n        writer = pypdf.PdfWriter()\n        writer.add_page(page)\n        tmp_path = f\"{output_dir}/page_{i+1:03d}.pdf\"\n        with open(tmp_path, \"wb\") as f:\n            writer.write(f)\n\n        # base64 编码\n        with open(tmp_path, \"rb\") as f:\n            b64 = base64.b64encode(f.read()).decode(\"utf-8\")\n\n        # 调用 API\n        payload = json.dumps({\n            \"model\": \"Pro/deepseek-ai/DeepSeek-V3\",\n            \"messages\": [{\n                \"role\": \"user\",\n                \"content\": [\n                    {\"type\": \"image_url\", \"image_url\": {\n                        \"url\": f\"data:application/pdf;base64,{b64}\",\n                        \"detail\": \"high\"\n                    }},\n                    {\"type\": \"text\", \"text\": \"<image>\\n<|grounding|>Convert the document to markdown.\"}\n                ]\n            }],\n            \"max_tokens\": 4096\n        }).encode(\"utf-8\")\n\n        req = urllib.request.Request(\n            \"https://api.siliconflow.cn/v1/chat/completions\",\n            data=payload,\n            headers={\"Authorization\": f\"Bearer {api_key}\", \"Content-Type\": \"application/json\"}\n        )\n        with urllib.request.urlopen(req) as resp:\n            result = json.loads(resp.read())\n            content = result[\"choices\"][0][\"message\"][\"content\"]\n            all_results.append(f\"## 第 {i+1} 页\\n\\n{content}\")\n            print(f\"✅ 第 {i+1}/{total_pages} 页识别完成\")\n\n    # 合并输出\n    merged = \"\\n\\n---\\n\\n\".join(all_results)\n    output_path = f\"{output_dir}/result.md\"\n    with open(output_path, \"w\", encoding=\"utf-8\") as f:\n        f.write(merged)\n    print(f\"\\n✅ 全部完成，结果已保存：{output_path}\")\n\nexcept ImportError:\n    print(\"需要安装 pypdf：pip install pypdf\")\n    print(\"安装后重新运行\")\n    sys.exit(1)\nPYEOF\n```\n\n---\n\n## Step 4：格式化输出\n\n识别完成后根据用户需求输出：\n\n### 文档转 Markdown（保留结构）\n\n```\n直接输出 Markdown 内容，保留：\n  - 标题层级（# ## ###）\n  - 列表（- * 1.）\n  - 表格（| 列1 | 列2 |）\n  - 代码块（```）\n  - 加粗、斜体等格式\n```\n\n### 发票/证件识别（结构化输出）\n\n```\n📄 发票识别结果\n━━━━━━━━━━━━━━━━━━━━\n发票类型：增值税专用发票\n发票号码：XXXXXXXXXXXXXXXX\n开票日期：2026年03月21日\n购买方：[公司名称]\n销售方：[公司名称]\n商品/服务：[明细]\n不含税金额：¥X,XXX.XX\n税率：13%\n税额：¥XXX.XX\n价税合计：¥X,XXX.XX\n```\n\n### 表格数据（CSV 友好格式）\n\n```\n识别结果同时输出：\n1. Markdown 表格（可读）\n2. 询问用户是否需要 CSV 格式（方便导入 Excel）\n```\n\n---\n\n## 输出文件保存\n\n```bash\nOUTPUT_DIR=\"${OPENCLAW_WORKSPACE:-$PWD}/ocr_$(date +%Y%m%d_%H%M%S)\"\nmkdir -p \"$OUTPUT_DIR\"\n\n# 保存 Markdown 结果\ncat > \"$OUTPUT_DIR/result.md\" << 'RESULT'\n{OCR识别内容}\nRESULT\n\necho \"✅ OCR 识别完成\"\necho \"结果已保存：$OUTPUT_DIR/result.md\"\n```\n\n---\n\n## 错误处理\n\n```\n文件不存在           → 提示用户确认路径\n文件过大（>10MB）    → 建议压缩或分页处理\n图片分辨率过低       → 提示识别效果可能较差，建议重新拍摄\nPDF 加密            → 提示需要先解密：qpdf --decrypt input.pdf output.pdf\n识别结果为空         → 可能是纯图片型PDF，尝试截图后重新识别\n401 错误            → API Key 失效，重新获取\n429 错误            → 请求频率超限，等待后重试\n```\n\n---\n\n## 注意事项\n\n- 图片最小 56×56，最大 3584×3584 像素，超出会自动压缩\n- PDF 支持 base64 编码输入，DeepSeek-OCR 同时支持 PDF URL\n- 多页 PDF 需要安装 pypdf：`pip install pypdf`\n- 识别结果保存到工作区，长期保留\n- detail=high 时按实际像素计费，detail=low 统一约256 token，复杂文档建议用 high\n- 发票/证件等隐私文件处理后请及时删除工作区临时文件\n\nFile v1.1.0:_meta.json\n\n{\n  \"ownerId\": \"kn75z6gevjsyrznm7dg2ez6sen82h8sz\",\n  \"slug\": \"china-doc-ocr\",\n  \"version\": \"1.1.0\",\n  \"publishedAt\": 1774486668370\n}\n\nFile v1.1.0:references/models.md\n\n# 模型选择说明\n\n来源：硅基流动官方文档\n\n## 主力模型：Pro/deepseek-ai/DeepSeek-V3\n\n```\n模型名：Pro/deepseek-ai/DeepSeek-V3\n特点：\n  - 支持图片和 PDF 输入（base64 或 URL）\n  - 中文文档识别准确率极高\n  - 支持专用 OCR prompt 格式（<image>\\n<|grounding|>...）\n  - 支持多图对比分析\n适用：所有文档类型的首选模型\n```\n\n## 备用模型：Qwen2.5-VL-72B\n\n```\n模型名：Qwen/Qwen2.5-VL-72B-Instruct\n特点：\n  - 超大参数量，复杂文档理解力强\n  - 支持图片输入（不支持 PDF base64）\n  - 中英文双语文档效果好\n适用：DeepSeek-OCR 效果不佳时的备选\n```\n\n## PaddleOCR-VL（专业 OCR 场景）\n\n```\n模型名：PaddlePaddle/PaddleOCR-VL\n特点：\n  - 专为 OCR 任务优化\n  - 支持 CLI 和 API 两种调用方式\n  - 对复杂版面（多列、表格）效果好\n适用：需要精确版面还原的场景\n```\n\n## 图像输入计费（detail 参数影响）\n\n```\ndetail=low：\n  统一压缩为 448×448，约 256 token\n  适合：文字较大、布局简单的文档\n\ndetail=high（推荐）：\n  按实际像素计费：ceil(h/28) × ceil(w/28) token\n  适合：字体较小、布局复杂、表格密集的文档\n\n建议默认使用 detail=high，确保识别准确率\n```\n\n## 模型选择决策\n\n```\n发票/收据/证件        → DeepSeek-V3（中文场景最佳）\n普通文档/报告         → DeepSeek-V3\n复杂表格/多列版面     → DeepSeek-V3 或 PaddleOCR-VL\n英文为主的文档        → Qwen2.5-VL-72B\nDeepSeek 结果不满意  → 改用 Qwen2.5-VL-72B 重试\n```\n\nFile v1.1.0:references/prompts.md\n\n# OCR 提示词模板\n\n来源：硅基流动官方文档 DeepSeek-OCR 专用 prompt 格式\n\n## 使用方式\n\n所有 prompt 在 text 字段中传入，格式固定为：\n`<image>\\n<|grounding|>{具体指令}`\n\n---\n\n## 1. 文档转 Markdown（默认推荐）\n\n```\n<image>\n<|grounding|>Convert the document to markdown.\n```\n\n输出：保留标题层级、列表、表格、加粗等所有格式\n适用：报告、合同、论文、书籍页面、网页截图\n\n---\n\n## 2. 通用 OCR（纯文字提取）\n\n```\n<image>\n<|grounding|>OCR this image.\n```\n\n输出：按阅读顺序提取所有文字，保留基本换行\n适用：简单图片、截图文字、无复杂格式的文档\n\n---\n\n## 3. 无布局 OCR（去除所有格式）\n\n```\n<image>\nFree OCR.\n```\n\n注意：这个格式不加 `<|grounding|>` 前缀\n输出：纯文字流，不保留任何格式信息\n适用：只需要文字内容、不关心排版的场景\n\n---\n\n## 4. 图表/表格解析\n\n```\n<image>\n<|grounding|>Parse the figure.\n```\n\n输出：结构化描述图表内容，表格转为 Markdown 格式\n适用：柱状图、折线图、饼图、数据表格、流程图\n\n---\n\n## 5. 图片详细描述\n\n```\n<image>\n<|grounding|>Describe this image in detail.\n```\n\n输出：详细描述图片内容，包括文字、图形、布局\n适用：混合图文内容，需要全面理解图片的场景\n\n---\n\n## 6. 文字定位\n\n```\n<image>\n<|grounding|>Locate <|ref|>要定位的文字<|/ref|> in the image.\n```\n\n输出：指出特定文字在图片中的位置\n适用：需要找到特定字段位置的场景\n\n---\n\n## 场景专用 Prompt\n\n### 发票识别\n\n```\n请识别这张发票的所有信息，以结构化格式输出：\n发票类型、发票号码、开票日期、\n购买方信息（名称/税号/地址）、\n销售方信息（名称/税号/地址）、\n商品明细（名称/数量/单价/金额）、\n税率、税额、价税合计、备注。\n如有字段无法识别，标注\"不清晰\"。\n```\n\n### 身份证识别\n\n```\n请识别这张身份证的信息：\n姓名、性别、民族、出生日期、住址、公民身份号码。\n输出结构化格式，隐私字段用*号部分隐藏。\n```\n\n### 营业执照识别\n\n```\n请识别这份营业执照的信息：\n公司名称、统一社会信用代码、类型、法定代表人、\n注册资本、成立日期、营业期限、经营范围、注册地址。\n```\n\n### 银行流水/对账单\n\n```\n请识别这份银行流水的所有交易记录，\n以表格格式输出：日期、摘要、支出、收入、余额。\n如有多页请按时间顺序排列。\n```\n\n### 合同关键信息提取\n\n```\n请识别这份合同的关键信息：\n合同编号、签订日期、甲方、乙方、\n合同金额、付款方式、合同期限、\n主要条款摘要（不超过200字）、\n双方签字/盖章情况。\n```\n\n### 学术论文结构化\n\n```\n<image>\n<|grounding|>Convert the document to markdown.\n请保留所有标题层级、公式（用LaTeX格式）、\n参考文献编号和表格结构。\n```\n\n### 表格数据提取（输出 CSV）\n\n```\n请识别这个表格中的所有数据，\n以 CSV 格式输出（逗号分隔），\n第一行为表头，之后每行为一条数据。\n如有合并单元格，请拆分填充。\n```\n\nFile v1.1.0:LICENSE.txt\n\nMIT No Attribution (MIT-0)\n\nCopyright 2026 ToBeWin\n\nPermission is hereby granted, free of charge, to any person obtaining a copy\nof this software and associated documentation files (the \"Software\"), to deal\nin the Software without restriction, including without limitation the rights\nto use, copy, modify, merge, publish, distribute, sublicense, and/or sell\ncopies of the Software, and to permit persons to whom the Software is\nfurnished to do so.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR\nIMPLIED.\n\nArchive v1.0.1: 5 files, 8113 bytes\n\nFiles: LICENSE.txt (533b), references/models.md (1601b), references/prompts.md (3242b), SKILL.md (11093b), _meta.json (132b)\n\nFile v1.0.1:SKILL.md\n\n---\nname: china-doc-ocr\ndescription: 智能文档OCR识别与结构化提取。Use when the user has a complex document, PDF, scanned image, photo, invoice, receipt, ID card, table, or chart that needs to be recognized and converted to text or Markdown. Handles complex layouts, multi-column text, tables, and mixed image-text that OpenClaw cannot process natively. Uses SiliconFlow DeepSeek-OCR and PaddleOCR-VL models — domestic access, no VPN, same API key as china-image-gen and china-tts.\nversion: 1.0.0\nlicense: MIT-0\nmetadata: {\"openclaw\": {\"emoji\": \"📄\", \"requires\": {\"bins\": [\"curl\", \"python3\"], \"env\": [\"SILICONFLOW_API_KEY\"]}, \"primaryEnv\": \"SILICONFLOW_API_KEY\"}}\n---\n\n# 智能文档 OCR China Doc OCR\n\n识别并提取复杂文档内容：PDF、图片、扫描件、发票、表格、证件等。\n使用硅基流动 DeepSeek-OCR / PaddleOCR-VL，国内直连，无需翻墙。\n与 china-image-gen、china-tts 共用同一个 SILICONFLOW_API_KEY。\n\n模型选择与参数说明 → `references/models.md`\n各场景提示词模板 → `references/prompts.md`\n\n## 触发时机\n\n- \"帮我识别这个PDF/图片里的内容\"\n- \"把这张发票/收据的信息提取出来\"\n- \"将这份扫描合同转成可编辑文字\"\n- \"这个表格里的数据帮我提取一下\"\n- \"帮我把这张截图的文字识别出来\"\n- \"这份报告转成 Markdown 格式\"\n- \"识别这张身份证/营业执照的信息\"\n\n---\n\n## Step 0：环境检查\n\n```bash\n# 检查 API Key\nif [ -z \"$SILICONFLOW_API_KEY\" ]; then\n  echo \"❌ 缺少 SILICONFLOW_API_KEY\"\n  echo \"配置方法：\"\n  echo \"  1. 访问 cloud.siliconflow.cn 注册（国内直连）\"\n  echo \"  2. 进入「API密钥」页面创建 Key\"\n  echo \"  3. export SILICONFLOW_API_KEY='sk-xxxxxxxx'\"\n  echo \"  或写入 ~/.openclaw/.env\"\n  exit 1\nfi\n\n# 检查 python3（用于 base64 编码）\nif ! command -v python3 &> /dev/null; then\n  echo \"❌ 需要 python3（用于文件 base64 编码）\"\n  echo \"  macOS:  brew install python3\"\n  echo \"  Ubuntu: sudo apt install python3\"\n  exit 1\nfi\n\necho \"✅ 环境检查通过\"\n```\n\n---\n\n## Step 1：识别内容类型，选择处理模式\n\n```\n用户提供文件路径或 URL → 判断类型：\n\n文件扩展名/用户描述 → 处理模式：\n\n.pdf                    → PDF 模式（见 Step 3）\n.jpg/.jpeg/.png/.webp   → 图片模式（见 Step 2）\n.bmp/.tiff/.gif         → 图片模式（先转换格式）\nURL（http/https开头）   → URL 直接模式（见 Step 2B）\n用户粘贴了 base64       → 直接使用\n\n用户意图 → 选择 Prompt 模式（见 references/prompts.md）：\n\n\"转成文字/提取文字\"     → 通用OCR\n\"转成Markdown/保留格式\" → 文档转Markdown\n\"提取表格/表格数据\"     → 图表解析\n\"发票/收据/单据\"        → 发票识别\n\"身份证/证件/执照\"      → 证件识别\n\"图表/图形/柱状图\"      → 图表解析\n未指定                  → 默认文档转Markdown\n```\n\n---\n\n## Step 2：图片 OCR（本地文件）\n\n### Step 2A：本地图片文件\n\n```bash\nIMAGE_PATH=\"/path/to/image.jpg\"  # 用户提供的图片路径\nPROMPT=\"<image>\\n<|grounding|>Convert the document to markdown.\"  # 见 references/prompts.md\n\n# 将图片编码为 base64\nBASE64_DATA=$(python3 -c \"\nimport base64, sys\nwith open('$IMAGE_PATH', 'rb') as f:\n    data = base64.b64encode(f.read()).decode('utf-8')\nprint(data)\n\")\n\n# 判断图片格式（用于 data URL）\nEXT=\"${IMAGE_PATH##*.}\"\ncase \"$EXT\" in\n  jpg|jpeg) MIME=\"image/jpeg\" ;;\n  png)      MIME=\"image/png\" ;;\n  webp)     MIME=\"image/webp\" ;;\n  bmp)      MIME=\"image/bmp\" ;;\n  *)        MIME=\"image/jpeg\" ;;\nesac\n\n# 调用 DeepSeek-OCR\ncurl -s -X POST \"https://api.siliconflow.cn/v1/chat/completions\" \\\n  -H \"Authorization: Bearer $SILICONFLOW_API_KEY\" \\\n  -H \"Content-Type: application/json\" \\\n  -d \"{\n    \\\"model\\\": \\\"Pro/deepseek-ai/DeepSeek-V3\\\",\n    \\\"messages\\\": [\n      {\n        \\\"role\\\": \\\"user\\\",\n        \\\"content\\\": [\n          {\n            \\\"type\\\": \\\"image_url\\\",\n            \\\"image_url\\\": {\n              \\\"url\\\": \\\"data:${MIME};base64,${BASE64_DATA}\\\",\n              \\\"detail\\\": \\\"high\\\"\n            }\n          },\n          {\n            \\\"type\\\": \\\"text\\\",\n            \\\"text\\\": \\\"$PROMPT\\\"\n          }\n        ]\n      }\n    ],\n    \\\"max_tokens\\\": 4096,\n    \\\"stream\\\": false\n  }\" | python3 -c \"\nimport sys, json\ndata = json.load(sys.stdin)\nif 'choices' in data:\n    print(data['choices'][0]['message']['content'])\nelse:\n    print('错误：', json.dumps(data, ensure_ascii=False))\n\"\n```\n\n### Step 2B：图片 URL（无需下载，直接传 URL）\n\n```bash\nIMAGE_URL=\"https://example.com/document.jpg\"\n\ncurl -s -X POST \"https://api.siliconflow.cn/v1/chat/completions\" \\\n  -H \"Authorization: Bearer $SILICONFLOW_API_KEY\" \\\n  -H \"Content-Type: application/json\" \\\n  -d \"{\n    \\\"model\\\": \\\"Pro/deepseek-ai/DeepSeek-V3\\\",\n    \\\"messages\\\": [\n      {\n        \\\"role\\\": \\\"user\\\",\n        \\\"content\\\": [\n          {\n            \\\"type\\\": \\\"image_url\\\",\n            \\\"image_url\\\": {\n              \\\"url\\\": \\\"$IMAGE_URL\\\",\n              \\\"detail\\\": \\\"high\\\"\n            }\n          },\n          {\n            \\\"type\\\": \\\"text\\\",\n            \\\"text\\\": \\\"<image>\\\\n<|grounding|>Convert the document to markdown.\\\"\n          }\n        ]\n      }\n    ],\n    \\\"max_tokens\\\": 4096\n  }\" | python3 -c \"\nimport sys, json\ndata = json.load(sys.stdin)\nprint(data['choices'][0]['message']['content'])\n\"\n```\n\n---\n\n## Step 3：PDF OCR\n\n### 单页或少页 PDF（直接整体处理）\n\n```bash\nPDF_PATH=\"/path/to/document.pdf\"\n\n# PDF 转 base64\nBASE64_PDF=$(python3 -c \"\nimport base64\nwith open('$PDF_PATH', 'rb') as f:\n    print(base64.b64encode(f.read()).decode('utf-8'))\n\")\n\ncurl -s -X POST \"https://api.siliconflow.cn/v1/chat/completions\" \\\n  -H \"Authorization: Bearer $SILICONFLOW_API_KEY\" \\\n  -H \"Content-Type: application/json\" \\\n  -d \"{\n    \\\"model\\\": \\\"Pro/deepseek-ai/DeepSeek-V3\\\",\n    \\\"messages\\\": [\n      {\n        \\\"role\\\": \\\"user\\\",\n        \\\"content\\\": [\n          {\n            \\\"type\\\": \\\"image_url\\\",\n            \\\"image_url\\\": {\n              \\\"url\\\": \\\"data:application/pdf;base64,${BASE64_PDF}\\\",\n              \\\"detail\\\": \\\"high\\\"\n            }\n          },\n          {\n            \\\"type\\\": \\\"text\\\",\n            \\\"text\\\": \\\"<image>\\\\n<|grounding|>Convert the document to markdown.\\\"\n          }\n        ]\n      }\n    ],\n    \\\"max_tokens\\\": 8192\n  }\" | python3 -c \"\nimport sys, json\ndata = json.load(sys.stdin)\nprint(data['choices'][0]['message']['content'])\n\"\n```\n\n### 多页 PDF 分页处理\n\n```bash\nPDF_PATH=\"/path/to/multipage.pdf\"\nOUTPUT_DIR=\"${OPENCLAW_WORKSPACE:-$PWD}/ocr_$(date +%Y%m%d_%H%M%S)\"\nmkdir -p \"$OUTPUT_DIR\"\n\n# 用 python3 分页提取并分别识别\npython3 << PYEOF\nimport base64, json, urllib.request, os, sys\n\npdf_path = \"$PDF_PATH\"\noutput_dir = \"$OUTPUT_DIR\"\napi_key = \"$SILICONFLOW_API_KEY\"\n\n# 尝试用 pypdf 分页\ntry:\n    import pypdf\n    reader = pypdf.PdfReader(pdf_path)\n    total_pages = len(reader.pages)\n    print(f\"PDF 共 {total_pages} 页，开始逐页识别...\")\n\n    all_results = []\n    for i, page in enumerate(reader.pages):\n        # 单页写成临时 PDF\n        writer = pypdf.PdfWriter()\n        writer.add_page(page)\n        tmp_path = f\"{output_dir}/page_{i+1:03d}.pdf\"\n        with open(tmp_path, \"wb\") as f:\n            writer.write(f)\n\n        # base64 编码\n        with open(tmp_path, \"rb\") as f:\n            b64 = base64.b64encode(f.read()).decode(\"utf-8\")\n\n        # 调用 API\n        payload = json.dumps({\n            \"model\": \"Pro/deepseek-ai/DeepSeek-V3\",\n            \"messages\": [{\n                \"role\": \"user\",\n                \"content\": [\n                    {\"type\": \"image_url\", \"image_url\": {\n                        \"url\": f\"data:application/pdf;base64,{b64}\",\n                        \"detail\": \"high\"\n                    }},\n                    {\"type\": \"text\", \"text\": \"<image>\\n<|grounding|>Convert the document to markdown.\"}\n                ]\n            }],\n            \"max_tokens\": 4096\n        }).encode(\"utf-8\")\n\n        req = urllib.request.Request(\n            \"https://api.siliconflow.cn/v1/chat/completions\",\n            data=payload,\n            headers={\"Authorization\": f\"Bearer {api_key}\", \"Content-Type\": \"application/json\"}\n        )\n        with urllib.request.urlopen(req) as resp:\n            result = json.loads(resp.read())\n            content = result[\"choices\"][0][\"message\"][\"content\"]\n            all_results.append(f\"## 第 {i+1} 页\\n\\n{content}\")\n            print(f\"✅ 第 {i+1}/{total_pages} 页识别完成\")\n\n    # 合并输出\n    merged = \"\\n\\n---\\n\\n\".join(all_results)\n    output_path = f\"{output_dir}/result.md\"\n    with open(output_path, \"w\", encoding=\"utf-8\") as f:\n        f.write(merged)\n    print(f\"\\n✅ 全部完成，结果已保存：{output_path}\")\n\nexcept ImportError:\n    print(\"需要安装 pypdf：pip install pypdf\")\n    print(\"安装后重新运行\")\n    sys.exit(1)\nPYEOF\n```\n\n---\n\n## Step 4：格式化输出\n\n识别完成后根据用户需求输出：\n\n### 文档转 Markdown（保留结构）\n\n```\n直接输出 Markdown 内容，保留：\n  - 标题层级（# ## ###）\n  - 列表（- * 1.）\n  - 表格（| 列1 | 列2 |）\n  - 代码块（```）\n  - 加粗、斜体等格式\n```\n\n### 发票/证件识别（结构化输出）\n\n```\n📄 发票识别结果\n━━━━━━━━━━━━━━━━━━━━\n发票类型：增值税专用发票\n发票号码：XXXXXXXXXXXXXXXX\n开票日期：2026年03月21日\n购买方：[公司名称]\n销售方：[公司名称]\n商品/服务：[明细]\n不含税金额：¥X,XXX.XX\n税率：13%\n税额：¥XXX.XX\n价税合计：¥X,XXX.XX\n```\n\n### 表格数据（CSV 友好格式）\n\n```\n识别结果同时输出：\n1. Markdown 表格（可读）\n2. 询问用户是否需要 CSV 格式（方便导入 Excel）\n```\n\n---\n\n## 输出文件保存\n\n```bash\nOUTPUT_DIR=\"${OPENCLAW_WORKSPACE:-$PWD}/ocr_$(date +%Y%m%d_%H%M%S)\"\nmkdir -p \"$OUTPUT_DIR\"\n\n# 保存 Markdown 结果\ncat > \"$OUTPUT_DIR/result.md\" << 'RESULT'\n{OCR识别内容}\nRESULT\n\necho \"✅ OCR 识别完成\"\necho \"结果已保存：$OUTPUT_DIR/result.md\"\n```\n\n---\n\n## 错误处理\n\n```\n文件不存在           → 提示用户确认路径\n文件过大（>10MB）    → 建议压缩或分页处理\n图片分辨率过低       → 提示识别效果可能较差，建议重新拍摄\nPDF 加密            → 提示需要先解密：qpdf --decrypt input.pdf output.pdf\n识别结果为空         → 可能是纯图片型PDF，尝试截图后重新识别\n401 错误            → API Key 失效，重新获取\n429 错误            → 请求频率超限，等待后重试\n```\n\n---\n\n## 注意事项\n\n- 图片最小 56×56，最大 3584×3584 像素，超出会自动压缩\n- PDF 支持 base64 编码输入，DeepSeek-OCR 同时支持 PDF URL\n- 多页 PDF 需要安装 pypdf：`pip install pypdf`\n- 识别结果保存到工作区，长期保留\n- detail=high 时按实际像素计费，detail=low 统一约256 token，复杂文档建议用 high\n- 发票/证件等隐私文件处理后请及时删除工作区临时文件\n\nFile v1.0.1:_meta.json\n\n{\n  \"ownerId\": \"kn75z6gevjsyrznm7dg2ez6sen82h8sz\",\n  \"slug\": \"china-doc-ocr\",\n  \"version\": \"1.0.1\",\n  \"publishedAt\": 1774244843681\n}\n\nFile v1.0.1:references/models.md\n\n# 模型选择说明\n\n来源：硅基流动官方文档\n\n## 主力模型：Pro/deepseek-ai/DeepSeek-V3\n\n```\n模型名：Pro/deepseek-ai/DeepSeek-V3\n特点：\n  - 支持图片和 PDF 输入（base64 或 URL）\n  - 中文文档识别准确率极高\n  - 支持专用 OCR prompt 格式（<image>\\n<|grounding|>...）\n  - 支持多图对比分析\n适用：所有文档类型的首选模型\n```\n\n## 备用模型：Qwen2.5-VL-72B\n\n```\n模型名：Qwen/Qwen2.5-VL-72B-Instruct\n特点：\n  - 超大参数量，复杂文档理解力强\n  - 支持图片输入（不支持 PDF base64）\n  - 中英文双语文档效果好\n适用：DeepSeek-OCR 效果不佳时的备选\n```\n\n## PaddleOCR-VL（专业 OCR 场景）\n\n```\n模型名：PaddlePaddle/PaddleOCR-VL\n特点：\n  - 专为 OCR 任务优化\n  - 支持 CLI 和 API 两种调用方式\n  - 对复杂版面（多列、表格）效果好\n适用：需要精确版面还原的场景\n```\n\n## 图像输入计费（detail 参数影响）\n\n```\ndetail=low：\n  统一压缩为 448×448，约 256 token\n  适合：文字较大、布局简单的文档\n\ndetail=high（推荐）：\n  按实际像素计费：ceil(h/28) × ceil(w/28) token\n  适合：字体较小、布局复杂、表格密集的文档\n\n建议默认使用 detail=high，确保识别准确率\n```\n\n## 模型选择决策\n\n```\n发票/收据/证件        → DeepSeek-V3（中文场景最佳）\n普通文档/报告         → DeepSeek-V3\n复杂表格/多列版面     → DeepSeek-V3 或 PaddleOCR-VL\n英文为主的文档        → Qwen2.5-VL-72B\nDeepSeek 结果不满意  → 改用 Qwen2.5-VL-72B 重试\n```\n\nFile v1.0.1:references/prompts.md\n\n# OCR 提示词模板\n\n来源：硅基流动官方文档 DeepSeek-OCR 专用 prompt 格式\n\n## 使用方式\n\n所有 prompt 在 text 字段中传入，格式固定为：\n`<image>\\n<|grounding|>{具体指令}`\n\n---\n\n## 1. 文档转 Markdown（默认推荐）\n\n```\n<image>\n<|grounding|>Convert the document to markdown.\n```\n\n输出：保留标题层级、列表、表格、加粗等所有格式\n适用：报告、合同、论文、书籍页面、网页截图\n\n---\n\n## 2. 通用 OCR（纯文字提取）\n\n```\n<image>\n<|grounding|>OCR this image.\n```\n\n输出：按阅读顺序提取所有文字，保留基本换行\n适用：简单图片、截图文字、无复杂格式的文档\n\n---\n\n## 3. 无布局 OCR（去除所有格式）\n\n```\n<image>\nFree OCR.\n```\n\n注意：这个格式不加 `<|grounding|>` 前缀\n输出：纯文字流，不保留任何格式信息\n适用：只需要文字内容、不关心排版的场景\n\n---\n\n## 4. 图表/表格解析\n\n```\n<image>\n<|grounding|>Parse the figure.\n```\n\n输出：结构化描述图表内容，表格转为 Markdown 格式\n适用：柱状图、折线图、饼图、数据表格、流程图\n\n---\n\n## 5. 图片详细描述\n\n```\n<image>\n<|grounding|>Describe this image in detail.\n```\n\n输出：详细描述图片内容，包括文字、图形、布局\n适用：混合图文内容，需要全面理解图片的场景\n\n---\n\n## 6. 文字定位\n\n```\n<image>\n<|grounding|>Locate <|ref|>要定位的文字<|/ref|> in the image.\n```\n\n输出：指出特定文字在图片中的位置\n适用：需要找到特定字段位置的场景\n\n---\n\n## 场景专用 Prompt\n\n### 发票识别\n\n```\n请识别这张发票的所有信息，以结构化格式输出：\n发票类型、发票号码、开票日期、\n购买方信息（名称/税号/地址）、\n销售方信息（名称/税号/地址）、\n商品明细（名称/数量/单价/金额）、\n税率、税额、价税合计、备注。\n如有字段无法识别，标注\"不清晰\"。\n```\n\n### 身份证识别\n\n```\n请识别这张身份证的信息：\n姓名、性别、民族、出生日期、住址、公民身份号码。\n输出结构化格式，隐私字段用*号部分隐藏。\n```\n\n### 营业执照识别\n\n```\n请识别这份营业执照的信息：\n公司名称、统一社会信用代码、类型、法定代表人、\n注册资本、成立日期、营业期限、经营范围、注册地址。\n```\n\n### 银行流水/对账单\n\n```\n请识别这份银行流水的所有交易记录，\n以表格格式输出：日期、摘要、支出、收入、余额。\n如有多页请按时间顺序排列。\n```\n\n### 合同关键信息提取\n\n```\n请识别这份合同的关键信息：\n合同编号、签订日期、甲方、乙方、\n合同金额、付款方式、合同期限、\n主要条款摘要（不超过200字）、\n双方签字/盖章情况。\n```\n\n### 学术论文结构化\n\n```\n<image>\n<|grounding|>Convert the document to markdown.\n请保留所有标题层级、公式（用LaTeX格式）、\n参考文献编号和表格结构。\n```\n\n### 表格数据提取（输出 CSV）\n\n```\n请识别这个表格中的所有数据，\n以 CSV 格式输出（逗号分隔），\n第一行为表头，之后每行为一条数据。\n如有合并单元格，请拆分填充。\n```\n\nFile v1.0.1:LICENSE.txt\n\nMIT No Attribution (MIT-0)\n\nCopyright 2026 ToBeWin\n\nPermission is hereby granted, free of charge, to any person obtaining a copy\nof this software and associated documentation files (the \"Software\"), to deal\nin the Software without restriction, including without limitation the rights\nto use, copy, modify, merge, publish, distribute, sublicense, and/or sell\ncopies of the Software, and to permit persons to whom the Software is\nfurnished to do so.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR\nIMPLIED.\n\nArchive v1.0.0: 5 files, 8115 bytes\n\nFiles: LICENSE.txt (533b), references/models.md (1601b), references/prompts.md (3242b), SKILL.md (11186b), _meta.json (132b)\n\nFile v1.0.0:SKILL.md\n\n---\nname: china-doc-ocr\ndescription: 智能文档OCR识别与结构化提取。Use when the user has a complex document, PDF, scanned image, photo, invoice, receipt, ID card, table, or chart that needs to be recognized and converted to text or Markdown. Handles complex layouts, multi-column text, tables, and mixed image-text that OpenClaw cannot process natively. Uses SiliconFlow DeepSeek-OCR and PaddleOCR-VL models — domestic access, no VPN, same API key as china-image-gen and china-tts.\nversion: 1.0.0\nlicense: MIT-0\nmetadata:\n  openclaw:\n    emoji: \"📄\"\n    requires:\n      bins:\n        - curl\n        - python3\n    requires_env:\n      - name: SILICONFLOW_API_KEY\n        description: 硅基流动 API Key，与 china-image-gen、china-tts 共用同一个 Key\n---\n\n# 智能文档 OCR China Doc OCR\n\n识别并提取复杂文档内容：PDF、图片、扫描件、发票、表格、证件等。\n使用硅基流动 DeepSeek-OCR / PaddleOCR-VL，国内直连，无需翻墙。\n与 china-image-gen、china-tts 共用同一个 SILICONFLOW_API_KEY。\n\n模型选择与参数说明 → `references/models.md`\n各场景提示词模板 → `references/prompts.md`\n\n## 触发时机\n\n- \"帮我识别这个PDF/图片里的内容\"\n- \"把这张发票/收据的信息提取出来\"\n- \"将这份扫描合同转成可编辑文字\"\n- \"这个表格里的数据帮我提取一下\"\n- \"帮我把这张截图的文字识别出来\"\n- \"这份报告转成 Markdown 格式\"\n- \"识别这张身份证/营业执照的信息\"\n\n---\n\n## Step 0：环境检查\n\n```bash\n# 检查 API Key\nif [ -z \"$SILICONFLOW_API_KEY\" ]; then\n  echo \"❌ 缺少 SILICONFLOW_API_KEY\"\n  echo \"配置方法：\"\n  echo \"  1. 访问 cloud.siliconflow.cn 注册（国内直连）\"\n  echo \"  2. 进入「API密钥」页面创建 Key\"\n  echo \"  3. export SILICONFLOW_API_KEY='sk-xxxxxxxx'\"\n  echo \"  或写入 ~/.openclaw/.env\"\n  exit 1\nfi\n\n# 检查 python3（用于 base64 编码）\nif ! command -v python3 &> /dev/null; then\n  echo \"❌ 需要 python3（用于文件 base64 编码）\"\n  echo \"  macOS:  brew install python3\"\n  echo \"  Ubuntu: sudo apt install python3\"\n  exit 1\nfi\n\necho \"✅ 环境检查通过\"\n```\n\n---\n\n## Step 1：识别内容类型，选择处理模式\n\n```\n用户提供文件路径或 URL → 判断类型：\n\n文件扩展名/用户描述 → 处理模式：\n\n.pdf                    → PDF 模式（见 Step 3）\n.jpg/.jpeg/.png/.webp   → 图片模式（见 Step 2）\n.bmp/.tiff/.gif         → 图片模式（先转换格式）\nURL（http/https开头）   → URL 直接模式（见 Step 2B）\n用户粘贴了 base64       → 直接使用\n\n用户意图 → 选择 Prompt 模式（见 references/prompts.md）：\n\n\"转成文字/提取文字\"     → 通用OCR\n\"转成Markdown/保留格式\" → 文档转Markdown\n\"提取表格/表格数据\"     → 图表解析\n\"发票/收据/单据\"        → 发票识别\n\"身份证/证件/执照\"      → 证件识别\n\"图表/图形/柱状图\"      → 图表解析\n未指定                  → 默认文档转Markdown\n```\n\n---\n\n## Step 2：图片 OCR（本地文件）\n\n### Step 2A：本地图片文件\n\n```bash\nIMAGE_PATH=\"/path/to/image.jpg\"  # 用户提供的图片路径\nPROMPT=\"<image>\\n<|grounding|>Convert the document to markdown.\"  # 见 references/prompts.md\n\n# 将图片编码为 base64\nBASE64_DATA=$(python3 -c \"\nimport base64, sys\nwith open('$IMAGE_PATH', 'rb') as f:\n    data = base64.b64encode(f.read()).decode('utf-8')\nprint(data)\n\")\n\n# 判断图片格式（用于 data URL）\nEXT=\"${IMAGE_PATH##*.}\"\ncase \"$EXT\" in\n  jpg|jpeg) MIME=\"image/jpeg\" ;;\n  png)      MIME=\"image/png\" ;;\n  webp)     MIME=\"image/webp\" ;;\n  bmp)      MIME=\"image/bmp\" ;;\n  *)        MIME=\"image/jpeg\" ;;\nesac\n\n# 调用 DeepSeek-OCR\ncurl -s -X POST \"https://api.siliconflow.cn/v1/chat/completions\" \\\n  -H \"Authorization: Bearer $SILICONFLOW_API_KEY\" \\\n  -H \"Content-Type: application/json\" \\\n  -d \"{\n    \\\"model\\\": \\\"Pro/deepseek-ai/DeepSeek-V3\\\",\n    \\\"messages\\\": [\n      {\n        \\\"role\\\": \\\"user\\\",\n        \\\"content\\\": [\n          {\n            \\\"type\\\": \\\"image_url\\\",\n            \\\"image_url\\\": {\n              \\\"url\\\": \\\"data:${MIME};base64,${BASE64_DATA}\\\",\n              \\\"detail\\\": \\\"high\\\"\n            }\n          },\n          {\n            \\\"type\\\": \\\"text\\\",\n            \\\"text\\\": \\\"$PROMPT\\\"\n          }\n        ]\n      }\n    ],\n    \\\"max_tokens\\\": 4096,\n    \\\"stream\\\": false\n  }\" | python3 -c \"\nimport sys, json\ndata = json.load(sys.stdin)\nif 'choices' in data:\n    print(data['choices'][0]['message']['content'])\nelse:\n    print('错误：', json.dumps(data, ensure_ascii=False))\n\"\n```\n\n### Step 2B：图片 URL（无需下载，直接传 URL）\n\n```bash\nIMAGE_URL=\"https://example.com/document.jpg\"\n\ncurl -s -X POST \"https://api.siliconflow.cn/v1/chat/completions\" \\\n  -H \"Authorization: Bearer $SILICONFLOW_API_KEY\" \\\n  -H \"Content-Type: application/json\" \\\n  -d \"{\n    \\\"model\\\": \\\"Pro/deepseek-ai/DeepSeek-V3\\\",\n    \\\"messages\\\": [\n      {\n        \\\"role\\\": \\\"user\\\",\n        \\\"content\\\": [\n          {\n            \\\"type\\\": \\\"image_url\\\",\n            \\\"image_url\\\": {\n              \\\"url\\\": \\\"$IMAGE_URL\\\",\n              \\\"detail\\\": \\\"high\\\"\n            }\n          },\n          {\n            \\\"type\\\": \\\"text\\\",\n            \\\"text\\\": \\\"<image>\\\\n<|grounding|>Convert the document to markdown.\\\"\n          }\n        ]\n      }\n    ],\n    \\\"max_tokens\\\": 4096\n  }\" | python3 -c \"\nimport sys, json\ndata = json.load(sys.stdin)\nprint(data['choices'][0]['message']['content'])\n\"\n```\n\n---\n\n## Step 3：PDF OCR\n\n### 单页或少页 PDF（直接整体处理）\n\n```bash\nPDF_PATH=\"/path/to/document.pdf\"\n\n# PDF 转 base64\nBASE64_PDF=$(python3 -c \"\nimport base64\nwith open('$PDF_PATH', 'rb') as f:\n    print(base64.b64encode(f.read()).decode('utf-8'))\n\")\n\ncurl -s -X POST \"https://api.siliconflow.cn/v1/chat/completions\" \\\n  -H \"Authorization: Bearer $SILICONFLOW_API_KEY\" \\\n  -H \"Content-Type: application/json\" \\\n  -d \"{\n    \\\"model\\\": \\\"Pro/deepseek-ai/DeepSeek-V3\\\",\n    \\\"messages\\\": [\n      {\n        \\\"role\\\": \\\"user\\\",\n        \\\"content\\\": [\n          {\n            \\\"type\\\": \\\"image_url\\\",\n            \\\"image_url\\\": {\n              \\\"url\\\": \\\"data:application/pdf;base64,${BASE64_PDF}\\\",\n              \\\"detail\\\": \\\"high\\\"\n            }\n          },\n          {\n            \\\"type\\\": \\\"text\\\",\n            \\\"text\\\": \\\"<image>\\\\n<|grounding|>Convert the document to markdown.\\\"\n          }\n        ]\n      }\n    ],\n    \\\"max_tokens\\\": 8192\n  }\" | python3 -c \"\nimport sys, json\ndata = json.load(sys.stdin)\nprint(data['choices'][0]['message']['content'])\n\"\n```\n\n### 多页 PDF 分页处理\n\n```bash\nPDF_PATH=\"/path/to/multipage.pdf\"\nOUTPUT_DIR=\"${OPENCLAW_WORKSPACE:-$PWD}/ocr_$(date +%Y%m%d_%H%M%S)\"\nmkdir -p \"$OUTPUT_DIR\"\n\n# 用 python3 分页提取并分别识别\npython3 << PYEOF\nimport base64, json, urllib.request, os, sys\n\npdf_path = \"$PDF_PATH\"\noutput_dir = \"$OUTPUT_DIR\"\napi_key = \"$SILICONFLOW_API_KEY\"\n\n# 尝试用 pypdf 分页\ntry:\n    import pypdf\n    reader = pypdf.PdfReader(pdf_path)\n    total_pages = len(reader.pages)\n    print(f\"PDF 共 {total_pages} 页，开始逐页识别...\")\n\n    all_results = []\n    for i, page in enumerate(reader.pages):\n        # 单页写成临时 PDF\n        writer = pypdf.PdfWriter()\n        writer.add_page(page)\n        tmp_path = f\"{output_dir}/page_{i+1:03d}.pdf\"\n        with open(tmp_path, \"wb\") as f:\n            writer.write(f)\n\n        # base64 编码\n        with open(tmp_path, \"rb\") as f:\n            b64 = base64.b64encode(f.read()).decode(\"utf-8\")\n\n        # 调用 API\n        payload = json.dumps({\n            \"model\": \"Pro/deepseek-ai/DeepSeek-V3\",\n            \"messages\": [{\n                \"role\": \"user\",\n                \"content\": [\n                    {\"type\": \"image_url\", \"image_url\": {\n                        \"url\": f\"data:application/pdf;base64,{b64}\",\n                        \"detail\": \"high\"\n                    }},\n                    {\"type\": \"text\", \"text\": \"<image>\\n<|grounding|>Convert the document to markdown.\"}\n                ]\n            }],\n            \"max_tokens\": 4096\n        }).encode(\"utf-8\")\n\n        req = urllib.request.Request(\n            \"https://api.siliconflow.cn/v1/chat/completions\",\n            data=payload,\n            headers={\"Authorization\": f\"Bearer {api_key}\", \"Content-Type\": \"application/json\"}\n        )\n        with urllib.request.urlopen(req) as resp:\n            result = json.loads(resp.read())\n            content = result[\"choices\"][0][\"message\"][\"content\"]\n            all_results.append(f\"## 第 {i+1} 页\\n\\n{content}\")\n            print(f\"✅ 第 {i+1}/{total_pages} 页识别完成\")\n\n    # 合并输出\n    merged = \"\\n\\n---\\n\\n\".join(all_results)\n    output_path = f\"{output_dir}/result.md\"\n    with open(output_path, \"w\", encoding=\"utf-8\") as f:\n        f.write(merged)\n    print(f\"\\n✅ 全部完成，结果已保存：{output_path}\")\n\nexcept ImportError:\n    print(\"需要安装 pypdf：pip install pypdf\")\n    print(\"安装后重新运行\")\n    sys.exit(1)\nPYEOF\n```\n\n---\n\n## Step 4：格式化输出\n\n识别完成后根据用户需求输出：\n\n### 文档转 Markdown（保留结构）\n\n```\n直接输出 Markdown 内容，保留：\n  - 标题层级（# ## ###）\n  - 列表（- * 1.）\n  - 表格（| 列1 | 列2 |）\n  - 代码块（```）\n  - 加粗、斜体等格式\n```\n\n### 发票/证件识别（结构化输出）\n\n```\n📄 发票识别结果\n━━━━━━━━━━━━━━━━━━━━\n发票类型：增值税专用发票\n发票号码：XXXXXXXXXXXXXXXX\n开票日期：2026年03月21日\n购买方：[公司名称]\n销售方：[公司名称]\n商品/服务：[明细]\n不含税金额：¥X,XXX.XX\n税率：13%\n税额：¥XXX.XX\n价税合计：¥X,XXX.XX\n```\n\n### 表格数据（CSV 友好格式）\n\n```\n识别结果同时输出：\n1. Markdown 表格（可读）\n2. 询问用户是否需要 CSV 格式（方便导入 Excel）\n```\n\n---\n\n## 输出文件保存\n\n```bash\nOUTPUT_DIR=\"${OPENCLAW_WORKSPACE:-$PWD}/ocr_$(date +%Y%m%d_%H%M%S)\"\nmkdir -p \"$OUTPUT_DIR\"\n\n# 保存 Markdown 结果\ncat > \"$OUTPUT_DIR/result.md\" << 'RESULT'\n{OCR识别内容}\nRESULT\n\necho \"✅ OCR 识别完成\"\necho \"结果已保存：$OUTPUT_DIR/result.md\"\n```\n\n---\n\n## 错误处理\n\n```\n文件不存在           → 提示用户确认路径\n文件过大（>10MB）    → 建议压缩或分页处理\n图片分辨率过低       → 提示识别效果可能较差，建议重新拍摄\nPDF 加密            → 提示需要先解密：qpdf --decrypt input.pdf output.pdf\n识别结果为空         → 可能是纯图片型PDF，尝试截图后重新识别\n401 错误            → API Key 失效，重新获取\n429 错误            → 请求频率超限，等待后重试\n```\n\n---\n\n## 注意事项\n\n- 图片最小 56×56，最大 3584×3584 像素，超出会自动压缩\n- PDF 支持 base64 编码输入，DeepSeek-OCR 同时支持 PDF URL\n- 多页 PDF 需要安装 pypdf：`pip install pypdf`\n- 识别结果保存到工作区，长期保留\n- detail=high 时按实际像素计费，detail=low 统一约256 token，复杂文档建议用 high\n- 发票/证件等隐私文件处理后请及时删除工作区临时文件\n\nFile v1.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn75z6gevjsyrznm7dg2ez6sen82h8sz\",\n  \"slug\": \"china-doc-ocr\",\n  \"version\": \"1.0.0\",\n  \"publishedAt\": 1774108057366\n}\n\nFile v1.0.0:references/models.md\n\n# 模型选择说明\n\n来源：硅基流动官方文档\n\n## 主力模型：Pro/deepseek-ai/DeepSeek-V3\n\n```\n模型名：Pro/deepseek-ai/DeepSeek-V3\n特点：\n  - 支持图片和 PDF 输入（base64 或 URL）\n  - 中文文档识别准确率极高\n  - 支持专用 OCR prompt 格式（<image>\\n<|grounding|>...）\n  - 支持多图对比分析\n适用：所有文档类型的首选模型\n```\n\n## 备用模型：Qwen2.5-VL-72B\n\n```\n模型名：Qwen/Qwen2.5-VL-72B-Instruct\n特点：\n  - 超大参数量，复杂文档理解力强\n  - 支持图片输入（不支持 PDF base64）\n  - 中英文双语文档效果好\n适用：DeepSeek-OCR 效果不佳时的备选\n```\n\n## PaddleOCR-VL（专业 OCR 场景）\n\n```\n模型名：PaddlePaddle/PaddleOCR-VL\n特点：\n  - 专为 OCR 任务优化\n  - 支持 CLI 和 API 两种调用方式\n  - 对复杂版面（多列、表格）效果好\n适用：需要精确版面还原的场景\n```\n\n## 图像输入计费（detail 参数影响）\n\n```\ndetail=low：\n  统一压缩为 448×448，约 256 token\n  适合：文字较大、布局简单的文档\n\ndetail=high（推荐）：\n  按实际像素计费：ceil(h/28) × ceil(w/28) token\n  适合：字体较小、布局复杂、表格密集的文档\n\n建议默认使用 detail=high，确保识别准确率\n```\n\n## 模型选择决策\n\n```\n发票/收据/证件        → DeepSeek-V3（中文场景最佳）\n普通文档/报告         → DeepSeek-V3\n复杂表格/多列版面     → DeepSeek-V3 或 PaddleOCR-VL\n英文为主的文档        → Qwen2.5-VL-72B\nDeepSeek 结果不满意  → 改用 Qwen2.5-VL-72B 重试\n```\n\nFile v1.0.0:references/prompts.md\n\n# OCR 提示词模板\n\n来源：硅基流动官方文档 DeepSeek-OCR 专用 prompt 格式\n\n## 使用方式\n\n所有 prompt 在 text 字段中传入，格式固定为：\n`<image>\\n<|grounding|>{具体指令}`\n\n---\n\n## 1. 文档转 Markdown（默认推荐）\n\n```\n<image>\n<|grounding|>Convert the document to markdown.\n```\n\n输出：保留标题层级、列表、表格、加粗等所有格式\n适用：报告、合同、论文、书籍页面、网页截图\n\n---\n\n## 2. 通用 OCR（纯文字提取）\n\n```\n<image>\n<|grounding|>OCR this image.\n```\n\n输出：按阅读顺序提取所有文字，保留基本换行\n适用：简单图片、截图文字、无复杂格式的文档\n\n---\n\n## 3. 无布局 OCR（去除所有格式）\n\n```\n<image>\nFree OCR.\n```\n\n注意：这个格式不加 `<|grounding|>` 前缀\n输出：纯文字流，不保留任何格式信息\n适用：只需要文字内容、不关心排版的场景\n\n---\n\n## 4. 图表/表格解析\n\n```\n<image>\n<|grounding|>Parse the figure.\n```\n\n输出：结构化描述图表内容，表格转为 Markdown 格式\n适用：柱状图、折线图、饼图、数据表格、流程图\n\n---\n\n## 5. 图片详细描述\n\n```\n<image>\n<|grounding|>Describe this image in detail.\n```\n\n输出：详细描述图片内容，包括文字、图形、布局\n适用：混合图文内容，需要全面理解图片的场景\n\n---\n\n## 6. 文字定位\n\n```\n<image>\n<|grounding|>Locate <|ref|>要定位的文字<|/ref|> in the image.\n```\n\n输出：指出特定文字在图片中的位置\n适用：需要找到特定字段位置的场景\n\n---\n\n## 场景专用 Prompt\n\n### 发票识别\n\n```\n请识别这张发票的所有信息，以结构化格式输出：\n发票类型、发票号码、开票日期、\n购买方信息（名称/税号/地址）、\n销售方信息（名称/税号/地址）、\n商品明细（名称/数量/单价/金额）、\n税率、税额、价税合计、备注。\n如有字段无法识别，标注\"不清晰\"。\n```\n\n### 身份证识别\n\n```\n请识别这张身份证的信息：\n姓名、性别、民族、出生日期、住址、公民身份号码。\n输出结构化格式，隐私字段用*号部分隐藏。\n```\n\n### 营业执照识别\n\n```\n请识别这份营业执照的信息：\n公司名称、统一社会信用代码、类型、法定代表人、\n注册资本、成立日期、营业期限、经营范围、注册地址。\n```\n\n### 银行流水/对账单\n\n```\n请识别这份银行流水的所有交易记录，\n以表格格式输出：日期、摘要、支出、收入、余额。\n如有多页请按时间顺序排列。\n```\n\n### 合同关键信息提取\n\n```\n请识别这份合同的关键信息：\n合同编号、签订日期、甲方、乙方、\n合同金额、付款方式、合同期限、\n主要条款摘要（不超过200字）、\n双方签字/盖章情况。\n```\n\n### 学术论文结构化\n\n```\n<image>\n<|grounding|>Convert the document to markdown.\n请保留所有标题层级、公式（用LaTeX格式）、\n参考文献编号和表格结构。\n```\n\n### 表格数据提取（输出 CSV）\n\n```\n请识别这个表格中的所有数据，\n以 CSV 格式输出（逗号分隔），\n第一行为表头，之后每行为一条数据。\n如有合并单元格，请拆分填充。\n```\n\nFile v1.0.0:LICENSE.txt\n\nMIT No Attribution (MIT-0)\n\nCopyright 2026 ToBeWin\n\nPermission is hereby granted, free of charge, to any person obtaining a copy\nof this software and associated documentation files (the \"Software\"), to deal\nin the Software without restriction, including without limitation the rights\nto use, copy, modify, merge, publish, distribute, sublicense, and/or sell\ncopies of the Software, and to permit persons to whom the Software is\nfurnished to do so.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR\nIMPLIED.","readmeExcerpt":"Skill: china-doc-ocr Owner: tobewin Summary: 智能文档OCR识别与结构化提取。Use when the user has a complex document, PDF, scanned image, photo, invoice, receipt, ID card, table, or chart that needs to be recognized a... Tags: china:1.2.0, document:1.2.0, image:1.2.0, invoice:1.2.0, latest:1.2.0, ocr:1.2.0, paddleocr:1.2.0, pdf:1.2.0, recognition:1.2.0, siliconflow:1.2.0 Version history: v1.2.0 | 2026-04-29T01:19:35.463Z | user v1.","codeSnippets":[],"executableExamples":[{"language":"text","snippet":"OCR优先级：\n1. PaddleOCR-VL-1.5 (免费、快速、专业OCR)\n2. DeepSeek-OCR (免费、效果好)\n3. Qwen2.5-VL-72B (视觉语言模型，OCR效果一般但可补充)\n\n默认使用 PaddleOCR-VL-1.5\n如果识别效果不好，降级到 DeepSeek-OCR\n如果仍然不好，降级到 Qwen2.5-VL-72B"},{"language":"bash","snippet":"# 检查 API Key\nif [ -z \"$SILICONFLOW_API_KEY\" ]; then\n  echo \"缺少 SILICONFLOW_API_KEY\"\n  echo \"配置方法：\"\n  echo \"  1. 访问 cloud.siliconflow.cn 注册（国内直连）\"\n  echo \"  2. 进入「API密钥」页面创建 Key\"\n  echo \"  3. export SILICONFLOW_API_KEY='sk-xxxxxxxx'\"\n  exit 1\nfi"},{"language":"text","snippet":"用户提供文件路径或 URL → 判断类型：\n\n文件扩展名/用户描述 → 处理模式：\n\n.pdf                    → PDF 模式\n.jpg/.jpeg/.png/.webp   → 图片模式\n.bmp/.tiff/.gif         → 图片模式（先转换格式）\nURL（http/https开头）   → URL 直接模式\n用户粘贴了 base64       → 直接使用\n\n用户意图 → 选择 Prompt 模式：\n\n\"转成文字/提取文字\"     → 通用OCR\n\"转成Markdown/保留格式\" → 文档转Markdown\n\"提取表格/表格数据\"     → 图表解析\n\"发票/收据/单据\"        → 发票识别\n\"身份证/证件/执照\"      → 证件识别\n\"图表/图形/柱状图\"      → 图表解析\n未指定                  → 默认文档转Markdown"},{"language":"bash","snippet":"python3 scripts/ocr.py \\\n  --image \"/path/to/image.jpg\" \\\n  --prompt \"Convert the document to markdown.\" \\\n  --model paddleocr"},{"language":"bash","snippet":"python3 scripts/ocr.py \\\n  --url \"https://example.com/document.jpg\" \\\n  --prompt \"Convert the document to markdown.\" \\\n  --model deepseek"},{"language":"bash","snippet":"# 使用 PaddleOCR（默认，推荐）\npython3 scripts/ocr.py --image photo.jpg --model paddleocr\n\n# 使用 DeepSeek-OCR\npython3 scripts/ocr.py --image photo.jpg --model deepseek\n\n# 使用 Qwen2.5-VL\npython3 scripts/ocr.py --image photo.jpg --model qwen"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: china-doc-ocr\ndescription: 智能文档OCR识别与结构化提取。Use when the user has a complex document, PDF, scanned image, photo, invoice, receipt, ID card, table, or chart that needs to be recognized and converted to text or Markdown. Uses PaddleOCR-VL-1.5 and DeepSeek-OCR. 文档OCR、发票识别、证件识别。\nversion: 1.2.0\nlicense: MIT-0\nmetadata: {\"openclaw\": {\"emoji\": \"📄\", \"requires\": {\"bins\": [\"python3\"], \"env\": [\"SILICONFLOW_API_KEY\"]}, \"primaryEnv\": \"SILICONFLOW_API_KEY\"}}\n---\n\n# 智能文档 OCR China Doc OCR\n\n识别并提取复杂文档内容：PDF、图片、扫描件、发票、表格、证件等。\n使用硅基流动 DeepSeek-OCR / PaddleOCR-VL，国内直连，无需翻墙。\n\n模型选择与参数说明 → `references/models.md`\n各场景提示词模板 → `references/prompts.md`\n\n## 触发时机\n\n- \"帮我识别这个PDF/图片里的内容\"\n- \"把这张发票/收据的信息提取出来\"\n- \"将这份扫描合同转成可编辑文字\"\n- \"这个表格里的数据帮我提取一下\"\n- \"帮我把这张截图的文字识别出来\"\n- \"这份报告转成 Markdown 格式\"\n- \"识别这张身份证/营业执照的信息\"\n\n---\n\n## 模型选择策略（优先OCR）\n\n```\nOCR优先级：\n1. PaddleOCR-VL-1.5 (免费、快速、专业OCR)\n2. DeepSeek-OCR (免费、效果好)\n3. Qwen2.5-VL-72B (视觉语言模型，OCR效果一般但可补充)\n\n默认使用 PaddleOCR-VL-1.5\n如果识别效果不好，降级到 DeepSeek-OCR\n如果仍然不好，降级到 Qwen2.5-VL-72B\n```\n\n---\n\n## Step 0：环境检查\n\n```bash\n# 检查 API Key\nif [ -z \"$SILICONFLOW_API_KEY\" ]; then\n  echo \"缺少 SILICONFLOW_API_KEY\"\n  echo \"配置方法：\"\n  echo \"  1. 访问 cloud.siliconflow.cn 注册（国内直连）\"\n  echo \"  2. 进入「API密钥」页面创建 Key\"\n  echo \"  3. export SILICONFLOW_API_KEY='sk-xxxxxxxx'\"\n  exit 1\nfi\n```\n\n---\n\n## Step 1：识别内容类型，选择处理模式\n\n```\n用户提供文件路径或 URL → 判断类型：\n\n文件扩展名/用户描述 → 处理模式：\n\n.pdf                    → PDF 模式\n.jpg/.jpeg/.png/.webp   → 图片模式\n.bmp/.tiff/.gif         → 图片模式（先转换格式）\nURL（http/https开头）   → URL 直接模式\n用户粘贴了 base64       → 直接使用\n\n用户意图 → 选择 Prompt 模式：\n\n\"转成文字/提取文字\"     → 通用OCR\n\"转成Markdown/保留格式\" → 文档转Markdown\n\"提取表格/表格数据\"     → 图表解析\n\"发票/收据/单据\"        → 发票识别\n\"身份证/证件/执照\"      → 证件识别\n\"图表/图形/柱状图\"      → 图表解析\n未指定                  → 默认文档转Markdown\n```\n\n---\n\n## Step 2：图片 OCR\n\n### 本地图片文件\n\n```bash\npython3 scripts/ocr.py \\\n  --image \"/path/to/image.jpg\" \\\n  --prompt \"Convert the document to markdown.\" \\\n  --model paddleocr\n```\n\n### 图片 URL\n\n```bash\npython3 scripts/ocr.py \\\n  --url \"https://example.com/document.jpg\" \\\n  --prompt \"Convert the document to markdown.\" \\\n  --model deepseek\n```\n\n### 指定模型\n\n```bash\n# 使用 PaddleOCR（默认，推荐）\npython3 scripts/ocr.py --image photo.jpg --model paddleocr\n\n# 使用 DeepSeek-OCR\npython3 scripts/ocr.py --image photo.jpg --model deepseek\n\n# 使用 Qwen2.5-VL\npython3 scripts/ocr.py --image photo.jpg --model qwen\n```\n\n---\n\n## Step 3：PDF OCR\n\n### 单页或少页 PDF\n\n```bash\npython3 scripts/ocr.py \\\n  --pdf \"/path/to/document.pdf\" \\\n  --prompt \"Convert the document to markdown.\" \\\n  --model deepseek\n```\n\n### 多页 PDF\n\n多页 PDF 需要分页处理。使用 Python 脚本：\n\n1. 使用 pypdf 分页\n2. 对每页分别调用 OCR\n3. 合并结果\n\n---\n\n## Step 4：格式化输出\n\n识别完成后根据用户需求输出：\n\n### 文档转 Markdown（保留结构）\n\n```\n直接输出 Markdown 内容，保留：\n  - 标题层级（# ## ###）\n  - 列表（- * 1.）\n  - 表格（| 列1 | 列2 |）\n  - 代码块（```）\n  - 加粗、斜体等格式\n```\n\n### 发票/证件识别（结构化输出）\n\n```\n发票识别结果\n━━━━━━━━━━━━━━━━━━━━\n发票类型：增值税专用发票\n发票号码：XXXXXXXXXXXXXXXX\n开票日期：2026年03月21日\n购买方：[公司名称]\n销售方：[公司名称]\n商品/服务：[明细]\n不含税金额：¥X,XXX.XX\n税率：13%\n税额：¥XXX.XX\n价税合计：¥X,XXX.XX\n```\n\n### 表格数据（CSV 友好格式）\n\n```\n识别结果同时输出：\n1. Markdown 表格（可"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn75z6gevjsyrznm7dg2ez6sen82h8sz\",\n  \"slug\": \"china-doc-ocr\",\n  \"version\": \"1.2.0\",\n  \"publishedAt\": 1777425575463\n}"},{"path":"references/models.md","content":"# 模型选择说明\n\n来源：硅基流动官方文档\n\n## 主力模型：Pro/deepseek-ai/DeepSeek-V3\n\n```\n模型名：Pro/deepseek-ai/DeepSeek-V3\n特点：\n  - 支持图片和 PDF 输入（base64 或 URL）\n  - 中文文档识别准确率极高\n  - 支持专用 OCR prompt 格式（<image>\\n<|grounding|>...）\n  - 支持多图对比分析\n适用：所有文档类型的首选模型\n```\n\n## 备用模型：Qwen2.5-VL-72B\n\n```\n模型名：Qwen/Qwen2.5-VL-72B-Instruct\n特点：\n  - 超大参数量，复杂文档理解力强\n  - 支持图片输入（不支持 PDF base64）\n  - 中英文双语文档效果好\n适用：DeepSeek-OCR 效果不佳时的备选\n```\n\n## PaddleOCR-VL（专业 OCR 场景）\n\n```\n模型名：PaddlePaddle/PaddleOCR-VL\n特点：\n  - 专为 OCR 任务优化\n  - 支持 CLI 和 API 两种调用方式\n  - 对复杂版面（多列、表格）效果好\n适用：需要精确版面还原的场景\n```\n\n## 图像输入计费（detail 参数影响）\n\n```\ndetail=low：\n  统一压缩为 448×448，约 256 token\n  适合：文字较大、布局简单的文档\n\ndetail=high（推荐）：\n  按实际像素计费：ceil(h/28) × ceil(w/28) token\n  适合：字体较小、布局复杂、表格密集的文档\n\n建议默认使用 detail=high，确保识别准确率\n```\n\n## 模型选择决策\n\n```\n发票/收据/证件        → DeepSeek-V3（中文场景最佳）\n普通文档/报告         → DeepSeek-V3\n复杂表格/多列版面     → DeepSeek-V3 或 PaddleOCR-VL\n英文为主的文档        → Qwen2.5-VL-72B\nDeepSeek 结果不满意  → 改用 Qwen2.5-VL-72B 重试\n```"},{"path":"references/prompts.md","content":"# OCR 提示词模板\n\n来源：硅基流动官方文档 DeepSeek-OCR 专用 prompt 格式\n\n## 使用方式\n\n所有 prompt 在 text 字段中传入，格式固定为：\n`<image>\\n<|grounding|>{具体指令}`\n\n---\n\n## 1. 文档转 Markdown（默认推荐）\n\n```\n<image>\n<|grounding|>Convert the document to markdown.\n```\n\n输出：保留标题层级、列表、表格、加粗等所有格式\n适用：报告、合同、论文、书籍页面、网页截图\n\n---\n\n## 2. 通用 OCR（纯文字提取）\n\n```\n<image>\n<|grounding|>OCR this image.\n```\n\n输出：按阅读顺序提取所有文字，保留基本换行\n适用：简单图片、截图文字、无复杂格式的文档\n\n---\n\n## 3. 无布局 OCR（去除所有格式）\n\n```\n<image>\nFree OCR.\n```\n\n注意：这个格式不加 `<|grounding|>` 前缀\n输出：纯文字流，不保留任何格式信息\n适用：只需要文字内容、不关心排版的场景\n\n---\n\n## 4. 图表/表格解析\n\n```\n<image>\n<|grounding|>Parse the figure.\n```\n\n输出：结构化描述图表内容，表格转为 Markdown 格式\n适用：柱状图、折线图、饼图、数据表格、流程图\n\n---\n\n## 5. 图片详细描述\n\n```\n<image>\n<|grounding|>Describe this image in detail.\n```\n\n输出：详细描述图片内容，包括文字、图形、布局\n适用：混合图文内容，需要全面理解图片的场景\n\n---\n\n## 6. 文字定位\n\n```\n<image>\n<|grounding|>Locate <|ref|>要定位的文字<|/ref|> in the image.\n```\n\n输出：指出特定文字在图片中的位置\n适用：需要找到特定字段位置的场景\n\n---\n\n## 场景专用 Prompt\n\n### 发票识别\n\n```\n请识别这张发票的所有信息，以结构化格式输出：\n发票类型、发票号码、开票日期、\n购买方信息（名称/税号/地址）、\n销售方信息（名称/税号/地址）、\n商品明细（名称/数量/单价/金额）、\n税率、税额、价税合计、备注。\n如有字段无法识别，标注\"不清晰\"。\n```\n\n### 身份证识别\n\n```\n请识别这张身份证的信息：\n姓名、性别、民族、出生日期、住址、公民身份号码。\n输出结构化格式，隐私字段用*号部分隐藏。\n```\n\n### 营业执照识别\n\n```\n请识别这份营业执照的信息：\n公司名称、统一社会信用代码、类型、法定代表人、\n注册资本、成立日期、营业期限、经营范围、注册地址。\n```\n\n### 银行流水/对账单\n\n```\n请识别这份银行流水的所有交易记录，\n以表格格式输出：日期、摘要、支出、收入、余额。\n如有多页请按时间顺序排列。\n```\n\n### 合同关键信息提取\n\n```\n请识别这份合同的关键信息：\n合同编号、签订日期、甲方、乙方、\n合同金额、付款方式、合同期限、\n主要条款摘要（不超过200字）、\n双方签字/盖章情况。\n```\n\n### 学术论文结构化\n\n```\n<image>\n<|grounding|>Convert the document to markdown.\n请保留所有标题层级、公式（用LaTeX格式）、\n参考文献编号和表格结构。\n```\n\n### 表格数据提取（输出 CSV）\n\n```\n请识别这个表格中的所有数据，\n以 CSV 格式输出（逗号分隔），\n第一行为表头，之后每行为一条数据。\n如有合并单元格，请拆分填充。\n```"},{"path":"skill-card.md","content":"## Description:\n\nChina Doc OCR helps agents recognize and structure content from documents, PDFs, scanned images, photos, invoices, receipts, ID cards, tables, and charts using SiliconFlow-hosted OCR models.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[tobewin](https://clawhub.ai/user/tobewin)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers, operators, and external users use this skill to convert Chinese and mixed-language document images, PDFs, invoices, IDs, tables, and charts into editable text, Markdown, or structured extraction results.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill sends documents to a third-party OCR API, which can expose sensitive personal, financial, legal, or regulated business information.\n\nMitigation: Use it only for documents the user is permitted to upload, and redact or avoid IDs, bank statements, contracts, invoices, and regulated records unless upload is approved.\n\nRisk: OCR results and temporary outputs may remain in the workspace after processing.\n\nMitigation: Delete retained workspace outputs and temporary files when the task is finished, especially after handling private documents.\n\nRisk: The skill requires a SiliconFlow API key for network calls.\n\nMitigation: Keep the API key in the environment, avoid pasting it into prompts or files, and rotate it if it may have been exposed.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/tobewin/skills/china-doc-ocr)\n- [Model selection reference](references/models.md)\n- [OCR prompt templates](references/prompts.md)\n- [SiliconFlow API endpoint](https://api.siliconflow.cn/v1/chat/completions)\n- [SiliconFlow console](https://cloud.siliconflow.cn)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, Shell commands, Configuration, Guidance]\n\n**Output Format:** [Markdown, structured text, and shell command snippets]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May save OCR results in the workspace; the helper supports image, PDF, or URL input and defaults to 4096 response tokens.]\n\n## Skill Version(s):\n\n1.2.0 (source: frontmatter, release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"智能文档OCR识别与结构化提取。Use when the user has a complex document, PDF, scanned image, photo, invoice, receipt, ID card, table, or chart that needs to be recognized a... Skill: china-doc-ocr Owner: tobewin Summary: 智能文档OCR识别与结构化提取。Use when the user has a complex document, PDF, scanned image, photo, invoice, receipt, ID card, table, or chart that needs to be recognized a... Tags: china:1.2.0, document:1.2.0, image:1.2.0, invoice:1.2.0, latest:1.2.0, ocr:1.2.0, paddleocr:1.2.0, pdf:1.2.0, recognition:1.2.0, siliconflow:1.2.0 Version history: v1.2.0 | 2026-04-29T01:19:35.463Z | user v1.","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":955,"uniquenessScore":54,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T16:13:35.034Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T16:13:35.034Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T18:22:05.938Z","emptyReason":null},"items":[{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-10T18:48:31.762Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}