{"id":"21894601-3970-48f0-ab91-b9093c309bfd","entityType":"agent","slug":"clawhub-bobholamovic-paddleocr-text-recognition","name":"PaddleOCR Text Recognition","canonicalUrl":"https://www.xpersona.co/agent/clawhub-bobholamovic-paddleocr-text-recognition","canonicalPath":"/agent/clawhub-bobholamovic-paddleocr-text-recognition","generatedAt":"2026-10-09T13:54:14.345Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-09T05:14:42.977Z","emptyReason":null},"description":"Use this skill whenever the user wants text extracted from images, photos, scans, screenshots, or scanned PDFs. Returns exact machine-readable strings with l... Skill: PaddleOCR Text Recognition Owner: bobholamovic Summary: Use this skill whenever the user wants text extracted from images, photos, scans, screenshots, or scanned PDFs. Returns exact machine-readable strings with l... Tags: latest:2.0.0 Version history: v2.0.0 | 2026-06-05T10:10:23.964Z | user - Major refactor: removed all local wrapper scripts and sample/reference files, switching to direct CLI usage. - SKILL.","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 4.5K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s17brd0e37a6ay53kynxvwrt6x83gx1w:paddleocr-text-recognition","sourceUrl":"https://clawhub.ai/bobholamovic/paddleocr-text-recognition","homepage":"https://clawhub.ai/bobholamovic/skills/paddleocr-text-recognition","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/bobholamovic/paddleocr-text-recognition","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/bobholamovic/skills/paddleocr-text-recognition","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":66,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Use this skill whenever the user wants text extracted from images, photos, scans, screenshots, or scanned PDFs. Returns exact machine-readable strings with l..."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T05:14:42.977Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T05:14:42.977Z","emptyReason":null},"stars":null,"forks":null,"downloads":4541,"packageName":null,"latestVersion":"2.0.0","tractionLabel":"4.5K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T05:14:42.977Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T05:14:42.977Z","lastCrawledAt":"2026-10-09T05:14:42.977Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T05:14:42.977Z","lastVerifiedAt":null,"highlights":[{"version":"2.0.0","createdAt":"2026-06-05T10:10:23.964Z","changelog":"- Major refactor: removed all local wrapper scripts and sample/reference files, switching to direct CLI usage. - SKILL.md is now much shorter and focuses on using the upstream paddleocr CLI tool directly. - Installation and environment details are simplified—API configuration only requires PADDLEOCR_ACCESS_TOKEN. - All previous internal instructions, JSON handling tips, and error guidance for wrapper scripts have been removed. - Usage examples now demonstrate running OCR directly with the paddleocr CLI, including preprocessing and output options.","fileCount":3,"zipByteSize":3422},{"version":"1.0.21","createdAt":"2026-04-03T04:26:20.432Z","changelog":"- License information has been removed from the YAML header. - No code or functionality changes; documentation and usage remain consistent with previous versions.","fileCount":7,"zipByteSize":14431},{"version":"1.0.20","createdAt":"2026-04-03T04:25:41.183Z","changelog":"No changes detected in this version. - Version 1.0.20 was published with no modifications to files or documentation.","fileCount":6,"zipByteSize":13200},{"version":"1.0.19","createdAt":"2026-04-03T04:21:25.478Z","changelog":"- Switched to inline Python dependency declaration (PEP 723) for uv compatibility; no requirements.txt file needed. - Updated installation instructions to use uv for zero-config dependency management. - Minor metadata improvements: added license, compatibility, and clarified usage notes. - No functional changes to core OCR features or command-line interface.","fileCount":6,"zipByteSize":13200},{"version":"1.0.18","createdAt":"2026-03-30T08:36:50.356Z","changelog":"- Documentation streamlined: redundant information, especially duplicate and overly detailed first-time configuration steps, has been trimmed for clarity. - API configuration instructions updated to clarify language/model selection steps. - Core usage steps are unchanged; skill usage and command-line examples remain the same. - No code or functional changes; this is a documentation-only update for improved usability.","fileCount":7,"zipByteSize":13104},{"version":"1.0.17","createdAt":"2026-03-28T03:41:35.485Z","changelog":"No file changes detected for version 1.0.17. - No changes or updates in this release; all documentation and code remain the same.","fileCount":7,"zipByteSize":13173},{"version":"1.0.16","createdAt":"2026-03-27T12:35:11.807Z","changelog":"No code or documentation changes detected in this version. - No file or documentation updates found between versions. - Functionality and usage remain unchanged.","fileCount":7,"zipByteSize":13113},{"version":"1.0.15","createdAt":"2026-03-27T10:36:05.224Z","changelog":"- Documentation streamlined for clarity: removed duplication, reorganized sections, and improved formatting for easier reference. - Added performance notes about expected OCR processing times for images and large PDFs. - Clarified criteria for when to use/not use the skill and when to use \"Document Parsing\" instead. - Made output requirements explicit: display the entire recognized text, avoid truncation, and outlined correct/incorrect examples. - Updated configuration and error handling instructions to be briefer and more direct. - Provided usage examples with file-type explanation and concise guidance on result handling.","fileCount":7,"zipByteSize":12367}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17brd0e37a6ay53kynxvwrt6x83gx1w:paddleocr-text-recognition","setupComplexity":"low","setupSteps":["Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-bobholamovic-paddleocr-text-recognition/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-bobholamovic-paddleocr-text-recognition/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-bobholamovic-paddleocr-text-recognition/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-bobholamovic-paddleocr-text-recognition/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-bobholamovic-paddleocr-text-recognition/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-bobholamovic-paddleocr-text-recognition/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-09T13:54:14.342Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-bobholamovic-paddleocr-text-recognition/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-bobholamovic-paddleocr-text-recognition/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-bobholamovic-paddleocr-text-recognition/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-bobholamovic-paddleocr-text-recognition/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-09T05:14:42.977Z","emptyReason":null},"readme":"Skill: PaddleOCR Text Recognition\n\nOwner: bobholamovic\n\nSummary: Use this skill whenever the user wants text extracted from images, photos, scans, screenshots, or scanned PDFs. Returns exact machine-readable strings with l...\n\nTags: latest:2.0.0\n\nVersion history:\n\nv2.0.0 | 2026-06-05T10:10:23.964Z | user\n\n- Major refactor: removed all local wrapper scripts and sample/reference files, switching to direct CLI usage.\n- SKILL.md is now much shorter and focuses on using the upstream paddleocr CLI tool directly.\n- Installation and environment details are simplified—API configuration only requires PADDLEOCR_ACCESS_TOKEN.\n- All previous internal instructions, JSON handling tips, and error guidance for wrapper scripts have been removed.\n- Usage examples now demonstrate running OCR directly with the paddleocr CLI, including preprocessing and output options.\n\nv1.0.21 | 2026-04-03T04:26:20.432Z | user\n\n- License information has been removed from the YAML header.\n- No code or functionality changes; documentation and usage remain consistent with previous versions.\n\nv1.0.20 | 2026-04-03T04:25:41.183Z | user\n\nNo changes detected in this version.\n\n- Version 1.0.20 was published with no modifications to files or documentation.\n\nv1.0.19 | 2026-04-03T04:21:25.478Z | user\n\n- Switched to inline Python dependency declaration (PEP 723) for uv compatibility; no requirements.txt file needed.\n- Updated installation instructions to use uv for zero-config dependency management.\n- Minor metadata improvements: added license, compatibility, and clarified usage notes.\n- No functional changes to core OCR features or command-line interface.\n\nv1.0.18 | 2026-03-30T08:36:50.356Z | user\n\n- Documentation streamlined: redundant information, especially duplicate and overly detailed first-time configuration steps, has been trimmed for clarity.\n- API configuration instructions updated to clarify language/model selection steps.\n- Core usage steps are unchanged; skill usage and command-line examples remain the same.\n- No code or functional changes; this is a documentation-only update for improved usability.\n\nv1.0.17 | 2026-03-28T03:41:35.485Z | user\n\nNo file changes detected for version 1.0.17.\n\n- No changes or updates in this release; all documentation and code remain the same.\n\nv1.0.16 | 2026-03-27T12:35:11.807Z | user\n\nNo code or documentation changes detected in this version.\n\n- No file or documentation updates found between versions.\n- Functionality and usage remain unchanged.\n\nv1.0.15 | 2026-03-27T10:36:05.224Z | user\n\n- Documentation streamlined for clarity: removed duplication, reorganized sections, and improved formatting for easier reference.\n- Added performance notes about expected OCR processing times for images and large PDFs.\n- Clarified criteria for when to use/not use the skill and when to use \"Document Parsing\" instead.\n- Made output requirements explicit: display the entire recognized text, avoid truncation, and outlined correct/incorrect examples.\n- Updated configuration and error handling instructions to be briefer and more direct.\n- Provided usage examples with file-type explanation and concise guidance on result handling.\n\nv1.0.14 | 2026-03-27T06:37:47.388Z | user\n\n- requirements.txt moved from scripts/ to the top-level skill directory.\n- Environment variable PADDLEOCR_OCR_TIMEOUT removed from required list; now only API URL and access token are mandatory.\n- Documentation updated to install requirements from the new path.\n- Minor metadata and env var requirements cleanup.\n\nv1.0.13 | 2026-03-26T15:51:37.025Z | user\n\nNo code changes detected in this release.  \nSkill documentation updated only.\n\n- Updated the skill description for improved clarity, detail, and accuracy of supported use cases.\n- Enhanced trigger terms and usage explanation in the YAML metadata and introductory sections.\n- No changes to scripts, logic, or runtime behavior.  \n- Safe to use as a documentation/metadata update only.\n\nv1.0.12 | 2026-03-24T17:49:19.738Z | user\n\nNo user-facing changes detected in this version.\n\n- No changes found between the previous and current version files.\n- Behavior, guidance, and usage remain the same as in the last release.\n\nv1.0.11 | 2026-03-24T17:43:24.110Z | user\n\n- Added bilingual trigger keywords and routing instructions to the description field for improved discovery (now includes both Chinese and English terms such as OCR, 文字识别, plain text extraction, bbox, etc.).\n- Updated usage instructions and \"When to Use This Skill\" section to clarify routing and trigger logic.\n- No changes to code or functional behavior; only SKILL.md metadata and documentation updated.\n\nv1.0.10 | 2026-03-17T04:13:28.678Z | user\n\n- Improved first-time configuration instructions: users are now guided to get the API URL and token from the official PaddleOCR website (with model and configuration details).\n- Added specific note that supported model is PP-OCRv5, and clarified environment variable setup guidance (including example for OpenClaw).\n- Enhanced credential handling: warns users about sharing sensitive data in chat, and recommends secure environment variable configuration.\n- No functional or code changes; documentation only.\n\nv1.0.9 | 2026-03-14T06:58:09.278Z | user\n\nVersion 1.0.9 Changelog\n\n- Added explicit installation instructions for required Python dependencies using `pip install -r scripts/requirements.txt`.\n- Clarified the need to install dependencies before using the skill.\n- No changes to functionality or script usage—documentation improvements only.\n\nv1.0.8 | 2026-03-13T16:11:37.700Z | auto\n\n- Removed the no-longer-needed `scripts/configure.py` script.\n- Updated documentation in SKILL.md for improved clarity and security guidance.\n- Enhanced error handling instructions and user security warnings in configuration sections.\n- Refactored script code and requirements for maintainability.\n- Added clarifications on parsing, saving, and presenting OCR results.\n\nv1.0.7 | 2026-03-13T15:37:10.903Z | user\n\n## paddleocr-text-recognition 1.0.7\n\n- No code or documentation changes detected in this release.\n- No user-facing updates or features introduced.\n\nv1.0.6 | 2026-03-13T10:07:54.156Z | user\n\n- Simplified and clarified the skill description for easier understanding.\n- Updated configuration instructions to assume environment variables are already set, unless an OCR task fails due to configuration issues.\n- Improved guidance for handling credentials, including stronger recommendations against providing secrets in chat or creating local files.\n- Removed redundant and overly detailed workflow text for easier use and maintenance.\n- Error handling and output display guidance remain strict: always show complete OCR results and exact error messages.\n\nv1.0.5 | 2026-03-11T14:34:10.709Z | user\n\nVersion 1.0.5\n\n- Updated configuration instructions to recommend secure credential setup via the host application, rather than pasting credentials in chat.\n- Added explicit security warning if credentials are provided in chat, highlighting that such information may be stored in conversation history.\n- Clarified environment variable setup steps and emphasized secure configuration.\n- No functional changes to the skill’s OCR handling or results workflow.\n\nv1.0.4 | 2026-03-11T14:06:49.421Z | user\n\npaddleocr-text-recognition v1.0.4\n\n- Updated environment variable naming: now requires PADDLEOCR_OCR_TIMEOUT instead of PADDLEOCR_TIMEOUT.\n- Minor documentation improvements in reference links and metadata.\n- No functional or code changes detected.\n\nv1.0.3 | 2026-03-11T14:02:05.665Z | user\n\n- Added metadata block specifying required environment variables, dependencies, emoji, and homepage URL for improved integration and discoverability.\n- No functional or workflow changes to the skill itself.\n- Documentation now reflects explicit environment variable and binary dependencies for clarity.\n\nv1.0.2 | 2026-03-11T13:58:55.609Z | user\n\n- Updated first-time configuration instructions: users are now directed to set required environment variables (API URL and token) in the host application or runtime environment, instead of configuring them in-band via script.\n- Emphasized NOT to run the skill's configure script or create local `.env` files by default, especially in host-managed environments.\n- Clarified the workflow for parsing credentials, validation, and when to retry OCR after configuration.\n- Removed internal metadata from the documentation.\n- General documentation clean-up and improvements for clarity.\n\nv1.0.1 | 2026-03-10T10:05:01.345Z | user\n\n- Added metadata section to SKILL.md, including required environment variables, binary dependencies, and homepage link.\n- No code or functionality changes detected.\n\nv1.0.0 | 2026-03-10T09:14:26.536Z | user\n\n- Initial release of paddleocr-text-recognition skill.\n- Enables text extraction from images, PDFs, and document files via PaddleOCR API.\n- Supports processing with both URLs and local file paths.\n- Returns results as structured JSON, including complete recognized text.\n- Strict usage and error handling guidelines: only PaddleOCR API is used, no fallback methods allowed.\n- Automatically guides users through configuration if API credentials are missing.\n\nArchive index:\n\nArchive v2.0.0: 3 files, 3422 bytes\n\nFiles: skill-card.md (2411b), SKILL.md (3726b), _meta.json (145b)\n\nFile v2.0.0:SKILL.md\n\n---\nname: paddleocr-text-recognition\ndescription: >-\n  Use this skill whenever the user wants text extracted from images, photos, scans, screenshots,\n  or scanned PDFs. Returns exact machine-readable strings with line-level text and optional bbox\n  coordinates. Strong accuracy for CJK, small print, and handwritten text.\n  Trigger terms: OCR, 文字识别, 图片转文字, 截图识字, 提取图中文字, 扫描识字, 识字, 纯文字,\n  plain text extraction, 坐标, 检测框, bbox, bounding box, image to text, screenshot, photo scan,\n  recognize text.\nlicense: Apache-2.0\nmetadata:\n  openclaw:\n    requires:\n      env:\n        - PADDLEOCR_ACCESS_TOKEN\n      bins:\n        - paddleocr\n    primaryEnv: PADDLEOCR_ACCESS_TOKEN\n    emoji: \"🔤\"\n    install:\n      - kind: uv\n        package: paddleocr\n        bins: [paddleocr]\n---\n\n# PaddleOCR Text Recognition\n\n## When to Use This Skill\n\n**Use this skill for**:\n\n- Extract text from images (screenshots, photos, scans)\n- Extract text from PDFs or document images when the goal is **line/box-level text**\n- Extract text from URLs or local files that point to images/PDFs\n\n**Do not use for**:\n\n- Documents with tables, formulas, charts, or complex layouts — use Document Parsing instead\n\n## Usage\n\n### Basic OCR\n\nFrom URL:\n\n```bash\npaddleocr api \\\n  --model_type ocr \\\n  --file_url \"https://example.com/image.png\"\n```\n\nFrom local file:\n\n```bash\npaddleocr api \\\n  --model_type ocr \\\n  --file_path \"./document.pdf\"\n```\n\n### Common Options\n\n```bash\n# With specific model\npaddleocr api \\\n  --model_type ocr \\\n  --model PP-OCRv5 \\\n  --file_path \"./report.pdf\"\n\n# Disable preprocessing (faster, for flat/well-oriented images)\npaddleocr api \\\n  --model_type ocr \\\n  --file_path \"./document.pdf\" \\\n  --use_doc_unwarping False \\\n  --use_doc_orientation_classify False\n\n# Save result to file\npaddleocr api \\\n  --model_type ocr \\\n  --file_url \"https://...\" \\\n  --output result.json\n\n# Page ranges\npaddleocr api \\\n  --model_type ocr \\\n  --file_path \"./large.pdf\" \\\n  --page_ranges \"1-5,10,15-20\"\n```\n\n### Output Format\n\n```json\n{\n  \"jobId\": \"job-xxx\",\n  \"pages\": [\n    {\n      \"prunedResult\": {\n        \"rec_texts\": [\"Line 1\", \"Line 2\"],\n        \"rec_scores\": [0.98, 0.95]\n      },\n      \"ocrImageUrl\": \"https://...\"\n    }\n  ]\n}\n```\n\n## Important Notes\n\n**Preprocessing options**: By default, the API enables document preprocessing (unwarping and orientation classification). For flat, well-oriented images (screenshots, properly scanned documents), you can disable preprocessing for faster results:\n\n```bash\npaddleocr api --model_type ocr --file_path \"./document.pdf\" --use_doc_unwarping False --use_doc_orientation_classify False\n```\n\nKeep preprocessing enabled when:\n- The input is a photo of a curved or folded document\n- The document has significant perspective distortion\n- Orientation is uncertain (rotated 90/180/270 degrees)\n\n**Display complete results**: Always show the full extracted content to users. Do not truncate with \"...\" unless content exceeds 10,000 characters. When multiple pages are processed, summarize if needed but provide complete results when explicitly requested.\n\n**Handle errors gracefully**: When the CLI returns an error, inform the user of the specific issue rather than silently failing or falling back to your own vision capabilities. Common errors:\n- Authentication: `PADDLEOCR_ACCESS_TOKEN` invalid or missing\n- Quota: API rate limit exceeded\n- No content detected: Image may be blank or contain no text\n\n## CLI Reference\n\nRun `paddleocr api --help` for all options.\n\nFor full documentation, see: [PaddleOCR Official Documentation](https://www.paddleocr.ai/latest/en/version3.x/inference_deployment/serving/paddleocr_official_api/cli.html)\n\nFile v2.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn77zppfj1a2fc620aygaf9z9980ewfa\",\n  \"slug\": \"paddleocr-text-recognition\",\n  \"version\": \"2.0.0\",\n  \"publishedAt\": 1780654223964\n}\n\nFile v2.0.0:skill-card.md\n\n## Description:\n\nExtracts machine-readable OCR text from images, screenshots, scans, and scanned PDFs, with line-level text and optional bounding boxes.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[bobholamovic](https://clawhub.ai/user/bobholamovic)\n\n### License/Terms of Use:\n\nApache-2.0\n\n## Use Case:\n\nDevelopers and agent users use this skill to extract plain text from images, screenshots, photos, scans, and scanned PDFs through the PaddleOCR CLI. It is best suited for line-level OCR output and not for tables, formulas, charts, or complex document layouts.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Local images or PDFs may be sent to PaddleOCR's external API without a clear privacy confirmation.\n\nMitigation: Use only approved data with the external OCR service; avoid submitting secrets, regulated records, internal screenshots, or confidential documents unless that service is approved for that data.\n\nRisk: Runtime dependency behavior can change if the PaddleOCR package is installed without controls.\n\nMitigation: Use a pinned dependency version or controlled runtime for higher-risk use.\n\nRisk: OCR output may be incomplete or unsuitable for tables, formulas, charts, and complex document layouts.\n\nMitigation: Use a document parsing workflow for complex layouts and review OCR results before relying on them.\n\n## Reference(s):\n\n- [PaddleOCR Official CLI Documentation](https://www.paddleocr.ai/latest/en/version3.x/inference_deployment/serving/paddleocr_official_api/cli.html)\n- [ClawHub Skill Page](https://clawhub.ai/bobholamovic/skills/paddleocr-text-recognition)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown guidance with shell command examples and JSON OCR result descriptions]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Returns complete extracted content when feasible; examples include line text, recognition scores, optional bounding boxes, page ranges, and saved JSON output.]\n\n## Skill Version(s):\n\n2.0.0 (source: ClawHub release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v1.0.21: 7 files, 14431 bytes\n\nFiles: references/output_schema.md (3128b), scripts/lib.py (12257b), scripts/ocr_caller.py (4475b), scripts/smoke_test.py (4147b), skill-card.md (2324b), SKILL.md (9602b), _meta.json (146b)\n\nFile v1.0.21:SKILL.md\n\n---\nname: paddleocr-text-recognition\ndescription: >-\n  Use this skill whenever the user wants text extracted from images, photos, scans, screenshots,\n  or scanned PDFs. Returns exact machine-readable strings with line-level text and optional bbox\n  coordinates. Strong accuracy for CJK, small print, and handwritten text.\n  Trigger terms: OCR, 文字识别, 图片转文字, 截图识字, 提取图中文字, 扫描识字, 识字, 纯文字,\n  plain text extraction, 坐标, 检测框, bbox, bounding box, image to text, screenshot, photo scan,\n  recognize text.\ncompatibility: Requires Python 3.9+, uv, and internet access.\nmetadata:\n  openclaw:\n    requires:\n      env:\n        - PADDLEOCR_OCR_API_URL\n        - PADDLEOCR_ACCESS_TOKEN\n      bins:\n        - uv\n    primaryEnv: PADDLEOCR_ACCESS_TOKEN\n    emoji: \"🔤\"\n    homepage: https://github.com/PaddlePaddle/PaddleOCR/tree/main/skills/paddleocr-text-recognition\n---\n\n# PaddleOCR Text Recognition Skill\n\n## When to Use This Skill\n\n**Trigger keywords (routing)**: Bilingual trigger terms (Chinese and English) are listed in the YAML `description` above—use that field for discovery and routing.\n\n**Use this skill for**:\n\n- Extract text from images (screenshots, photos, scans)\n- Extract text from PDFs or document images when the goal is **line/box-level text**, not recovering table grids, formulas, or full reading-order layout\n- Extract text from URLs or local files that point to images/PDFs\n\n**Do not use for**:\n\n- Plain text files, code files, or markdown documents that can be read directly as text\n- Documents with tables, formulas, charts, or complex layouts — use Document Parsing instead\n- Tasks that do not involve image-to-text conversion\n\n## Installation\n\nScripts declare their dependencies inline ([PEP 723](https://peps.python.org/pep-0723/)). No separate install step is needed — [uv](https://docs.astral.sh/uv/) resolves dependencies automatically:\n\n```bash\nuv run scripts/ocr_caller.py --help\n```\n\n## How to Use This Skill\n\n> **Working directory**: All `uv run scripts/...` commands below should be run from this skill's root directory (the directory containing this SKILL.md file).\n\n### Basic Workflow\n\n1. **Identify the input source**:\n   - User provides URL: Use the `--file-url` parameter\n   - User provides local file path: Use the `--file-path` parameter\n\n2. **Execute OCR**:\n\n   ```bash\n   uv run scripts/ocr_caller.py --file-url \"URL provided by user\" --pretty\n   ```\n\n   Or for local files:\n\n   ```bash\n   uv run scripts/ocr_caller.py --file-path \"file path\" --pretty\n   ```\n\n   > **Performance note**: Parsing time scales with document complexity. Single-page images typically complete in 1-3 seconds; large PDFs (50+ pages) may take several minutes. Allow adequate time before assuming a timeout.\n\n   **Default behavior: save raw JSON to a temp file**:\n   - If `--output` is omitted, the script saves automatically under the system temp directory\n   - Default path pattern: `<system-temp>/paddleocr/text-recognition/results/result_<timestamp>_<id>.json`\n   - If `--output` is provided, it overrides the default temp-file destination\n   - If `--stdout` is provided, JSON is printed to stdout and no file is saved\n   - In save mode, the script prints the absolute saved path on stderr: `Result saved to: /absolute/path/...`\n   - In default/custom save mode, read and parse the saved JSON file before responding\n   - Use `--stdout` only when you explicitly want to skip file persistence\n\n3. **Parse JSON response**:\n   - In default/custom save mode, load JSON from the saved file path shown by the script\n   - Check the `ok` field: `true` means success, `false` means error\n   - Extract text: `text` field contains all recognized text\n   - If `--stdout` is used, parse the stdout JSON directly\n   - Handle errors: If `ok` is false, display `error.message`\n\n4. **Present results to user**:\n   - Display extracted text in a readable format\n   - If the text is empty, the image may contain no text\n   - In save mode, always tell the user the saved file path and that full raw JSON is available there\n\n### What to Do After Extraction\n\nCommon next steps once you have the recognized text:\n\n- **Save to file**: Write the `text` field to a `.txt` or `.md` file\n- **Search the content**: Search the saved output file for keywords\n- **Feed to another pipeline**: The `text` field is clean plain text, ready for downstream processing\n- **Poor results**: See \"Tips for Better Results\" below before retrying\n\n### Complete Output Display\n\nAlways display the COMPLETE recognized text to the user. The user typically needs the full content for downstream use — truncation silently loses data they may not notice is missing.\n\n- Display the entire `text` field, no matter how long\n- Do not use phrases like \"Here's a summary\" or \"The text begins with...\"\n- Do not truncate with \"...\" unless the text truly exceeds reasonable display limits (>10,000 chars)\n\n**Example - Correct**:\n\n```\nUser: \"Extract the text from this image\"\nAgent: I've extracted the text from the image. Here's the complete content:\n\n[Display the entire text here]\n```\n\n**Example - Incorrect**:\n\n```\nUser: \"Extract the text from this image\"\nAgent: I found some text in the image. Here's a preview:\n\"The quick brown fox...\" (truncated)\n```\n\n### Understanding the Output\n\nThe script returns a JSON envelope with `ok`, `text`, `result`, and `error` fields. Use `text` for the recognized content; `result` contains the raw API response for debugging.\n\nFor the full schema and field-level details, see `references/output_schema.md`.\n\n> Raw result location (default): the temp-file path printed by the script on stderr\n\n### Usage Examples\n\n**Example 1: URL OCR**\n\n```bash\nuv run scripts/ocr_caller.py --file-url \"https://example.com/invoice.jpg\" --pretty\n```\n\n**Example 2: Local File OCR**\n\n```bash\nuv run scripts/ocr_caller.py --file-path \"./document.pdf\" --pretty\n```\n\n**Example 3: OCR With Explicit File Type**\n\n```bash\nuv run scripts/ocr_caller.py --file-url \"https://example.com/input\" --file-type 1 --pretty\n```\n\n- `--file-type 0`: PDF\n- `--file-type 1`: image\n- If omitted, the type is auto-detected from the file extension. For local files, a recognized extension (`.pdf`, `.png`, `.jpg`, `.jpeg`, `.bmp`, `.tiff`, `.tif`, `.webp`) is required; otherwise pass `--file-type` explicitly. For URLs with unrecognized extensions, the service attempts inference.\n\n**Example 4: Print JSON Without Saving**\n\n```bash\nuv run scripts/ocr_caller.py --file-url \"https://example.com/input\" --stdout --pretty\n```\n\n### First-Time Configuration\n\n**When API is not configured**, the script outputs:\n\n```json\n{\n  \"ok\": false,\n  \"text\": \"\",\n  \"result\": null,\n  \"error\": {\n    \"code\": \"CONFIG_ERROR\",\n    \"message\": \"PADDLEOCR_OCR_API_URL not configured. Get your API at: https://paddleocr.com\"\n  }\n}\n```\n\n**Configuration workflow**:\n\n1. **Show the exact error message** to the user.\n\n2. **Guide the user to obtain credentials**: Visit the [PaddleOCR website](https://www.paddleocr.com), click **API**, select the `PP-OCRv5` model, select the language, then copy the `API_URL` and `Token`. They map to these environment variables:\n   - `PADDLEOCR_OCR_API_URL` — full endpoint URL ending with `/ocr`\n   - `PADDLEOCR_ACCESS_TOKEN` — 40-character alphanumeric string\n\n   Optionally configure `PADDLEOCR_OCR_TIMEOUT` for request timeout. Recommend using the host application's standard configuration method rather than pasting credentials in chat.\n\n3. **Apply credentials** — one of:\n   - **User configured via the host UI**: ask the user to confirm, then retry.\n   - **User pastes credentials in chat**: warn that they may be stored in conversation history, help the user persist them using the host's standard configuration method, then retry.\n\n### Error Handling\n\nAll errors return JSON with `ok: false`. Show the error message and stop — do not fall back to your own vision capabilities. Identify the issue from `error.code` and `error.message`:\n\n**Authentication failed (403)** — `error.message` contains \"Authentication failed\"\n\n- Token is invalid, reconfigure with correct credentials\n\n**Quota exceeded (429)** — `error.message` contains \"API rate limit exceeded\"\n\n- Daily API quota exhausted, inform user to wait or upgrade\n\n**Unsupported format** — `error.message` contains \"Unsupported file format\"\n\n- File format not supported, convert to PDF/PNG/JPG\n\n**No text detected**:\n\n- `text` field is empty\n- Image may be blank, corrupted, or contain no text\n\n### Tips for Better Results\n\nIf recognition quality is poor:\n\n- **Low resolution**: Provide a higher resolution image (≥300 DPI works well for most printed text)\n- **Noisy background**: A cleaner scan or screenshot typically yields better results than a phone photo\n- **Check confidence**: The raw JSON (`result.result.ocrResults[n].prunedResult.rec_scores`) shows per-line confidence scores — low values identify uncertain regions worth reviewing\n\n## Reference Documentation\n\n- `references/output_schema.md` — Full output schema, field descriptions, and command examples\n\n> **Note**: Model version, capabilities, and supported file formats are determined by your API endpoint (`PADDLEOCR_OCR_API_URL`) and its official API documentation.\n\n## Testing the Skill\n\nTo verify the skill is working properly:\n\n```bash\nuv run scripts/smoke_test.py\nuv run scripts/smoke_test.py --skip-api-test\nuv run scripts/smoke_test.py --test-url \"https://...\"\n```\n\nThe first form tests configuration and API connectivity. `--skip-api-test` checks configuration only. `--test-url` overrides the default sample image URL.\n\nFile v1.0.21:_meta.json\n\n{\n  \"ownerId\": \"kn77zppfj1a2fc620aygaf9z9980ewfa\",\n  \"slug\": \"paddleocr-text-recognition\",\n  \"version\": \"1.0.21\",\n  \"publishedAt\": 1775190380432\n}\n\nFile v1.0.21:references/output_schema.md\n\n# PaddleOCR Text Recognition Output Schema\n\nThis document defines the output envelope returned by `ocr_caller.py`.\n\nBy default, `ocr_caller.py` saves the JSON envelope to a unique file under the system temp directory and prints the absolute saved path to `stderr`. Use `--output` when you need a custom destination, or `--stdout` when you want to skip file saving and print JSON directly.\n\n## Output Envelope\n\n`ocr_caller.py` wraps provider response in a stable structure:\n\n```json\n{\n  \"ok\": true,\n  \"text\": \"Extracted text from all pages\",\n  \"result\": { ... },  // raw provider response\n  \"error\": null\n}\n```\n\nOn error:\n\n```json\n{\n  \"ok\": false,\n  \"text\": \"\",\n  \"result\": null,\n  \"error\": {\n    \"code\": \"ERROR_CODE\",\n    \"message\": \"Human-readable message\"\n  }\n}\n```\n\n## Error Codes\n\n| Code           | Description                                                                |\n| -------------- | -------------------------------------------------------------------------- |\n| `INPUT_ERROR`  | Invalid or unusable input (arguments, file source, format, types).         |\n| `CONFIG_ERROR` | Missing or invalid API / client configuration.                             |\n| `API_ERROR`    | Request or response handling failed (network, HTTP, body parsing, schema). |\n\n## Raw Result Notes\n\nThe `result` field contains raw provider output.  \nRaw fields may vary by model version and endpoint.\n\n## Raw Result Example\n\n```json\n{\n  \"logId\": \"request-uuid\",\n  \"errorCode\": 0,\n  \"errorMsg\": \"Success\",\n  \"result\": {\n    \"ocrResults\": [\n      {\n        \"prunedResult\": {\n          \"rec_texts\": [\"First line\", \"Second line\"],\n          \"rec_scores\": [0.98, 0.95],\n          \"...\": \"other OCR fields\"\n        },\n        \"ocrImage\": \"https://...\",\n        \"inputImage\": \"https://...\",\n        \"...\": \"other model-specific fields\"\n      }\n    ],\n    \"dataInfo\": {\n      \"numPages\": 1,\n      \"type\": \"pdf\",\n      \"...\": \"other metadata\"\n    },\n    \"...\": \"other top-level fields\"\n  }\n}\n```\n\n## Stable Fields for Downstream Use\n\nPaths are relative to the output envelope root.\n\n- `result.result.ocrResults[n].prunedResult`  \n  Structured OCR data for page `n`.\n\n- `result.result.ocrResults[n].prunedResult.rec_texts`  \n  Recognized text lines for page `n`.\n\n- `result.result.ocrResults[n].prunedResult.rec_scores`  \n  Confidence scores for recognized text lines.\n\n## Text Extraction\n\n`ocr_caller.py` extracts top-level `text` from `result.result.ocrResults[n].prunedResult.rec_texts`, joins lines with `\\n`, and joins pages with `\\n\\n`.\n\n## Command Examples\n\n```bash\n# OCR from URL (result auto-saves to the system temp directory)\nuv run scripts/ocr_caller.py --file-url \"URL\" --pretty\n\n# OCR local file (result auto-saves to the system temp directory)\nuv run scripts/ocr_caller.py --file-path \"doc.pdf\" --pretty\n\n# OCR with explicit file type\nuv run scripts/ocr_caller.py --file-url \"URL\" --file-type 1 --pretty\n\n# Save result to a custom file path\nuv run scripts/ocr_caller.py --file-url \"URL\" --output \"./result.json\" --pretty\n\n# Print JSON to stdout without saving a file\nuv run scripts/ocr_caller.py --file-url \"URL\" --stdout --pretty\n```\n\nFile v1.0.21:skill-card.md\n\n## Description: <br>\nExtracts exact machine-readable text from images, photos, scans, screenshots, and scanned PDFs, with line-level text and optional bounding box details. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[bobholamovic](https://clawhub.ai/user/bobholamovic) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers, operators, and end users use this skill to run PaddleOCR text recognition on image or PDF inputs and return complete extracted text for downstream review, search, saving, or processing. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: OCR inputs and recognized text may contain sensitive content and are sent to the configured PaddleOCR endpoint. <br>\nMitigation: Use an approved endpoint for sensitive documents, configure credentials through the host application's standard secret handling, and avoid sharing raw OCR results or logs. <br>\nRisk: Saved OCR result files can contain full extracted text, raw provider data, source URLs, request identifiers, and local paths. <br>\nMitigation: Use stdout mode for sensitive one-off documents or delete the saved temp JSON after use, and limit access to any persisted result files. <br>\n\n\n## Reference(s): <br>\n- [PaddleOCR Text Recognition skill source](https://github.com/PaddlePaddle/PaddleOCR/tree/main/skills/paddleocr-text-recognition) <br>\n- [Output schema](references/output_schema.md) <br>\n- [PaddleOCR API website](https://www.paddleocr.com) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [Text, JSON, Shell commands, Guidance] <br>\n**Output Format:** [Plain text extracted from OCR plus a JSON envelope saved to a file or printed to stdout] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [May include line-level recognized text, confidence scores, bounding box details, source URLs, request identifiers, and provider raw result fields.] <br>\n\n## Skill Version(s): <br>\n1.0.21 (source: server release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.0.20: 6 files, 13200 bytes\n\nFiles: references/output_schema.md (3100b), scripts/lib.py (12257b), scripts/ocr_caller.py (4475b), scripts/smoke_test.py (4147b), SKILL.md (9622b), _meta.json (146b)\n\nFile v1.0.20:SKILL.md\n\n---\nname: paddleocr-text-recognition\ndescription: >-\n  Use this skill whenever the user wants text extracted from images, photos, scans, screenshots,\n  or scanned PDFs. Returns exact machine-readable strings with line-level text and optional bbox\n  coordinates. Strong accuracy for CJK, small print, and handwritten text.\n  Trigger terms: OCR, 文字识别, 图片转文字, 截图识字, 提取图中文字, 扫描识字, 识字, 纯文字,\n  plain text extraction, 坐标, 检测框, bbox, bounding box, image to text, screenshot, photo scan,\n  recognize text.\nlicense: Apache-2.0\ncompatibility: Requires Python 3.9+, uv, and internet access.\nmetadata:\n  openclaw:\n    requires:\n      env:\n        - PADDLEOCR_OCR_API_URL\n        - PADDLEOCR_ACCESS_TOKEN\n      bins:\n        - uv\n    primaryEnv: PADDLEOCR_ACCESS_TOKEN\n    emoji: \"🔤\"\n    homepage: https://github.com/PaddlePaddle/PaddleOCR/tree/main/skills/paddleocr-text-recognition\n---\n\n# PaddleOCR Text Recognition Skill\n\n## When to Use This Skill\n\n**Trigger keywords (routing)**: Bilingual trigger terms (Chinese and English) are listed in the YAML `description` above—use that field for discovery and routing.\n\n**Use this skill for**:\n\n- Extract text from images (screenshots, photos, scans)\n- Extract text from PDFs or document images when the goal is **line/box-level text**, not recovering table grids, formulas, or full reading-order layout\n- Extract text from URLs or local files that point to images/PDFs\n\n**Do not use for**:\n\n- Plain text files, code files, or markdown documents that can be read directly as text\n- Documents with tables, formulas, charts, or complex layouts — use Document Parsing instead\n- Tasks that do not involve image-to-text conversion\n\n## Installation\n\nScripts declare their dependencies inline ([PEP 723](https://peps.python.org/pep-0723/)). No separate install step is needed — [uv](https://docs.astral.sh/uv/) resolves dependencies automatically:\n\n```bash\nuv run scripts/ocr_caller.py --help\n```\n\n## How to Use This Skill\n\n> **Working directory**: All `uv run scripts/...` commands below should be run from this skill's root directory (the directory containing this SKILL.md file).\n\n### Basic Workflow\n\n1. **Identify the input source**:\n   - User provides URL: Use the `--file-url` parameter\n   - User provides local file path: Use the `--file-path` parameter\n\n2. **Execute OCR**:\n\n   ```bash\n   uv run scripts/ocr_caller.py --file-url \"URL provided by user\" --pretty\n   ```\n\n   Or for local files:\n\n   ```bash\n   uv run scripts/ocr_caller.py --file-path \"file path\" --pretty\n   ```\n\n   > **Performance note**: Parsing time scales with document complexity. Single-page images typically complete in 1-3 seconds; large PDFs (50+ pages) may take several minutes. Allow adequate time before assuming a timeout.\n\n   **Default behavior: save raw JSON to a temp file**:\n   - If `--output` is omitted, the script saves automatically under the system temp directory\n   - Default path pattern: `<system-temp>/paddleocr/text-recognition/results/result_<timestamp>_<id>.json`\n   - If `--output` is provided, it overrides the default temp-file destination\n   - If `--stdout` is provided, JSON is printed to stdout and no file is saved\n   - In save mode, the script prints the absolute saved path on stderr: `Result saved to: /absolute/path/...`\n   - In default/custom save mode, read and parse the saved JSON file before responding\n   - Use `--stdout` only when you explicitly want to skip file persistence\n\n3. **Parse JSON response**:\n   - In default/custom save mode, load JSON from the saved file path shown by the script\n   - Check the `ok` field: `true` means success, `false` means error\n   - Extract text: `text` field contains all recognized text\n   - If `--stdout` is used, parse the stdout JSON directly\n   - Handle errors: If `ok` is false, display `error.message`\n\n4. **Present results to user**:\n   - Display extracted text in a readable format\n   - If the text is empty, the image may contain no text\n   - In save mode, always tell the user the saved file path and that full raw JSON is available there\n\n### What to Do After Extraction\n\nCommon next steps once you have the recognized text:\n\n- **Save to file**: Write the `text` field to a `.txt` or `.md` file\n- **Search the content**: Search the saved output file for keywords\n- **Feed to another pipeline**: The `text` field is clean plain text, ready for downstream processing\n- **Poor results**: See \"Tips for Better Results\" below before retrying\n\n### Complete Output Display\n\nAlways display the COMPLETE recognized text to the user. The user typically needs the full content for downstream use — truncation silently loses data they may not notice is missing.\n\n- Display the entire `text` field, no matter how long\n- Do not use phrases like \"Here's a summary\" or \"The text begins with...\"\n- Do not truncate with \"...\" unless the text truly exceeds reasonable display limits (>10,000 chars)\n\n**Example - Correct**:\n\n```\nUser: \"Extract the text from this image\"\nAgent: I've extracted the text from the image. Here's the complete content:\n\n[Display the entire text here]\n```\n\n**Example - Incorrect**:\n\n```\nUser: \"Extract the text from this image\"\nAgent: I found some text in the image. Here's a preview:\n\"The quick brown fox...\" (truncated)\n```\n\n### Understanding the Output\n\nThe script returns a JSON envelope with `ok`, `text`, `result`, and `error` fields. Use `text` for the recognized content; `result` contains the raw API response for debugging.\n\nFor the full schema and field-level details, see `references/output_schema.md`.\n\n> Raw result location (default): the temp-file path printed by the script on stderr\n\n### Usage Examples\n\n**Example 1: URL OCR**\n\n```bash\nuv run scripts/ocr_caller.py --file-url \"https://example.com/invoice.jpg\" --pretty\n```\n\n**Example 2: Local File OCR**\n\n```bash\nuv run scripts/ocr_caller.py --file-path \"./document.pdf\" --pretty\n```\n\n**Example 3: OCR With Explicit File Type**\n\n```bash\nuv run scripts/ocr_caller.py --file-url \"https://example.com/input\" --file-type 1 --pretty\n```\n\n- `--file-type 0`: PDF\n- `--file-type 1`: image\n- If omitted, the type is auto-detected from the file extension. For local files, a recognized extension (`.pdf`, `.png`, `.jpg`, `.jpeg`, `.bmp`, `.tiff`, `.tif`, `.webp`) is required; otherwise pass `--file-type` explicitly. For URLs with unrecognized extensions, the service attempts inference.\n\n**Example 4: Print JSON Without Saving**\n\n```bash\nuv run scripts/ocr_caller.py --file-url \"https://example.com/input\" --stdout --pretty\n```\n\n### First-Time Configuration\n\n**When API is not configured**, the script outputs:\n\n```json\n{\n  \"ok\": false,\n  \"text\": \"\",\n  \"result\": null,\n  \"error\": {\n    \"code\": \"CONFIG_ERROR\",\n    \"message\": \"PADDLEOCR_OCR_API_URL not configured. Get your API at: https://paddleocr.com\"\n  }\n}\n```\n\n**Configuration workflow**:\n\n1. **Show the exact error message** to the user.\n\n2. **Guide the user to obtain credentials**: Visit the [PaddleOCR website](https://www.paddleocr.com), click **API**, select the `PP-OCRv5` model, select the language, then copy the `API_URL` and `Token`. They map to these environment variables:\n   - `PADDLEOCR_OCR_API_URL` — full endpoint URL ending with `/ocr`\n   - `PADDLEOCR_ACCESS_TOKEN` — 40-character alphanumeric string\n\n   Optionally configure `PADDLEOCR_OCR_TIMEOUT` for request timeout. Recommend using the host application's standard configuration method rather than pasting credentials in chat.\n\n3. **Apply credentials** — one of:\n   - **User configured via the host UI**: ask the user to confirm, then retry.\n   - **User pastes credentials in chat**: warn that they may be stored in conversation history, help the user persist them using the host's standard configuration method, then retry.\n\n### Error Handling\n\nAll errors return JSON with `ok: false`. Show the error message and stop — do not fall back to your own vision capabilities. Identify the issue from `error.code` and `error.message`:\n\n**Authentication failed (403)** — `error.message` contains \"Authentication failed\"\n\n- Token is invalid, reconfigure with correct credentials\n\n**Quota exceeded (429)** — `error.message` contains \"API rate limit exceeded\"\n\n- Daily API quota exhausted, inform user to wait or upgrade\n\n**Unsupported format** — `error.message` contains \"Unsupported file format\"\n\n- File format not supported, convert to PDF/PNG/JPG\n\n**No text detected**:\n\n- `text` field is empty\n- Image may be blank, corrupted, or contain no text\n\n### Tips for Better Results\n\nIf recognition quality is poor:\n\n- **Low resolution**: Provide a higher resolution image (≥300 DPI works well for most printed text)\n- **Noisy background**: A cleaner scan or screenshot typically yields better results than a phone photo\n- **Check confidence**: The raw JSON (`result.result.ocrResults[n].prunedResult.rec_scores`) shows per-line confidence scores — low values identify uncertain regions worth reviewing\n\n## Reference Documentation\n\n- `references/output_schema.md` — Full output schema, field descriptions, and command examples\n\n> **Note**: Model version, capabilities, and supported file formats are determined by your API endpoint (`PADDLEOCR_OCR_API_URL`) and its official API documentation.\n\n## Testing the Skill\n\nTo verify the skill is working properly:\n\n```bash\nuv run scripts/smoke_test.py\nuv run scripts/smoke_test.py --skip-api-test\nuv run scripts/smoke_test.py --test-url \"https://...\"\n```\n\nThe first form tests configuration and API connectivity. `--skip-api-test` checks configuration only. `--test-url` overrides the default sample image URL.\n\nFile v1.0.20:_meta.json\n\n{\n  \"ownerId\": \"kn77zppfj1a2fc620aygaf9z9980ewfa\",\n  \"slug\": \"paddleocr-text-recognition\",\n  \"version\": \"1.0.20\",\n  \"publishedAt\": 1775190341183\n}\n\nFile v1.0.20:references/output_schema.md\n\n# PaddleOCR Text Recognition Output Schema\n\nThis document defines the output envelope returned by `ocr_caller.py`.\n\nBy default, `ocr_caller.py` saves the JSON envelope to a unique file under the system temp directory and prints the absolute saved path to `stderr`. Use `--output` when you need a custom destination, or `--stdout` when you want to skip file saving and print JSON directly.\n\n## Output Envelope\n\n`ocr_caller.py` wraps provider response in a stable structure:\n\n```json\n{\n  \"ok\": true,\n  \"text\": \"Extracted text from all pages\",\n  \"result\": { ... },  // raw provider response\n  \"error\": null\n}\n```\n\nOn error:\n\n```json\n{\n  \"ok\": false,\n  \"text\": \"\",\n  \"result\": null,\n  \"error\": {\n    \"code\": \"ERROR_CODE\",\n    \"message\": \"Human-readable message\"\n  }\n}\n```\n\n## Error Codes\n\n| Code           | Description                                                                 |\n| -------------- | --------------------------------------------------------------------------- |\n| `INPUT_ERROR`  | Invalid or unusable input (arguments, file source, format, types). |\n| `CONFIG_ERROR` | Missing or invalid API / client configuration.       |\n| `API_ERROR`    | Request or response handling failed (network, HTTP, body parsing, schema). |\n\n## Raw Result Notes\n\nThe `result` field contains raw provider output.  \nRaw fields may vary by model version and endpoint.\n\n## Raw Result Example\n\n```json\n{\n  \"logId\": \"request-uuid\",\n  \"errorCode\": 0,\n  \"errorMsg\": \"Success\",\n  \"result\": {\n    \"ocrResults\": [\n      {\n        \"prunedResult\": {\n          \"rec_texts\": [\"First line\", \"Second line\"],\n          \"rec_scores\": [0.98, 0.95],\n          \"...\": \"other OCR fields\"\n        },\n        \"ocrImage\": \"https://...\",\n        \"inputImage\": \"https://...\",\n        \"...\": \"other model-specific fields\"\n      }\n    ],\n    \"dataInfo\": {\n      \"numPages\": 1,\n      \"type\": \"pdf\",\n      \"...\": \"other metadata\"\n    },\n    \"...\": \"other top-level fields\"\n  }\n}\n```\n\n## Stable Fields for Downstream Use\n\nPaths are relative to the output envelope root.\n\n- `result.result.ocrResults[n].prunedResult`  \n  Structured OCR data for page `n`.\n\n- `result.result.ocrResults[n].prunedResult.rec_texts`  \n  Recognized text lines for page `n`.\n\n- `result.result.ocrResults[n].prunedResult.rec_scores`  \n  Confidence scores for recognized text lines.\n\n## Text Extraction\n\n`ocr_caller.py` extracts top-level `text` from `result.result.ocrResults[n].prunedResult.rec_texts`, joins lines with `\\n`, and joins pages with `\\n\\n`.\n\n## Command Examples\n\n```bash\n# OCR from URL (result auto-saves to the system temp directory)\nuv run scripts/ocr_caller.py --file-url \"URL\" --pretty\n\n# OCR local file (result auto-saves to the system temp directory)\nuv run scripts/ocr_caller.py --file-path \"doc.pdf\" --pretty\n\n# OCR with explicit file type\nuv run scripts/ocr_caller.py --file-url \"URL\" --file-type 1 --pretty\n\n# Save result to a custom file path\nuv run scripts/ocr_caller.py --file-url \"URL\" --output \"./result.json\" --pretty\n\n# Print JSON to stdout without saving a file\nuv run scripts/ocr_caller.py --file-url \"URL\" --stdout --pretty\n```\n\nArchive v1.0.19: 6 files, 13200 bytes\n\nFiles: references/output_schema.md (3100b), scripts/lib.py (12257b), scripts/ocr_caller.py (4475b), scripts/smoke_test.py (4147b), SKILL.md (9622b), _meta.json (146b)\n\nFile v1.0.19:SKILL.md\n\n---\nname: paddleocr-text-recognition\ndescription: >-\n  Use this skill whenever the user wants text extracted from images, photos, scans, screenshots,\n  or scanned PDFs. Returns exact machine-readable strings with line-level text and optional bbox\n  coordinates. Strong accuracy for CJK, small print, and handwritten text.\n  Trigger terms: OCR, 文字识别, 图片转文字, 截图识字, 提取图中文字, 扫描识字, 识字, 纯文字,\n  plain text extraction, 坐标, 检测框, bbox, bounding box, image to text, screenshot, photo scan,\n  recognize text.\nlicense: Apache-2.0\ncompatibility: Requires Python 3.9+, uv, and internet access.\nmetadata:\n  openclaw:\n    requires:\n      env:\n        - PADDLEOCR_OCR_API_URL\n        - PADDLEOCR_ACCESS_TOKEN\n      bins:\n        - uv\n    primaryEnv: PADDLEOCR_ACCESS_TOKEN\n    emoji: \"🔤\"\n    homepage: https://github.com/PaddlePaddle/PaddleOCR/tree/main/skills/paddleocr-text-recognition\n---\n\n# PaddleOCR Text Recognition Skill\n\n## When to Use This Skill\n\n**Trigger keywords (routing)**: Bilingual trigger terms (Chinese and English) are listed in the YAML `description` above—use that field for discovery and routing.\n\n**Use this skill for**:\n\n- Extract text from images (screenshots, photos, scans)\n- Extract text from PDFs or document images when the goal is **line/box-level text**, not recovering table grids, formulas, or full reading-order layout\n- Extract text from URLs or local files that point to images/PDFs\n\n**Do not use for**:\n\n- Plain text files, code files, or markdown documents that can be read directly as text\n- Documents with tables, formulas, charts, or complex layouts — use Document Parsing instead\n- Tasks that do not involve image-to-text conversion\n\n## Installation\n\nScripts declare their dependencies inline ([PEP 723](https://peps.python.org/pep-0723/)). No separate install step is needed — [uv](https://docs.astral.sh/uv/) resolves dependencies automatically:\n\n```bash\nuv run scripts/ocr_caller.py --help\n```\n\n## How to Use This Skill\n\n> **Working directory**: All `uv run scripts/...` commands below should be run from this skill's root directory (the directory containing this SKILL.md file).\n\n### Basic Workflow\n\n1. **Identify the input source**:\n   - User provides URL: Use the `--file-url` parameter\n   - User provides local file path: Use the `--file-path` parameter\n\n2. **Execute OCR**:\n\n   ```bash\n   uv run scripts/ocr_caller.py --file-url \"URL provided by user\" --pretty\n   ```\n\n   Or for local files:\n\n   ```bash\n   uv run scripts/ocr_caller.py --file-path \"file path\" --pretty\n   ```\n\n   > **Performance note**: Parsing time scales with document complexity. Single-page images typically complete in 1-3 seconds; large PDFs (50+ pages) may take several minutes. Allow adequate time before assuming a timeout.\n\n   **Default behavior: save raw JSON to a temp file**:\n   - If `--output` is omitted, the script saves automatically under the system temp directory\n   - Default path pattern: `<system-temp>/paddleocr/text-recognition/results/result_<timestamp>_<id>.json`\n   - If `--output` is provided, it overrides the default temp-file destination\n   - If `--stdout` is provided, JSON is printed to stdout and no file is saved\n   - In save mode, the script prints the absolute saved path on stderr: `Result saved to: /absolute/path/...`\n   - In default/custom save mode, read and parse the saved JSON file before responding\n   - Use `--stdout` only when you explicitly want to skip file persistence\n\n3. **Parse JSON response**:\n   - In default/custom save mode, load JSON from the saved file path shown by the script\n   - Check the `ok` field: `true` means success, `false` means error\n   - Extract text: `text` field contains all recognized text\n   - If `--stdout` is used, parse the stdout JSON directly\n   - Handle errors: If `ok` is false, display `error.message`\n\n4. **Present results to user**:\n   - Display extracted text in a readable format\n   - If the text is empty, the image may contain no text\n   - In save mode, always tell the user the saved file path and that full raw JSON is available there\n\n### What to Do After Extraction\n\nCommon next steps once you have the recognized text:\n\n- **Save to file**: Write the `text` field to a `.txt` or `.md` file\n- **Search the content**: Search the saved output file for keywords\n- **Feed to another pipeline**: The `text` field is clean plain text, ready for downstream processing\n- **Poor results**: See \"Tips for Better Results\" below before retrying\n\n### Complete Output Display\n\nAlways display the COMPLETE recognized text to the user. The user typically needs the full content for downstream use — truncation silently loses data they may not notice is missing.\n\n- Display the entire `text` field, no matter how long\n- Do not use phrases like \"Here's a summary\" or \"The text begins with...\"\n- Do not truncate with \"...\" unless the text truly exceeds reasonable display limits (>10,000 chars)\n\n**Example - Correct**:\n\n```\nUser: \"Extract the text from this image\"\nAgent: I've extracted the text from the image. Here's the complete content:\n\n[Display the entire text here]\n```\n\n**Example - Incorrect**:\n\n```\nUser: \"Extract the text from this image\"\nAgent: I found some text in the image. Here's a preview:\n\"The quick brown fox...\" (truncated)\n```\n\n### Understanding the Output\n\nThe script returns a JSON envelope with `ok`, `text`, `result`, and `error` fields. Use `text` for the recognized content; `result` contains the raw API response for debugging.\n\nFor the full schema and field-level details, see `references/output_schema.md`.\n\n> Raw result location (default): the temp-file path printed by the script on stderr\n\n### Usage Examples\n\n**Example 1: URL OCR**\n\n```bash\nuv run scripts/ocr_caller.py --file-url \"https://example.com/invoice.jpg\" --pretty\n```\n\n**Example 2: Local File OCR**\n\n```bash\nuv run scripts/ocr_caller.py --file-path \"./document.pdf\" --pretty\n```\n\n**Example 3: OCR With Explicit File Type**\n\n```bash\nuv run scripts/ocr_caller.py --file-url \"https://example.com/input\" --file-type 1 --pretty\n```\n\n- `--file-type 0`: PDF\n- `--file-type 1`: image\n- If omitted, the type is auto-detected from the file extension. For local files, a recognized extension (`.pdf`, `.png`, `.jpg`, `.jpeg`, `.bmp`, `.tiff`, `.tif`, `.webp`) is required; otherwise pass `--file-type` explicitly. For URLs with unrecognized extensions, the service attempts inference.\n\n**Example 4: Print JSON Without Saving**\n\n```bash\nuv run scripts/ocr_caller.py --file-url \"https://example.com/input\" --stdout --pretty\n```\n\n### First-Time Configuration\n\n**When API is not configured**, the script outputs:\n\n```json\n{\n  \"ok\": false,\n  \"text\": \"\",\n  \"result\": null,\n  \"error\": {\n    \"code\": \"CONFIG_ERROR\",\n    \"message\": \"PADDLEOCR_OCR_API_URL not configured. Get your API at: https://paddleocr.com\"\n  }\n}\n```\n\n**Configuration workflow**:\n\n1. **Show the exact error message** to the user.\n\n2. **Guide the user to obtain credentials**: Visit the [PaddleOCR website](https://www.paddleocr.com), click **API**, select the `PP-OCRv5` model, select the language, then copy the `API_URL` and `Token`. They map to these environment variables:\n   - `PADDLEOCR_OCR_API_URL` — full endpoint URL ending with `/ocr`\n   - `PADDLEOCR_ACCESS_TOKEN` — 40-character alphanumeric string\n\n   Optionally configure `PADDLEOCR_OCR_TIMEOUT` for request timeout. Recommend using the host application's standard configuration method rather than pasting credentials in chat.\n\n3. **Apply credentials** — one of:\n   - **User configured via the host UI**: ask the user to confirm, then retry.\n   - **User pastes credentials in chat**: warn that they may be stored in conversation history, help the user persist them using the host's standard configuration method, then retry.\n\n### Error Handling\n\nAll errors return JSON with `ok: false`. Show the error message and stop — do not fall back to your own vision capabilities. Identify the issue from `error.code` and `error.message`:\n\n**Authentication failed (403)** — `error.message` contains \"Authentication failed\"\n\n- Token is invalid, reconfigure with correct credentials\n\n**Quota exceeded (429)** — `error.message` contains \"API rate limit exceeded\"\n\n- Daily API quota exhausted, inform user to wait or upgrade\n\n**Unsupported format** — `error.message` contains \"Unsupported file format\"\n\n- File format not supported, convert to PDF/PNG/JPG\n\n**No text detected**:\n\n- `text` field is empty\n- Image may be blank, corrupted, or contain no text\n\n### Tips for Better Results\n\nIf recognition quality is poor:\n\n- **Low resolution**: Provide a higher resolution image (≥300 DPI works well for most printed text)\n- **Noisy background**: A cleaner scan or screenshot typically yields better results than a phone photo\n- **Check confidence**: The raw JSON (`result.result.ocrResults[n].prunedResult.rec_scores`) shows per-line confidence scores — low values identify uncertain regions worth reviewing\n\n## Reference Documentation\n\n- `references/output_schema.md` — Full output schema, field descriptions, and command examples\n\n> **Note**: Model version, capabilities, and supported file formats are determined by your API endpoint (`PADDLEOCR_OCR_API_URL`) and its official API documentation.\n\n## Testing the Skill\n\nTo verify the skill is working properly:\n\n```bash\nuv run scripts/smoke_test.py\nuv run scripts/smoke_test.py --skip-api-test\nuv run scripts/smoke_test.py --test-url \"https://...\"\n```\n\nThe first form tests configuration and API connectivity. `--skip-api-test` checks configuration only. `--test-url` overrides the default sample image URL.\n\nFile v1.0.19:_meta.json\n\n{\n  \"ownerId\": \"kn77zppfj1a2fc620aygaf9z9980ewfa\",\n  \"slug\": \"paddleocr-text-recognition\",\n  \"version\": \"1.0.19\",\n  \"publishedAt\": 1775190085478\n}\n\nFile v1.0.19:references/output_schema.md\n\n# PaddleOCR Text Recognition Output Schema\n\nThis document defines the output envelope returned by `ocr_caller.py`.\n\nBy default, `ocr_caller.py` saves the JSON envelope to a unique file under the system temp directory and prints the absolute saved path to `stderr`. Use `--output` when you need a custom destination, or `--stdout` when you want to skip file saving and print JSON directly.\n\n## Output Envelope\n\n`ocr_caller.py` wraps provider response in a stable structure:\n\n```json\n{\n  \"ok\": true,\n  \"text\": \"Extracted text from all pages\",\n  \"result\": { ... },  // raw provider response\n  \"error\": null\n}\n```\n\nOn error:\n\n```json\n{\n  \"ok\": false,\n  \"text\": \"\",\n  \"result\": null,\n  \"error\": {\n    \"code\": \"ERROR_CODE\",\n    \"message\": \"Human-readable message\"\n  }\n}\n```\n\n## Error Codes\n\n| Code           | Description                                                                 |\n| -------------- | --------------------------------------------------------------------------- |\n| `INPUT_ERROR`  | Invalid or unusable input (arguments, file source, format, types). |\n| `CONFIG_ERROR` | Missing or invalid API / client configuration.       |\n| `API_ERROR`    | Request or response handling failed (network, HTTP, body parsing, schema). |\n\n## Raw Result Notes\n\nThe `result` field contains raw provider output.  \nRaw fields may vary by model version and endpoint.\n\n## Raw Result Example\n\n```json\n{\n  \"logId\": \"request-uuid\",\n  \"errorCode\": 0,\n  \"errorMsg\": \"Success\",\n  \"result\": {\n    \"ocrResults\": [\n      {\n        \"prunedResult\": {\n          \"rec_texts\": [\"First line\", \"Second line\"],\n          \"rec_scores\": [0.98, 0.95],\n          \"...\": \"other OCR fields\"\n        },\n        \"ocrImage\": \"https://...\",\n        \"inputImage\": \"https://...\",\n        \"...\": \"other model-specific fields\"\n      }\n    ],\n    \"dataInfo\": {\n      \"numPages\": 1,\n      \"type\": \"pdf\",\n      \"...\": \"other metadata\"\n    },\n    \"...\": \"other top-level fields\"\n  }\n}\n```\n\n## Stable Fields for Downstream Use\n\nPaths are relative to the output envelope root.\n\n- `result.result.ocrResults[n].prunedResult`  \n  Structured OCR data for page `n`.\n\n- `result.result.ocrResults[n].prunedResult.rec_texts`  \n  Recognized text lines for page `n`.\n\n- `result.result.ocrResults[n].prunedResult.rec_scores`  \n  Confidence scores for recognized text lines.\n\n## Text Extraction\n\n`ocr_caller.py` extracts top-level `text` from `result.result.ocrResults[n].prunedResult.rec_texts`, joins lines with `\\n`, and joins pages with `\\n\\n`.\n\n## Command Examples\n\n```bash\n# OCR from URL (result auto-saves to the system temp directory)\nuv run scripts/ocr_caller.py --file-url \"URL\" --pretty\n\n# OCR local file (result auto-saves to the system temp directory)\nuv run scripts/ocr_caller.py --file-path \"doc.pdf\" --pretty\n\n# OCR with explicit file type\nuv run scripts/ocr_caller.py --file-url \"URL\" --file-type 1 --pretty\n\n# Save result to a custom file path\nuv run scripts/ocr_caller.py --file-url \"URL\" --output \"./result.json\" --pretty\n\n# Print JSON to stdout without saving a file\nuv run scripts/ocr_caller.py --file-url \"URL\" --stdout --pretty\n```\n\nArchive v1.0.18: 7 files, 13104 bytes\n\nFiles: references/output_schema.md (3100b), requirements.txt (57b), scripts/lib.py (12257b), scripts/ocr_caller.py (4206b), scripts/smoke_test.py (4100b), SKILL.md (9458b), _meta.json (146b)\n\nFile v1.0.18:SKILL.md\n\n---\nname: paddleocr-text-recognition\ndescription: >-\n  Use this skill whenever the user wants text extracted from images, photos, scans, screenshots,\n  or scanned PDFs. Returns exact machine-readable strings with line-level text and optional bbox\n  coordinates. Strong accuracy for CJK, small print, and handwritten text.\n  Trigger terms: OCR, 文字识别, 图片转文字, 截图识字, 提取图中文字, 扫描识字, 识字, 纯文字,\n  plain text extraction, 坐标, 检测框, bbox, bounding box, image to text, screenshot, photo scan,\n  recognize text.\nmetadata:\n  openclaw:\n    requires:\n      env:\n        - PADDLEOCR_OCR_API_URL\n        - PADDLEOCR_ACCESS_TOKEN\n      bins:\n        - python\n    primaryEnv: PADDLEOCR_ACCESS_TOKEN\n    emoji: \"🔤\"\n    homepage: https://github.com/PaddlePaddle/PaddleOCR/tree/main/skills/paddleocr-text-recognition\n---\n\n# PaddleOCR Text Recognition Skill\n\n## When to Use This Skill\n\n**Trigger keywords (routing)**: Bilingual trigger terms (Chinese and English) are listed in the YAML `description` above—use that field for discovery and routing.\n\n**Use this skill for**:\n\n- Extract text from images (screenshots, photos, scans)\n- Extract text from PDFs or document images when the goal is **line/box-level text**, not recovering table grids, formulas, or full reading-order layout\n- Extract text from URLs or local files that point to images/PDFs\n\n**Do not use for**:\n\n- Plain text files, code files, or markdown documents that can be read directly as text\n- Documents with tables, formulas, charts, or complex layouts — use Document Parsing instead\n- Tasks that do not involve image-to-text conversion\n\n## Installation\n\nInstall Python dependencies before using this skill. From the skill directory (`skills/paddleocr-text-recognition`):\n\n```bash\npip install -r requirements.txt\n```\n\n## How to Use This Skill\n\n> **Working directory**: All `python scripts/...` commands below should be run from this skill's root directory (the directory containing this SKILL.md file).\n\n### Basic Workflow\n\n1. **Identify the input source**:\n   - User provides URL: Use the `--file-url` parameter\n   - User provides local file path: Use the `--file-path` parameter\n\n2. **Execute OCR**:\n\n   ```bash\n   python scripts/ocr_caller.py --file-url \"URL provided by user\" --pretty\n   ```\n\n   Or for local files:\n\n   ```bash\n   python scripts/ocr_caller.py --file-path \"file path\" --pretty\n   ```\n\n   > **Performance note**: Parsing time scales with document complexity. Single-page images typically complete in 1-3 seconds; large PDFs (50+ pages) may take several minutes. Allow adequate time before assuming a timeout.\n\n   **Default behavior: save raw JSON to a temp file**:\n   - If `--output` is omitted, the script saves automatically under the system temp directory\n   - Default path pattern: `<system-temp>/paddleocr/text-recognition/results/result_<timestamp>_<id>.json`\n   - If `--output` is provided, it overrides the default temp-file destination\n   - If `--stdout` is provided, JSON is printed to stdout and no file is saved\n   - In save mode, the script prints the absolute saved path on stderr: `Result saved to: /absolute/path/...`\n   - In default/custom save mode, read and parse the saved JSON file before responding\n   - Use `--stdout` only when you explicitly want to skip file persistence\n\n3. **Parse JSON response**:\n   - In default/custom save mode, load JSON from the saved file path shown by the script\n   - Check the `ok` field: `true` means success, `false` means error\n   - Extract text: `text` field contains all recognized text\n   - If `--stdout` is used, parse the stdout JSON directly\n   - Handle errors: If `ok` is false, display `error.message`\n\n4. **Present results to user**:\n   - Display extracted text in a readable format\n   - If the text is empty, the image may contain no text\n   - In save mode, always tell the user the saved file path and that full raw JSON is available there\n\n### What to Do After Extraction\n\nCommon next steps once you have the recognized text:\n\n- **Save to file**: Write the `text` field to a `.txt` or `.md` file\n- **Search the content**: Search the saved output file for keywords\n- **Feed to another pipeline**: The `text` field is clean plain text, ready for downstream processing\n- **Poor results**: See \"Tips for Better Results\" below before retrying\n\n### Complete Output Display\n\nAlways display the COMPLETE recognized text to the user. The user typically needs the full content for downstream use — truncation silently loses data they may not notice is missing.\n\n- Display the entire `text` field, no matter how long\n- Do not use phrases like \"Here's a summary\" or \"The text begins with...\"\n- Do not truncate with \"...\" unless the text truly exceeds reasonable display limits (>10,000 chars)\n\n**Example - Correct**:\n\n```\nUser: \"Extract the text from this image\"\nAgent: I've extracted the text from the image. Here's the complete content:\n\n[Display the entire text here]\n```\n\n**Example - Incorrect**:\n\n```\nUser: \"Extract the text from this image\"\nAgent: I found some text in the image. Here's a preview:\n\"The quick brown fox...\" (truncated)\n```\n\n### Understanding the Output\n\nThe script returns a JSON envelope with `ok`, `text`, `result`, and `error` fields. Use `text` for the recognized content; `result` contains the raw API response for debugging.\n\nFor the full schema and field-level details, see `references/output_schema.md`.\n\n> Raw result location (default): the temp-file path printed by the script on stderr\n\n### Usage Examples\n\n**Example 1: URL OCR**\n\n```bash\npython scripts/ocr_caller.py --file-url \"https://example.com/invoice.jpg\" --pretty\n```\n\n**Example 2: Local File OCR**\n\n```bash\npython scripts/ocr_caller.py --file-path \"./document.pdf\" --pretty\n```\n\n**Example 3: OCR With Explicit File Type**\n\n```bash\npython scripts/ocr_caller.py --file-url \"https://example.com/input\" --file-type 1 --pretty\n```\n\n- `--file-type 0`: PDF\n- `--file-type 1`: image\n- If omitted, the type is auto-detected from the file extension. For local files, a recognized extension (`.pdf`, `.png`, `.jpg`, `.jpeg`, `.bmp`, `.tiff`, `.tif`, `.webp`) is required; otherwise pass `--file-type` explicitly. For URLs with unrecognized extensions, the service attempts inference.\n\n**Example 4: Print JSON Without Saving**\n\n```bash\npython scripts/ocr_caller.py --file-url \"https://example.com/input\" --stdout --pretty\n```\n\n### First-Time Configuration\n\n**When API is not configured**, the script outputs:\n\n```json\n{\n  \"ok\": false,\n  \"text\": \"\",\n  \"result\": null,\n  \"error\": {\n    \"code\": \"CONFIG_ERROR\",\n    \"message\": \"PADDLEOCR_OCR_API_URL not configured. Get your API at: https://paddleocr.com\"\n  }\n}\n```\n\n**Configuration workflow**:\n\n1. **Show the exact error message** to the user.\n\n2. **Guide the user to obtain credentials**: Visit the [PaddleOCR website](https://www.paddleocr.com), click **API**, select the `PP-OCRv5` model, select the language, then copy the `API_URL` and `Token`. They map to these environment variables:\n   - `PADDLEOCR_OCR_API_URL` — full endpoint URL ending with `/ocr`\n   - `PADDLEOCR_ACCESS_TOKEN` — 40-character alphanumeric string\n\n   Optionally configure `PADDLEOCR_OCR_TIMEOUT` for request timeout. Recommend using the host application's standard configuration method rather than pasting credentials in chat.\n\n3. **Apply credentials** — one of:\n   - **User configured via the host UI**: ask the user to confirm, then retry.\n   - **User pastes credentials in chat**: warn that they may be stored in conversation history, help the user persist them using the host's standard configuration method, then retry.\n\n### Error Handling\n\nAll errors return JSON with `ok: false`. Show the error message and stop — do not fall back to your own vision capabilities. Identify the issue from `error.code` and `error.message`:\n\n**Authentication failed (403)** — `error.message` contains \"Authentication failed\"\n\n- Token is invalid, reconfigure with correct credentials\n\n**Quota exceeded (429)** — `error.message` contains \"API rate limit exceeded\"\n\n- Daily API quota exhausted, inform user to wait or upgrade\n\n**Unsupported format** — `error.message` contains \"Unsupported file format\"\n\n- File format not supported, convert to PDF/PNG/JPG\n\n**No text detected**:\n\n- `text` field is empty\n- Image may be blank, corrupted, or contain no text\n\n### Tips for Better Results\n\nIf recognition quality is poor:\n\n- **Low resolution**: Provide a higher resolution image (≥300 DPI works well for most printed text)\n- **Noisy background**: A cleaner scan or screenshot typically yields better results than a phone photo\n- **Check confidence**: The raw JSON (`result.result.ocrResults[n].prunedResult.rec_scores`) shows per-line confidence scores — low values identify uncertain regions worth reviewing\n\n## Reference Documentation\n\n- `references/output_schema.md` — Full output schema, field descriptions, and command examples\n\n> **Note**: Model version, capabilities, and supported file formats are determined by your API endpoint (`PADDLEOCR_OCR_API_URL`) and its official API documentation.\n\n## Testing the Skill\n\nTo verify the skill is working properly:\n\n```bash\npython scripts/smoke_test.py\npython scripts/smoke_test.py --skip-api-test\npython scripts/smoke_test.py --test-url \"https://...\"\n```\n\nThe first form tests configuration and API connectivity. `--skip-api-test` checks configuration only. `--test-url` overrides the default sample image URL.\n\nFile v1.0.18:_meta.json\n\n{\n  \"ownerId\": \"kn77zppfj1a2fc620aygaf9z9980ewfa\",\n  \"slug\": \"paddleocr-text-recognition\",\n  \"version\": \"1.0.18\",\n  \"publishedAt\": 1774859810356\n}\n\nFile v1.0.18:references/output_schema.md\n\n# PaddleOCR Text Recognition Output Schema\n\nThis document defines the output envelope returned by `ocr_caller.py`.\n\nBy default, `ocr_caller.py` saves the JSON envelope to a unique file under the system temp directory and prints the absolute saved path to `stderr`. Use `--output` when you need a custom destination, or `--stdout` when you want to skip file saving and print JSON directly.\n\n## Output Envelope\n\n`ocr_caller.py` wraps provider response in a stable structure:\n\n```json\n{\n  \"ok\": true,\n  \"text\": \"Extracted text from all pages\",\n  \"result\": { ... },  // raw provider response\n  \"error\": null\n}\n```\n\nOn error:\n\n```json\n{\n  \"ok\": false,\n  \"text\": \"\",\n  \"result\": null,\n  \"error\": {\n    \"code\": \"ERROR_CODE\",\n    \"message\": \"Human-readable message\"\n  }\n}\n```\n\n## Error Codes\n\n| Code           | Description                                                                 |\n| -------------- | --------------------------------------------------------------------------- |\n| `INPUT_ERROR`  | Invalid or unusable input (arguments, file source, format, types). |\n| `CONFIG_ERROR` | Missing or invalid API / client configuration.       |\n| `API_ERROR`    | Request or response handling failed (network, HTTP, body parsing, schema). |\n\n## Raw Result Notes\n\nThe `result` field contains raw provider output.  \nRaw fields may vary by model version and endpoint.\n\n## Raw Result Example\n\n```json\n{\n  \"logId\": \"request-uuid\",\n  \"errorCode\": 0,\n  \"errorMsg\": \"Success\",\n  \"result\": {\n    \"ocrResults\": [\n      {\n        \"prunedResult\": {\n          \"rec_texts\": [\"First line\", \"Second line\"],\n          \"rec_scores\": [0.98, 0.95],\n          \"...\": \"other OCR fields\"\n        },\n        \"ocrImage\": \"https://...\",\n        \"inputImage\": \"https://...\",\n        \"...\": \"other model-specific fields\"\n      }\n    ],\n    \"dataInfo\": {\n      \"numPages\": 1,\n      \"type\": \"pdf\",\n      \"...\": \"other metadata\"\n    },\n    \"...\": \"other top-level fields\"\n  }\n}\n```\n\n## Stable Fields for Downstream Use\n\nPaths are relative to the output envelope root.\n\n- `result.result.ocrResults[n].prunedResult`  \n  Structured OCR data for page `n`.\n\n- `result.result.ocrResults[n].prunedResult.rec_texts`  \n  Recognized text lines for page `n`.\n\n- `result.result.ocrResults[n].prunedResult.rec_scores`  \n  Confidence scores for recognized text lines.\n\n## Text Extraction\n\n`ocr_caller.py` extracts top-level `text` from `result.result.ocrResults[n].prunedResult.rec_texts`, joins lines with `\\n`, and joins pages with `\\n\\n`.\n\n## Command Examples\n\n```bash\n# OCR from URL (result auto-saves to the system temp directory)\npython scripts/ocr_caller.py --file-url \"URL\" --pretty\n\n# OCR local file (result auto-saves to the system temp directory)\npython scripts/ocr_caller.py --file-path \"doc.pdf\" --pretty\n\n# OCR with explicit file type\npython scripts/ocr_caller.py --file-url \"URL\" --file-type 1 --pretty\n\n# Save result to a custom file path\npython scripts/ocr_caller.py --file-url \"URL\" --output \"./result.json\" --pretty\n\n# Print JSON to stdout without saving a file\npython scripts/ocr_caller.py --file-url \"URL\" --stdout --pretty\n```\n\nFile v1.0.18:requirements.txt\n\n# PaddleOCR Text Recognition Dependencies\n\nhttpx>=0.24.0\n\nArchive v1.0.17: 7 files, 13173 bytes\n\nFiles: references/output_schema.md (3100b), requirements.txt (57b), scripts/lib.py (12257b), scripts/ocr_caller.py (4206b), scripts/smoke_test.py (4100b), SKILL.md (9684b), _meta.json (146b)\n\nFile v1.0.17:SKILL.md\n\n---\nname: paddleocr-text-recognition\ndescription: >-\n  Use this skill whenever the user wants text extracted from images, photos, scans, screenshots,\n  or scanned PDFs. Returns exact machine-readable strings with line-level text and optional bbox\n  coordinates. Strong accuracy for CJK, small print, and handwritten text.\n  Trigger terms: OCR, 文字识别, 图片转文字, 截图识字, 提取图中文字, 扫描识字, 识字, 纯文字,\n  plain text extraction, 坐标, 检测框, bbox, bounding box, image to text, screenshot, photo scan,\n  recognize text.\nmetadata:\n  openclaw:\n    requires:\n      env:\n        - PADDLEOCR_OCR_API_URL\n        - PADDLEOCR_ACCESS_TOKEN\n      bins:\n        - python\n    primaryEnv: PADDLEOCR_ACCESS_TOKEN\n    emoji: \"🔤\"\n    homepage: https://github.com/PaddlePaddle/PaddleOCR/tree/main/skills/paddleocr-text-recognition\n---\n\n# PaddleOCR Text Recognition Skill\n\n## When to Use This Skill\n\n**Trigger keywords (routing)**: Bilingual trigger terms (Chinese and English) are listed in the YAML `description` above—use that field for discovery and routing.\n\n**Use this skill for**:\n\n- Extract text from images (screenshots, photos, scans)\n- Extract text from PDFs or document images when the goal is **line/box-level text**, not recovering table grids, formulas, or full reading-order layout\n- Extract text from URLs or local files that point to images/PDFs\n\n**Do not use for**:\n\n- Plain text files, code files, or markdown documents that can be read directly as text\n- Documents with tables, formulas, charts, or complex layouts — use Document Parsing instead\n- Tasks that do not involve image-to-text conversion\n\n## Installation\n\nInstall Python dependencies before using this skill. From the skill directory (`skills/paddleocr-text-recognition`):\n\n```bash\npip install -r requirements.txt\n```\n\n## How to Use This Skill\n\n> **Working directory**: All `python scripts/...` commands below should be run from this skill's root directory (the directory containing this SKILL.md file).\n\n### Basic Workflow\n\n1. **Identify the input source**:\n   - User provides URL: Use the `--file-url` parameter\n   - User provides local file path: Use the `--file-path` parameter\n   - User uploads image: Save it first, then use `--file-path`\n\n   **Input type note**:\n   - Supported file types depend on the model and endpoint configuration.\n   - Follow the official endpoint/API documentation for the exact supported formats.\n\n2. **Execute OCR**:\n\n   ```bash\n   python scripts/ocr_caller.py --file-url \"URL provided by user\" --pretty\n   ```\n\n   Or for local files:\n\n   ```bash\n   python scripts/ocr_caller.py --file-path \"file path\" --pretty\n   ```\n\n   > **Performance note**: Parsing time scales with document complexity. Single-page images typically complete in 1-3 seconds; large PDFs (50+ pages) may take several minutes. Allow adequate time before assuming a timeout.\n\n   **Default behavior: save raw JSON to a temp file**:\n   - If `--output` is omitted, the script saves automatically under the system temp directory\n   - Default path pattern: `<system-temp>/paddleocr/text-recognition/results/result_<timestamp>_<id>.json`\n   - If `--output` is provided, it overrides the default temp-file destination\n   - If `--stdout` is provided, JSON is printed to stdout and no file is saved\n   - In save mode, the script prints the absolute saved path on stderr: `Result saved to: /absolute/path/...`\n   - In default/custom save mode, read and parse the saved JSON file before responding\n   - Use `--stdout` only when you explicitly want to skip file persistence\n\n3. **Parse JSON response**:\n   - In default/custom save mode, load JSON from the saved file path shown by the script\n   - Check the `ok` field: `true` means success, `false` means error\n   - Extract text: `text` field contains all recognized text\n   - If `--stdout` is used, parse the stdout JSON directly\n   - Handle errors: If `ok` is false, display `error.message`\n\n4. **Present results to user**:\n   - Display extracted text in a readable format\n   - If the text is empty, the image may contain no text\n   - In save mode, always tell the user the saved file path and that full raw JSON is available there\n\n### What to Do After Extraction\n\nCommon next steps once you have the recognized text:\n\n- **Save to file**: Write the `text` field to a `.txt` or `.md` file\n- **Search the content**: Search the saved output file for keywords\n- **Feed to another pipeline**: The `text` field is clean plain text, ready for downstream processing\n- **Poor results**: See \"Tips for Better Results\" below before retrying\n\n### Complete Output Display\n\nAlways display the COMPLETE recognized text to the user. The user typically needs the full content for downstream use — truncation silently loses data they may not notice is missing.\n\n- Display the entire `text` field, no matter how long\n- Do not use phrases like \"Here's a summary\" or \"The text begins with...\"\n- Do not truncate with \"...\" unless the text truly exceeds reasonable display limits (>10,000 chars)\n\n**Example - Correct**:\n\n```\nUser: \"Extract the text from this image\"\nAgent: I've extracted the text from the image. Here's the complete content:\n\n[Display the entire text here]\n```\n\n**Example - Incorrect**:\n\n```\nUser: \"Extract the text from this image\"\nAgent: I found some text in the image. Here's a preview:\n\"The quick brown fox...\" (truncated)\n```\n\n### Understanding the Output\n\nThe script returns a JSON envelope with `ok`, `text`, `result`, and `error` fields. Use `text` for the recognized content; `result` contains the raw API response for debugging.\n\nFor the full schema and field-level details, see `references/output_schema.md`.\n\n> Raw result location (default): the temp-file path printed by the script on stderr\n\n### Usage Examples\n\n**Example 1: URL OCR**\n\n```bash\npython scripts/ocr_caller.py --file-url \"https://example.com/invoice.jpg\" --pretty\n```\n\n**Example 2: Local File OCR**\n\n```bash\npython scripts/ocr_caller.py --file-path \"./document.pdf\" --pretty\n```\n\n**Example 3: OCR With Explicit File Type**\n\n```bash\npython scripts/ocr_caller.py --file-url \"https://example.com/input\" --file-type 1 --pretty\n```\n\n- `--file-type 0`: PDF\n- `--file-type 1`: image\n- If omitted, the type is auto-detected from the file extension. For local files, a recognized extension (`.pdf`, `.png`, `.jpg`, `.jpeg`, `.bmp`, `.tiff`, `.tif`, `.webp`) is required; otherwise pass `--file-type` explicitly. For URLs with unrecognized extensions, the service attempts inference.\n\n**Example 4: Print JSON Without Saving**\n\n```bash\npython scripts/ocr_caller.py --file-url \"https://example.com/input\" --stdout --pretty\n```\n\n### First-Time Configuration\n\n**When API is not configured**, the script outputs:\n\n```json\n{\n  \"ok\": false,\n  \"text\": \"\",\n  \"result\": null,\n  \"error\": {\n    \"code\": \"CONFIG_ERROR\",\n    \"message\": \"PADDLEOCR_OCR_API_URL not configured. Get your API at: https://paddleocr.com\"\n  }\n}\n```\n\n**Configuration workflow**:\n\n1. **Show the exact error message** to the user.\n\n2. **Guide the user to obtain credentials**: Visit the [PaddleOCR website](https://www.paddleocr.com), click **API**, select the `PP-OCRv5` model, then copy the `API_URL` and `Token`. They map to these environment variables:\n   - `PADDLEOCR_OCR_API_URL` — full endpoint URL ending with `/ocr`\n   - `PADDLEOCR_ACCESS_TOKEN` — 40-character alphanumeric string\n\n   Optionally configure `PADDLEOCR_OCR_TIMEOUT` for request timeout. Recommend using the host application's standard configuration method rather than pasting credentials in chat.\n\n3. **Apply credentials** — one of:\n   - **User configured via the host UI**: ask the user to confirm, then retry.\n   - **User pastes credentials in chat**: warn that they may be stored in conversation history, help the user persist them using the host's standard configuration method, then retry.\n\n### Error Handling\n\nAll errors return JSON with `ok: false`. Show the error message and stop — do not fall back to your own vision capabilities. Identify the issue from `error.code` and `error.message`:\n\n**Authentication failed (403)** — `error.message` contains \"Authentication failed\"\n\n- Token is invalid, reconfigure with correct credentials\n\n**Quota exceeded (429)** — `error.message` contains \"API rate limit exceeded\"\n\n- Daily API quota exhausted, inform user to wait or upgrade\n\n**Unsupported format** — `error.message` contains \"Unsupported file format\"\n\n- File format not supported, convert to PDF/PNG/JPG\n\n**No text detected**:\n\n- `text` field is empty\n- Image may be blank, corrupted, or contain no text\n\n### Tips for Better Results\n\nIf recognition quality is poor:\n\n- **Low resolution**: Provide a higher resolution image (≥300 DPI works well for most printed text)\n- **Noisy background**: A cleaner scan or screenshot typically yields better results than a phone photo\n- **Check confidence**: The raw JSON (`result.result.ocrResults[n].prunedResult.rec_scores`) shows per-line confidence scores — low values identify uncertain regions worth reviewing\n\n## Reference Documentation\n\n- `references/output_schema.md` — Full output schema, field descriptions, and command examples\n\n> **Note**: Model version, capabilities, and supported file formats are determined by your API endpoint (`PADDLEOCR_OCR_API_URL`) and its official API documentation.\n\n## Testing the Skill\n\nTo verify the skill is working properly:\n\n```bash\npython scripts/smoke_test.py\npython scripts/smoke_test.py --skip-api-test\npython scripts/smoke_test.py --test-url \"https://...\"\n```\n\nThe first form tests configuration and API connectivity. `--skip-api-test` checks configuration only. `--test-url` overrides the default sample image URL.\n\nFile v1.0.17:_meta.json\n\n{\n  \"ownerId\": \"kn77zppfj1a2fc620aygaf9z9980ewfa\",\n  \"slug\": \"paddleocr-text-recognition\",\n  \"version\": \"1.0.17\",\n  \"publishedAt\": 1774669295485\n}\n\nFile v1.0.17:references/output_schema.md\n\n# PaddleOCR Text Recognition Output Schema\n\nThis document defines the output envelope returned by `ocr_caller.py`.\n\nBy default, `ocr_caller.py` saves the JSON envelope to a unique file under the system temp directory and prints the absolute saved path to `stderr`. Use `--output` when you need a custom destination, or `--stdout` when you want to skip file saving and print JSON directly.\n\n## Output Envelope\n\n`ocr_caller.py` wraps provider response in a stable structure:\n\n```json\n{\n  \"ok\": true,\n  \"text\": \"Extracted text from all pages\",\n  \"result\": { ... },  // raw provider response\n  \"error\": null\n}\n```\n\nOn error:\n\n```json\n{\n  \"ok\": false,\n  \"text\": \"\",\n  \"result\": null,\n  \"error\": {\n    \"code\": \"ERROR_CODE\",\n    \"message\": \"Human-readable message\"\n  }\n}\n```\n\n## Error Codes\n\n| Code           | Description                                                                 |\n| -------------- | --------------------------------------------------------------------------- |\n| `INPUT_ERROR`  | Invalid or unusable input (arguments, file source, format, types). |\n| `CONFIG_ERROR` | Missing or invalid API / client configuration.       |\n| `API_ERROR`    | Request or response handling failed (network, HTTP, body parsing, schema). |\n\n## Raw Result Notes\n\nThe `result` field contains raw provider output.  \nRaw fields may vary by model version and endpoint.\n\n## Raw Result Example\n\n```json\n{\n  \"logId\": \"request-uuid\",\n  \"errorCode\": 0,\n  \"errorMsg\": \"Success\",\n  \"result\": {\n    \"ocrResults\": [\n      {\n        \"prunedResult\": {\n          \"rec_texts\": [\"First line\", \"Second line\"],\n          \"rec_scores\": [0.98, 0.95],\n          \"...\": \"other OCR fields\"\n        },\n        \"ocrImage\": \"https://...\",\n        \"inputImage\": \"https://...\",\n        \"...\": \"other model-specific fields\"\n      }\n    ],\n    \"dataInfo\": {\n      \"numPages\": 1,\n      \"type\": \"pdf\",\n      \"...\": \"other metadata\"\n    },\n    \"...\": \"other top-level fields\"\n  }\n}\n```\n\n## Stable Fields for Downstream Use\n\nPaths are relative to the output envelope root.\n\n- `result.result.ocrResults[n].prunedResult`  \n  Structured OCR data for page `n`.\n\n- `result.result.ocrResults[n].prunedResult.rec_texts`  \n  Recognized text lines for page `n`.\n\n- `result.result.ocrResults[n].prunedResult.rec_scores`  \n  Confidence scores for recognized text lines.\n\n## Text Extraction\n\n`ocr_caller.py` extracts top-level `text` from `result.result.ocrResults[n].prunedResult.rec_texts`, joins lines with `\\n`, and joins pages with `\\n\\n`.\n\n## Command Examples\n\n```bash\n# OCR from URL (result auto-saves to the system temp directory)\npython scripts/ocr_caller.py --file-url \"URL\" --pretty\n\n# OCR local file (result auto-saves to the system temp directory)\npython scripts/ocr_caller.py --file-path \"doc.pdf\" --pretty\n\n# OCR with explicit file type\npython scripts/ocr_caller.py --file-url \"URL\" --file-type 1 --pretty\n\n# Save result to a custom file path\npython scripts/ocr_caller.py --file-url \"URL\" --output \"./result.json\" --pretty\n\n# Print JSON to stdout without saving a file\npython scripts/ocr_caller.py --file-url \"URL\" --stdout --pretty\n```\n\nFile v1.0.17:requirements.txt\n\n# PaddleOCR Text Recognition Dependencies\n\nhttpx>=0.24.0\n\nArchive v1.0.16: 7 files, 13113 bytes\n\nFiles: references/output_schema.md (3100b), requirements.txt (57b), scripts/lib.py (12257b), scripts/ocr_caller.py (4206b), scripts/smoke_test.py (3938b), SKILL.md (9684b), _meta.json (146b)\n\nFile v1.0.16:SKILL.md\n\n---\nname: paddleocr-text-recognition\ndescription: >-\n  Use this skill whenever the user wants text extracted from images, photos, scans, screenshots,\n  or scanned PDFs. Returns exact machine-readable strings with line-level text and optional bbox\n  coordinates. Strong accuracy for CJK, small print, and handwritten text.\n  Trigger terms: OCR, 文字识别, 图片转文字, 截图识字, 提取图中文字, 扫描识字, 识字, 纯文字,\n  plain text extraction, 坐标, 检测框, bbox, bounding box, image to text, screenshot, photo scan,\n  recognize text.\nmetadata:\n  openclaw:\n    requires:\n      env:\n        - PADDLEOCR_OCR_API_URL\n        - PADDLEOCR_ACCESS_TOKEN\n      bins:\n        - python\n    primaryEnv: PADDLEOCR_ACCESS_TOKEN\n    emoji: \"🔤\"\n    homepage: https://github.com/PaddlePaddle/PaddleOCR/tree/main/skills/paddleocr-text-recognition\n---\n\n# PaddleOCR Text Recognition Skill\n\n## When to Use This Skill\n\n**Trigger keywords (routing)**: Bilingual trigger terms (Chinese and English) are listed in the YAML `description` above—use that field for discovery and routing.\n\n**Use this skill for**:\n\n- Extract text from images (screenshots, photos, scans)\n- Extract text from PDFs or document images when the goal is **line/box-level text**, not recovering table grids, formulas, or full reading-order layout\n- Extract text from URLs or local files that point to images/PDFs\n\n**Do not use for**:\n\n- Plain text files, code files, or markdown documents that can be read directly as text\n- Documents with tables, formulas, charts, or complex layouts — use Document Parsing instead\n- Tasks that do not involve image-to-text conversion\n\n## Installation\n\nInstall Python dependencies before using this skill. From the skill directory (`skills/paddleocr-text-recognition`):\n\n```bash\npip install -r requirements.txt\n```\n\n## How to Use This Skill\n\n> **Working directory**: All `python scripts/...` commands below should be run from this skill's root directory (the directory containing this SKILL.md file).\n\n### Basic Workflow\n\n1. **Identify the input source**:\n   - User provides URL: Use the `--file-url` parameter\n   - User provides local file path: Use the `--file-path` parameter\n   - User uploads image: Save it first, then use `--file-path`\n\n   **Input type note**:\n   - Supported file types depend on the model and endpoint configuration.\n   - Follow the official endpoint/API documentation for the exact supported formats.\n\n2. **Execute OCR**:\n\n   ```bash\n   python scripts/ocr_caller.py --file-url \"URL provided by user\" --pretty\n   ```\n\n   Or for local files:\n\n   ```bash\n   python scripts/ocr_caller.py --file-path \"file path\" --pretty\n   ```\n\n   > **Performance note**: Parsing time scales with document complexity. Single-page images typically complete in 1-3 seconds; large PDFs (50+ pages) may take several minutes. Allow adequate time before assuming a timeout.\n\n   **Default behavior: save raw JSON to a temp file**:\n   - If `--output` is omitted, the script saves automatically under the system temp directory\n   - Default path pattern: `<system-temp>/paddleocr/text-recognition/results/result_<timestamp>_<id>.json`\n   - If `--output` is provided, it overrides the default temp-file destination\n   - If `--stdout` is provided, JSON is printed to stdout and no file is saved\n   - In save mode, the script prints the absolute saved path on stderr: `Result saved to: /absolute/path/...`\n   - In default/custom save mode, read and parse the saved JSON file before responding\n   - Use `--stdout` only when you explicitly want to skip file persistence\n\n3. **Parse JSON response**:\n   - In default/custom save mode, load JSON from the saved file path shown by the script\n   - Check the `ok` field: `true` means success, `false` means error\n   - Extract text: `text` field contains all recognized text\n   - If `--stdout` is used, parse the stdout JSON directly\n   - Handle errors: If `ok` is false, display `error.message`\n\n4. **Present results to user**:\n   - Display extracted text in a readable format\n   - If the text is empty, the image may contain no text\n   - In save mode, always tell the user the saved file path and that full raw JSON is available there\n\n### What to Do After Extraction\n\nCommon next steps once you have the recognized text:\n\n- **Save to file**: Write the `text` field to a `.txt` or `.md` file\n- **Search the content**: Search the saved output file for keywords\n- **Feed to another pipeline**: The `text` field is clean plain text, ready for downstream processing\n- **Poor results**: See \"Tips for Better Results\" below before retrying\n\n### Complete Output Display\n\nAlways display the COMPLETE recognized text to the user. The user typically needs the full content for downstream use — truncation silently loses data they may not notice is missing.\n\n- Display the entire `text` field, no matter how long\n- Do not use phrases like \"Here's a summary\" or \"The text begins with...\"\n- Do not truncate with \"...\" unless the text truly exceeds reasonable display limits (>10,000 chars)\n\n**Example - Correct**:\n\n```\nUser: \"Extract the text from this image\"\nAgent: I've extracted the text from the image. Here's the complete content:\n\n[Display the entire text here]\n```\n\n**Example - Incorrect**:\n\n```\nUser: \"Extract the text from this image\"\nAgent: I found some text in the image. Here's a preview:\n\"The quick brown fox...\" (truncated)\n```\n\n### Understanding the Output\n\nThe script returns a JSON envelope with `ok`, `text`, `result`, and `error` fields. Use `text` for the recognized content; `result` contains the raw API response for debugging.\n\nFor the full schema and field-level details, see `references/output_schema.md`.\n\n> Raw result location (default): the temp-file path printed by the script on stderr\n\n### Usage Examples\n\n**Example 1: URL OCR**\n\n```bash\npython scripts/ocr_caller.py --file-url \"https://example.com/invoice.jpg\" --pretty\n```\n\n**Example 2: Local File OCR**\n\n```bash\npython scripts/ocr_caller.py --file-path \"./document.pdf\" --pretty\n```\n\n**Example 3: OCR With Explicit File Type**\n\n```bash\npython scripts/ocr_caller.py --file-url \"https://example.com/input\" --file-type 1 --pretty\n```\n\n- `--file-type 0`: PDF\n- `--file-type 1`: image\n- If omitted, the type is auto-detected from the file extension. For local files, a recognized extension (`.pdf`, `.png`, `.jpg`, `.jpeg`, `.bmp`, `.tiff`, `.tif`, `.webp`) is required; otherwise pass `--file-type` explicitly. For URLs with unrecognized extensions, the service attempts inference.\n\n**Example 4: Print JSON Without Saving**\n\n```bash\npython scripts/ocr_caller.py --file-url \"https://example.com/input\" --stdout --pretty\n```\n\n### First-Time Configuration\n\n**When API is not configured**, the script outputs:\n\n```json\n{\n  \"ok\": false,\n  \"text\": \"\",\n  \"result\": null,\n  \"error\": {\n    \"code\": \"CONFIG_ERROR\",\n    \"message\": \"PADDLEOCR_OCR_API_URL not configured. Get your API at: https://paddleocr.com\"\n  }\n}\n```\n\n**Configuration workflow**:\n\n1. **Show the exact error message** to the user.\n\n2. **Guide the user to obtain credentials**: Visit the [PaddleOCR website](https://www.paddleocr.com), click **API**, select the `PP-OCRv5` model, then copy the `API_URL` and `Token`. They map to these environment variables:\n   - `PADDLEOCR_OCR_API_URL` — full endpoint URL ending with `/ocr`\n   - `PADDLEOCR_ACCESS_TOKEN` — 40-character alphanumeric string\n\n   Optionally configure `PADDLEOCR_OCR_TIMEOUT` for request timeout. Recommend using the host application's standard configuration method rather than pasting credentials in chat.\n\n3. **Apply credentials** — one of:\n   - **User configured via the host UI**: ask the user to confirm, then retry.\n   - **User pastes credentials in chat**: warn that they may be stored in conversation history, help the user persist them using the host's standard configuration method, then retry.\n\n### Error Handling\n\nAll errors return JSON with `ok: false`. Show the error message and stop — do not fall back to your own vision capabilities. Identify the issue from `error.code` and `error.message`:\n\n**Authentication failed (403)** — `error.message` contains \"Authentication failed\"\n\n- Token is invalid, reconfigure with correct credentials\n\n**Quota exceeded (429)** — `error.message` contains \"API rate limit exceeded\"\n\n- Daily API quota exhausted, inform user to wait or upgrade\n\n**Unsupported format** — `error.message` contains \"Unsupported file format\"\n\n- File format not supported, convert to PDF/PNG/JPG\n\n**No text detected**:\n\n- `text` field is empty\n- Image may be blank, corrupted, or contain no text\n\n### Tips for Better Results\n\nIf recognition quality is poor:\n\n- **Low resolution**: Provide a higher resolution image (≥300 DPI works well for most printed text)\n- **Noisy background**: A cleaner scan or screenshot typically yields better results than a phone photo\n- **Check confidence**: The raw JSON (`result.result.ocrResults[n].prunedResult.rec_scores`) shows per-line confidence scores — low values identify uncertain regions worth reviewing\n\n## Reference Documentation\n\n- `references/output_schema.md` — Full output schema, field descriptions, and command examples\n\n> **Note**: Model version, capabilities, and supported file formats are determined by your API endpoint (`PADDLEOCR_OCR_API_URL`) and its official API documentation.\n\n## Testing the Skill\n\nTo verify the skill is working properly:\n\n```bash\npython scripts/smoke_test.py\npython scripts/smoke_test.py --skip-api-test\npython scripts/smoke_test.py --test-url \"https://...\"\n```\n\nThe first form tests configuration and API connectivity. `--skip-api-test` checks configuration only. `--test-url` overrides the default sample image URL.\n\nFile v1.0.16:_meta.json\n\n{\n  \"ownerId\": \"kn77zppfj1a2fc620aygaf9z9980ewfa\",\n  \"slug\": \"paddleocr-text-recognition\",\n  \"version\": \"1.0.16\",\n  \"publishedAt\": 1774614911807\n}\n\nFile v1.0.16:references/output_schema.md\n\n# PaddleOCR Text Recognition Output Schema\n\nThis document defines the output envelope returned by `ocr_caller.py`.\n\nBy default, `ocr_caller.py` saves the JSON envelope to a unique file under the system temp directory and prints the absolute saved path to `stderr`. Use `--output` when you need a custom destination, or `--stdout` when you want to skip file saving and print JSON directly.\n\n## Output Envelope\n\n`ocr_caller.py` wraps provider response in a stable structure:\n\n```json\n{\n  \"ok\": true,\n  \"text\": \"Extracted text from all pages\",\n  \"result\": { ... },  // raw provider response\n  \"error\": null\n}\n```\n\nOn error:\n\n```json\n{\n  \"ok\": false,\n  \"text\": \"\",\n  \"result\": null,\n  \"error\": {\n    \"code\": \"ERROR_CODE\",\n    \"message\": \"Human-readable message\"\n  }\n}\n```\n\n## Error Codes\n\n| Code           | Description                                                                 |\n| -------------- | --------------------------------------------------------------------------- |\n| `INPUT_ERROR`  | Invalid or unusable input (arguments, file source, format, types). |\n| `CONFIG_ERROR` | Missing or invalid API / client configuration.       |\n| `API_ERROR`    | Request or response handling failed (network, HTTP, body parsing, schema). |\n\n## Raw Result Notes\n\nThe `result` field contains raw provider output.  \nRaw fields may vary by model version and endpoint.\n\n## Raw Result Example\n\n```json\n{\n  \"logId\": \"request-uuid\",\n  \"errorCode\": 0,\n  \"errorMsg\": \"Success\",\n  \"result\": {\n    \"ocrResults\": [\n      {\n        \"prunedResult\": {\n          \"rec_texts\": [\"First line\", \"Second line\"],\n          \"rec_scores\": [0.98, 0.95],\n          \"...\": \"other OCR fields\"\n        },\n        \"ocrImage\": \"https://...\",\n        \"inputImage\": \"https://...\",\n        \"...\": \"other model-specific fields\"\n      }\n    ],\n    \"dataInfo\": {\n      \"numPages\": 1,\n      \"type\": \"pdf\",\n      \"...\": \"other metadata\"\n    },\n    \"...\": \"other top-level fields\"\n  }\n}\n```\n\n## Stable Fields for Downstream Use\n\nPaths are relative to the output envelope root.\n\n- `result.result.ocrResults[n].prunedResult`  \n  Structured OCR data for page `n`.\n\n- `result.result.ocrResults[n].prunedResult.rec_texts`  \n  Recognized text lines for page `n`.\n\n- `result.result.ocrResults[n].prunedResult.rec_scores`  \n  Confidence scores for recognized text lines.\n\n## Text Extraction\n\n`ocr_caller.py` extracts top-level `text` from `result.result.ocrResults[n].prunedResult.rec_texts`, joins lines with `\\n`, and joins pages with `\\n\\n`.\n\n## Command Examples\n\n```bash\n# OCR from URL (result auto-saves to the system temp directory)\npython scripts/ocr_caller.py --file-url \"URL\" --pretty\n\n# OCR local file (result auto-saves to the system temp directory)\npython scripts/ocr_caller.py --file-path \"doc.pdf\" --pretty\n\n# OCR with explicit file type\npython scripts/ocr_caller.py --file-url \"URL\" --file-type 1 --pretty\n\n# Save result to a custom file path\npython scripts/ocr_caller.py --file-url \"URL\" --output \"./result.json\" --pretty\n\n# Print JSON to stdout without saving a file\npython scripts/ocr_caller.py --file-url \"URL\" --stdout --pretty\n```\n\nFile v1.0.16:requirements.txt\n\n# PaddleOCR Text Recognition Dependencies\n\nhttpx>=0.24.0\n\nArchive v1.0.15: 7 files, 12367 bytes\n\nFiles: references/output_schema.md (3128b), requirements.txt (57b), scripts/lib.py (9552b), scripts/ocr_caller.py (4202b), scripts/smoke_test.py (3934b), SKILL.md (9477b), _meta.json (146b)\n\nFile v1.0.15:SKILL.md\n\n---\nname: paddleocr-text-recognition\ndescription: >-\n  Use this skill whenever the user wants text extracted from images, photos, scans, screenshots,\n  or scanned PDFs. Returns exact machine-readable strings with line-level text and optional bbox\n  coordinates. Strong accuracy for CJK, small print, and handwritten text.\n  Trigger terms: OCR, 文字识别, 图片转文字, 截图识字, 提取图中文字, 扫描识字, 识字, 纯文字,\n  plain text extraction, 坐标, 检测框, bbox, bounding box, image to text, screenshot, photo scan,\n  recognize text.\nmetadata:\n  openclaw:\n    requires:\n      env:\n        - PADDLEOCR_OCR_API_URL\n        - PADDLEOCR_ACCESS_TOKEN\n      bins:\n        - python\n    primaryEnv: PADDLEOCR_ACCESS_TOKEN\n    emoji: \"🔤\"\n    homepage: https://github.com/PaddlePaddle/PaddleOCR/tree/main/skills/paddleocr-text-recognition\n---\n\n# PaddleOCR Text Recognition Skill\n\n## When to Use This Skill\n\n**Trigger keywords (routing)**: Bilingual trigger terms (Chinese and English) are listed in the YAML `description` above—use that field for discovery and routing.\n\n**Use this skill for**:\n\n- Extract text from images (screenshots, photos, scans)\n- Extract text from PDFs or document images when the goal is **line/box-level text**, not recovering table grids, formulas, or full reading-order layout\n- Extract text from URLs or local files that point to images/PDFs\n\n**Do not use for**:\n\n- Plain text files, code files, or markdown documents that can be read directly as text\n- Documents with tables, formulas, charts, or complex layouts — use Document Parsing instead\n- Tasks that do not involve image-to-text conversion\n\n## Installation\n\nInstall Python dependencies before using this skill. From the skill directory (`skills/paddleocr-text-recognition`):\n\n```bash\npip install -r requirements.txt\n```\n\n## How to Use This Skill\n\n> **Working directory**: All `python scripts/...` commands below should be run from this skill's root directory (the directory containing this SKILL.md file).\n\n### Basic Workflow\n\n1. **Identify the input source**:\n   - User provides URL: Use the `--file-url` parameter\n   - User provides local file path: Use the `--file-path` parameter\n   - User uploads image: Save it first, then use `--file-path`\n\n   **Input type note**:\n   - Supported file types depend on the model and endpoint configuration.\n   - Follow the official endpoint/API documentation for the exact supported formats.\n\n2. **Execute OCR**:\n\n   ```bash\n   python scripts/ocr_caller.py --file-url \"URL provided by user\" --pretty\n   ```\n\n   Or for local files:\n\n   ```bash\n   python scripts/ocr_caller.py --file-path \"file path\" --pretty\n   ```\n\n   > **Performance note**: Parsing time scales with document complexity. Single-page images typically complete in 1-3 seconds; large PDFs (50+ pages) may take several minutes. Allow adequate time before assuming a timeout.\n\n   **Default behavior: save raw JSON to a temp file**:\n   - If `--output` is omitted, the script saves automatically under the system temp directory\n   - Default path pattern: `<system-temp>/paddleocr/text-recognition/results/result_<timestamp>_<id>.json`\n   - If `--output` is provided, it overrides the default temp-file destination\n   - If `--stdout` is provided, JSON is printed to stdout and no file is saved\n   - In save mode, the script prints the absolute saved path on stderr: `Result saved to: /absolute/path/...`\n   - In default/custom save mode, read and parse the saved JSON file before responding\n   - Use `--stdout` only when you explicitly want to skip file persistence\n\n3. **Parse JSON response**:\n   - In default/custom save mode, load JSON from the saved file path shown by the script\n   - Check the `ok` field: `true` means success, `false` means error\n   - Extract text: `text` field contains all recognized text\n   - If `--stdout` is used, parse the stdout JSON directly\n   - Handle errors: If `ok` is false, display `error.message`\n\n4. **Present results to user**:\n   - Display extracted text in a readable format\n   - If the text is empty, the image may contain no text\n   - In save mode, always tell the user the saved file path and that full raw JSON is available there\n\n### What to Do After Extraction\n\nCommon next steps once you have the recognized text:\n\n- **Save to file**: Write the `text` field to a `.txt` or `.md` file\n- **Search the content**: Search the saved output file for keywords\n- **Feed to another pipeline**: The `text` field is clean plain text, ready for downstream processing\n- **Poor results**: See \"Tips for Better Results\" below before retrying\n\n### Complete Output Display\n\nAlways display the COMPLETE recognized text to the user. The user typically needs the full content for downstream use — truncation silently loses data they may not notice is missing.\n\n- Display the entire `text` field, no matter how long\n- Do not use phrases like \"Here's a summary\" or \"The text begins with...\"\n- Do not truncate with \"...\" unless the text truly exceeds reasonable display limits (>10,000 chars)\n\n**Example - Correct**:\n\n```\nUser: \"Extract the text from this image\"\nAgent: I've extracted the text from the image. Here's the complete content:\n\n[Display the entire text here]\n```\n\n**Example - Incorrect**:\n\n```\nUser: \"Extract the text from this image\"\nAgent: I found some text in the image. Here's a preview:\n\"The quick brown fox...\" (truncated)\n```\n\n### Understanding the Output\n\nThe script returns a JSON envelope with `ok`, `text`, `result`, and `error` fields. Use `text` for the recognized content; `result` contains the raw API response for debugging.\n\nFor the full schema and field-level details, see `references/output_schema.md`.\n\n> Raw result location (default): the temp-file path printed by the script on stderr\n\n### Usage Examples\n\n**Example 1: URL OCR**\n\n```bash\npython scripts/ocr_caller.py --file-url \"https://example.com/invoice.jpg\" --pretty\n```\n\n**Example 2: Local File OCR**\n\n```bash\npython scripts/ocr_caller.py --file-path \"./document.pdf\" --pretty\n```\n\n**Example 3: OCR With Explicit File Type**\n\n```bash\npython scripts/ocr_caller.py --file-url \"https://example.com/input\" --file-type 1 --pretty\n```\n\n- `--file-type 0`: PDF\n- `--file-type 1`: image\n- If omitted, the type is auto-detected from the file extension. For local files, a recognized extension (`.pdf`, `.png`, `.jpg`, `.jpeg`, `.bmp`, `.tiff`, `.tif`, `.webp`) is required; otherwise pass `--file-type` explicitly. For URLs with unrecognized extensions, the service attempts inference.\n\n**Example 4: Print JSON Without Saving**\n\n```bash\npython scripts/ocr_caller.py --file-url \"https://example.com/input\" --stdout --pretty\n```\n\n### First-Time Configuration\n\n**When API is not configured**, the script outputs:\n\n```json\n{\n  \"ok\": false,\n  \"text\": \"\",\n  \"result\": null,\n  \"error\": {\n    \"code\": \"CONFIG_ERROR\",\n    \"message\": \"PADDLEOCR_OCR_API_URL not configured. Get your API at: https://paddleocr.com\"\n  }\n}\n```\n\n**Configuration workflow**:\n\n1. **Show the exact error message** to the user.\n\n2. **Guide the user to obtain credentials**: Visit the [PaddleOCR website](https://www.paddleocr.com), click **API**, select the `PP-OCRv5` model, then copy the `API_URL` and `Token`. They map to these environment variables:\n   - `PADDLEOCR_OCR_API_URL` — full endpoint URL ending with `/ocr`\n   - `PADDLEOCR_ACCESS_TOKEN` — 40-character alphanumeric string\n\n   Optionally configure `PADDLEOCR_OCR_TIMEOUT` for request timeout. Recommend using the host application's standard configuration method rather than pasting credentials in chat.\n\n3. **Apply credentials** — one of:\n   - **User configured via the host UI**: ask the user to confirm, then retry.\n   - **User pastes credentials in chat**: warn that they may be stored in conversation history, help the user persist them using the host's standard configuration method, then retry.\n\n### Error Handling\n\nAll errors return JSON with `ok: false`. Show the error message and stop — do not fall back to your own vision capabilities. Identify the issue from `error.code` and `error.message`:\n\n**Authentication failed (403)** — `error.message` contains \"Authentication failed\"\n\n- Token is invalid, reconfigure with correct credentials\n\n**Quota exceeded (429)** — `error.message` contains \"API rate limit exceeded\"\n\n- Daily API quota exhausted, inform user to wait or upgrade\n\n**Unsupported format** — `error.message` contains \"Unsupported file format\"\n\n- File format not supported, convert to PDF/PNG/JPG\n\n**No text detected**:\n\n- `text` field is empty\n- Image may be blank, corrupted, or contain no text\n\n### Tips for Better Results\n\nIf recognition quality is poor:\n\n- **Low resolution**: Provide a higher resolution image (≥300 DPI works well for most printed text)\n- **Noisy background**: A cleaner scan or screenshot typically yields better results than a phone photo\n- **Check confidence**: The raw JSON (`result.result.ocrResults[n].prunedResult.rec_scores`) shows per-line confidence scores — low values identify uncertain regions worth reviewing\n\n## Reference Documentation\n\n- `references/output_schema.md` — Full output schema, field descriptions, and command examples\n\n> **Note**: Model version, capabilities, and supported file formats are determined by your API endpoint (`PADDLEOCR_OCR_API_URL`) and its official API documentation.\n\n## Testing the Skill\n\nTo verify the skill is working properly:\n\n```bash\npython scripts/smoke_test.py\n```\n\nThis tests configuration and API connectivity.\n\nFile v1.0.15:_meta.json\n\n{\n  \"ownerId\": \"kn77zppfj1a2fc620aygaf9z9980ewfa\",\n  \"slug\": \"paddleocr-text-recognition\",\n  \"version\": \"1.0.15\",\n  \"publishedAt\": 1774607765224\n}\n\nFile v1.0.15:references/output_schema.md\n\n# PaddleOCR Text Recognition Output Schema\n\nThis document defines the output envelope returned by `ocr_caller.py`.\n\nBy default, `ocr_caller.py` saves the JSON envelope to a unique file under the system temp directory and prints the absolute saved path to `stderr`. Use `--output` when you need a custom destination, or `--stdout` when you want to skip file saving and print JSON directly.\n\n## Output Envelope\n\n`ocr_caller.py` wraps provider response in a stable structure:\n\n```json\n{\n  \"ok\": true,\n  \"text\": \"Extracted text from all pages\",\n  \"result\": { ... },  // raw provider response\n  \"error\": null\n}\n```\n\nOn error:\n\n```json\n{\n  \"ok\": false,\n  \"text\": \"\",\n  \"result\": null,\n  \"error\": {\n    \"code\": \"ERROR_CODE\",\n    \"message\": \"Human-readable message\"\n  }\n}\n```\n\n## Error Codes\n\n| Code           | Description                                                                |\n| -------------- | -------------------------------------------------------------------------- |\n| `INPUT_ERROR`  | Invalid input (missing file, unsupported format, invalid file type)        |\n| `CONFIG_ERROR` | API not configured                                                         |\n| `API_ERROR`    | API call failed (auth, timeout, service error, or invalid response schema) |\n\n## Raw Result Notes\n\nThe `result` field contains raw provider output.  \nRaw fields may vary by model version and endpoint.\n\n## Raw Result Example\n\n```json\n{\n  \"logId\": \"request-uuid\",\n  \"errorCode\": 0,\n  \"errorMsg\": \"Success\",\n  \"result\": {\n    \"ocrResults\": [\n      {\n        \"prunedResult\": {\n          \"rec_texts\": [\"First line\", \"Second line\"],\n          \"rec_scores\": [0.98, 0.95],\n          \"...\": \"other OCR fields\"\n        },\n        \"ocrImage\": \"https://...\",\n        \"inputImage\": \"https://...\",\n        \"...\": \"other model-specific fields\"\n      }\n    ],\n    \"dataInfo\": {\n      \"numPages\": 1,\n      \"type\": \"pdf\",\n      \"...\": \"other metadata\"\n    },\n    \"...\": \"other top-level fields\"\n  }\n}\n```\n\n## Stable Fields for Downstream Use\n\nPaths are relative to the output envelope root.\n\n- `result.result.ocrResults[n].prunedResult`  \n  Structured OCR data for page `n`.\n\n- `result.result.ocrResults[n].prunedResult.rec_texts`  \n  Recognized text lines for page `n`.\n\n- `result.result.ocrResults[n].prunedResult.rec_scores`  \n  Confidence scores for recognized text lines.\n\n## Text Extraction\n\n`ocr_caller.py` extracts top-level `text` from `result.result.ocrResults[n].prunedResult.rec_texts`, joins lines with `\\n`, and joins pages with `\\n\\n`.\n\n## Command Examples\n\n```bash\n# OCR from URL (result auto-saves to the system temp directory)\npython scripts/ocr_caller.py --file-url \"URL\" --pretty\n\n# OCR local file (result auto-saves to the system temp directory)\npython scripts/ocr_caller.py --file-path \"doc.pdf\" --pretty\n\n# OCR with explicit file type\npython scripts/ocr_caller.py --file-url \"URL\" --file-type 1 --pretty\n\n# Save result to a custom file path\npython scripts/ocr_caller.py --file-url \"URL\" --output \"./result.json\" --pretty\n\n# Print JSON to stdout without saving a file\npython scripts/ocr_caller.py --file-url \"URL\" --stdout --pretty\n```\n\nFile v1.0.15:requirements.txt\n\n# PaddleOCR Text Recognition Dependencies\n\nhttpx>=0.24.0\n\nArchive v1.0.14: 7 files, 13251 bytes\n\nFiles: references/output_schema.md (3018b), requirements.txt (57b), scripts/lib.py (9493b), scripts/ocr_caller.py (4811b), scripts/smoke_test.py (4580b), SKILL.md (9235b), _meta.json (146b)\n\nFile v1.0.14:SKILL.md\n\n---\nname: paddleocr-text-recognition\ndescription: >-\n  Use this skill whenever the user wants text extracted from images, photos, scans, screenshots,\n  or scanned PDFs. Returns exact machine-readable strings with line-level text and optional bbox\n  coordinates. Strong accuracy for CJK, small print, and handwritten text. Supports batch/multi-image\n  runs.\n  Trigger terms: OCR, 文字识别, 图片转文字, 截图识字, 提取图中文字, 扫描识字, 识字, 纯文字,\n  plain text extraction, 坐标, 检测框, bbox, bounding box, image to text, screenshot, photo scan,\n  recognize text.\nmetadata:\n  openclaw:\n    requires:\n      env:\n        - PADDLEOCR_OCR_API_URL\n        - PADDLEOCR_ACCESS_TOKEN\n      bins:\n        - python\n    primaryEnv: PADDLEOCR_ACCESS_TOKEN\n    emoji: \"🔤\"\n    homepage: https://github.com/PaddlePaddle/PaddleOCR/tree/main/skills/paddleocr-text-recognition\n---\n\n# PaddleOCR Text Recognition Skill\n\n## When to Use This Skill\n\n**Trigger keywords (routing)**: Bilingual trigger terms (Chinese and English) are listed in the YAML `description` above—use that field for discovery and routing.\n\nInvoke this skill in the following situations:\n\n- Extract text from images (screenshots, photos, scans)\n- Extract text from PDFs or document images when the goal is **line/box-level text**, not recovering table grids, formulas, or full reading-order layout\n- Extract text from URLs or local files that point to images/PDFs\n\nDo not use this skill in the following situations:\n\n- Plain text files that can be read directly with the Read tool\n- Code files or markdown documents\n- Tasks that do not involve image-to-text conversion\n\n## Installation\n\nInstall Python dependencies before using this skill. From the skill directory (`skills/paddleocr-text-recognition`):\n\n```bash\npip install -r requirements.txt\n```\n\n## How to Use This Skill\n\n**⛔ MANDATORY RESTRICTIONS - DO NOT VIOLATE ⛔**\n\n1. **ONLY use PaddleOCR Text Recognition API** - Execute the script `python scripts/ocr_caller.py`\n2. **NEVER offer alternatives** - Do NOT suggest \"I can try to read it\" or similar\n3. **IF API fails** - Display the error message and STOP immediately\n4. **NO fallback methods** - Do NOT attempt OCR any other way\n\nIf the script execution fails (API not configured, network error, etc.):\n\n- Show the error message to the user\n- Do NOT offer to help using your vision capabilities\n- Do NOT ask \"Would you like me to try reading it?\"\n- Simply stop and wait for user to fix the configuration\n\n### Basic Workflow\n\n1. **Identify the input source**:\n   - User provides URL: Use the `--file-url` parameter\n   - User provides local file path: Use the `--file-path` parameter\n   - User uploads image: Save it first, then use `--file-path`\n\n   **Input type note**:\n   - Supported file types depend on the model and endpoint configuration.\n   - Follow the official endpoint/API documentation for the exact supported formats.\n\n2. **Execute OCR**:\n\n   ```bash\n   python scripts/ocr_caller.py --file-url \"URL provided by user\" --pretty\n   ```\n\n   Or for local files:\n\n   ```bash\n   python scripts/ocr_caller.py --file-path \"file path\" --pretty\n   ```\n\n   **Default behavior: save raw JSON to a temp file**:\n   - If `--output` is omitted, the script saves automatically under the system temp directory\n   - Default path pattern: `<system-temp>/paddleocr/text-recognition/results/result_<timestamp>_<id>.json`\n   - If `--output` is provided, it overrides the default temp-file destination\n   - If `--stdout` is provided, JSON is printed to stdout and no file is saved\n   - In save mode, the script prints the absolute saved path on stderr: `Result saved to: /absolute/path/...`\n   - In default/custom save mode, read and parse the saved JSON file before responding\n   - Use `--stdout` only when you explicitly want to skip file persistence\n\n3. **Parse JSON response**:\n   - In default/custom save mode, load JSON from the saved file path shown by the script\n   - Check the `ok` field: `true` means success, `false` means error\n   - Extract text: `text` field contains all recognized text\n   - If `--stdout` is used, parse the stdout JSON directly\n   - Handle errors: If `ok` is false, display `error.message`\n\n4. **Present results to user**:\n   - Display extracted text in a readable format\n   - If the text is empty, the image may contain no text\n   - In save mode, always tell the user the saved file path and that full raw JSON is available there\n\n### IMPORTANT: Complete Output Display\n\n**CRITICAL**: Always display the COMPLETE recognized text to the user. Do NOT truncate or summarize the OCR results.\n\n- The output JSON contains complete output, including full text in `text` field\n- **You MUST display the entire `text` content to the user**, no matter how long it is\n- Do NOT use phrases like \"Here's a summary\" or \"The text begins with...\"\n- Do NOT truncate with \"...\" unless the text truly exceeds reasonable display limits\n- The user expects to see ALL the recognized text, not a preview or excerpt\n\n**Correct approach**:\n\n```\nI've extracted the text from the image. Here's the complete content:\n\n[Display the entire text here]\n```\n\n**Incorrect approach**:\n\n```\nI found some text in the image. Here's a preview:\n\"The quick brown fox...\" (truncated)\n```\n\n### Usage Examples\n\n**Example 1: URL OCR**:\n\n```bash\npython scripts/ocr_caller.py --file-url \"https://example.com/invoice.jpg\" --pretty\n```\n\n**Example 2: Local File OCR**:\n\n```bash\npython scripts/ocr_caller.py --file-path \"./document.pdf\" --pretty\n```\n\n**Example 3: OCR With Explicit File Type**:\n\n```bash\npython scripts/ocr_caller.py --file-url \"https://example.com/input\" --file-type 1 --pretty\n```\n\n**Example 4: Print JSON Without Saving**:\n\n```bash\npython scripts/ocr_caller.py --file-url \"https://example.com/input\" --stdout --pretty\n```\n\n### Understanding the Output\n\nThe output JSON structure is as follows:\n\n```json\n{\n  \"ok\": true,\n  \"text\": \"All recognized text here...\",\n  \"result\": { ... },\n  \"error\": null\n}\n```\n\n**Key fields**:\n\n- `ok`: `true` for success, `false` for error\n- `text`: Complete recognized text\n- `result`: Raw API response (for debugging)\n- `error`: Error details if `ok` is false\n\n> Raw result location (default): the temp-file path printed by the script on stderr\n\n### First-Time Configuration\n\n**When API is not configured**:\n\nThe error will show:\n\n```\nCONFIG_ERROR: PADDLEOCR_OCR_API_URL not configured. Get your API at: https://paddleocr.com\n```\n\n**Configuration workflow**:\n\n1. **Show the exact error message** to the user (including the URL).\n\n2. **Guide the user to configure securely**:\n   - Instruct the user to visit the [PaddleOCR website](https://www.paddleocr.com), click **API**, select the model you need, then copy the `API_URL` and `Token`. They correspond to the API URL (`PADDLEOCR_OCR_API_URL`) and access token (`PADDLEOCR_ACCESS_TOKEN`) used for authentication. Supported model: `PP-OCRv5`.\n   - Optionally, ask the user to configure the request timeout via `PADDLEOCR_OCR_TIMEOUT`.\n   - Recommend configuring through the host application's standard method (e.g., settings file, environment variable UI) rather than pasting credentials in chat. For example, in OpenClaw, environment variables can be set in `~/.openclaw/openclaw.json`.\n\n3. **If the user provides credentials in chat anyway** (accept any reasonable format), for example:\n   - `PADDLEOCR_OCR_API_URL=https://xxx.paddleocr.com/ocr, PADDLEOCR_ACCESS_TOKEN=abc123...`\n   - `Here's my API: https://xxx and token: abc123`\n   - Copy-pasted code format\n\n   Warn the user that credentials shared in chat may be stored in conversation history. Recommend setting them through the host application's configuration instead when possible.\n\n   Then parse and validate the values:\n   - Extract `PADDLEOCR_OCR_API_URL` (look for URLs with `paddleocr.com` or similar)\n   - Confirm `PADDLEOCR_OCR_API_URL` is a full endpoint ending with `/ocr`\n   - Extract `PADDLEOCR_ACCESS_TOKEN` (long alphanumeric string, usually 40+ chars)\n\n4. **Ask the user to confirm the environment is configured**.\n\n5. **Retry only after confirmation**:\n   - Once the user confirms the environment variables are available, retry the original OCR task\n\n### Error Handling\n\n**Authentication failed**:\n\n```\nAPI_ERROR: Authentication failed (403). Check your token.\n```\n\n- Token is invalid, reconfigure with correct credentials\n\n**Quota exceeded**:\n\n```\nAPI_ERROR: API rate limit exceeded (429)\n```\n\n- Daily API quota exhausted, inform user to wait or upgrade\n\n**No text detected**:\n\n- `text` field is empty\n- Image may be blank, corrupted, or contain no text\n\n### Tips for Better Results\n\nIf recognition quality is poor, suggest:\n\n- Check if the image is clear and contains text\n- Provide a higher resolution image if possible\n\n## Reference Documentation\n\nFor in-depth understanding of the OCR system, refer to:\n\n- `references/output_schema.md` - Output format specification\n\n> **Note**: Model version, capabilities, and supported file formats are determined by your API endpoint (`PADDLEOCR_OCR_API_URL`) and its official API documentation.\n\n## Testing the Skill\n\nTo verify the skill is working properly:\n\n```bash\npython scripts/smoke_test.py\n```\n\nThis tests configuration and API connectivity.\n\nFile v1.0.14:_meta.json\n\n{\n  \"ownerId\": \"kn77zppfj1a2fc620aygaf9z9980ewfa\",\n  \"slug\": \"paddleocr-text-recognition\",\n  \"version\": \"1.0.14\",\n  \"publishedAt\": 1774593467388\n}\n\nFile v1.0.14:references/output_schema.md\n\n# PaddleOCR Text Recognition Output Schema\n\nThis document defines the output envelope returned by `ocr_caller.py`.\n\nBy default, `ocr_caller.py` saves the JSON envelope to a unique file under the system temp directory and prints the absolute saved path to `stderr`. Use `--output` when you need a custom destination, or `--stdout` when you want to skip file saving and print JSON directly.\n\n## Output Envelope\n\n`ocr_caller.py` wraps provider response in a stable structure:\n\n```json\n{\n  \"ok\": true,\n  \"text\": \"Extracted text from all pages\",\n  \"result\": { ... },  // raw provider response\n  \"error\": null\n}\n```\n\nOn error:\n\n```json\n{\n  \"ok\": false,\n  \"text\": \"\",\n  \"result\": null,\n  \"error\": {\n    \"code\": \"ERROR_CODE\",\n    \"message\": \"Human-readable message\"\n  }\n}\n```\n\n## Error Codes\n\n| Code           | Description                                                                |\n| -------------- | -------------------------------------------------------------------------- |\n| `INPUT_ERROR`  | Invalid input (missing file, unsupported format, invalid file type)        |\n| `CONFIG_ERROR` | API not configured                                                         |\n| `API_ERROR`    | API call failed (auth, timeout, service error, or invalid response schema) |\n\n## Raw Result Notes\n\nThe `result` field contains raw provider output.  \nRaw fields may vary by model version and endpoint.\n\n## Raw Result Example\n\n```json\n{\n  \"logId\": \"request-uuid\",\n  \"errorCode\": 0,\n  \"errorMsg\": \"Success\",\n  \"result\": {\n    \"ocrResults\": [\n      {\n        \"prunedResult\": {\n          \"rec_texts\": [\"First line\", \"Second line\"],\n          \"rec_scores\": [0.98, 0.95],\n          \"...\": \"other OCR fields\"\n        },\n        \"ocrImage\": \"https://...\",\n        \"inputImage\": \"https://...\",\n        \"...\": \"other model-specific fields\"\n      }\n    ],\n    \"dataInfo\": {\n      \"numPages\": 1,\n      \"type\": \"pdf\",\n      \"...\": \"other metadata\"\n    },\n    \"...\": \"other top-level fields\"\n  }\n}\n```\n\n## Stable Fields for Downstream Use\n\n- `result[n].prunedResult`  \n  Structured OCR data for page `n`.\n\n- `result[n].prunedResult.rec_texts`  \n  Recognized text lines for page `n`.\n\n- `result[n].prunedResult.rec_scores`  \n  Confidence scores for recognized text lines.\n\n## Text Extraction\n\n`ocr_caller.py` extracts top-level `text` from `result.ocrResults[n].prunedResult.rec_texts`, joins lines with `\\n`, and joins pages with `\\n\\n`.\n\n## Command Examples\n\n```bash\n# OCR from URL (result auto-saves to the system temp directory)\npython scripts/ocr_caller.py --file-url \"URL\" --pretty\n\n# OCR local file (result auto-saves to the system temp directory)\npython scripts/ocr_caller.py --file-path \"doc.pdf\" --pretty\n\n# OCR with explicit file type\npython scripts/ocr_caller.py --file-url \"URL\" --file-type 1 --pretty\n\n# Save result to a custom file path\npython scripts/ocr_caller.py --file-url \"URL\" --output \"./result.json\" --pretty\n\n# Print JSON to stdout without saving a file\npython scripts/ocr_caller.py --file-url \"URL\" --stdout --pretty\n```\n\nFile v1.0.14:requirements.txt\n\n# PaddleOCR Text Recognition Dependencies\n\nhttpx>=0.24.0\n\nArchive v1.0.13: 7 files, 13235 bytes\n\nFiles: references/output_schema.md (2940b), scripts/lib.py (9283b), scripts/ocr_caller.py (5000b), scripts/requirements.txt (57b), scripts/smoke_test.py (4714b), SKILL.md (9255b), _meta.json (146b)\n\nFile v1.0.13:SKILL.md\n\n---\nname: paddleocr-text-recognition\ndescription: >-\n  Use this skill whenever the user wants text extracted from images, photos, scans, screenshots,\n  or scanned PDFs. Returns exact machine-readable strings with line-level text and optional bbox\n  coordinates. Strong accuracy for CJK, small print, and handwritten text. Supports batch/multi-image\n  runs.\n  Trigger terms: OCR, 文字识别, 图片转文字, 截图识字, 提取图中文字, 扫描识字, 识字, 纯文字,\n  plain text extraction, 坐标, 检测框, bbox, bounding box, image to text, screenshot, photo scan,\n  recognize text.\nmetadata:\n  openclaw:\n    requires:\n      env:\n        - PADDLEOCR_OCR_API_URL\n        - PADDLEOCR_ACCESS_TOKEN\n        - PADDLEOCR_OCR_TIMEOUT\n      bins:\n        - python\n    primaryEnv: PADDLEOCR_ACCESS_TOKEN\n    emoji: \"🔤\"\n    homepage: https://github.com/PaddlePaddle/PaddleOCR/tree/main/skills/paddleocr-text-recognition\n---\n\n# PaddleOCR Text Recognition Skill\n\n## When to Use This Skill\n\n**Trigger keywords (routing)**: Bilingual trigger terms (Chinese and English) are listed in the YAML `description` above—use that field for discovery and routing.\n\nInvoke this skill in the following situations:\n- Extract text from images (screenshots, photos, scans)\n- Extract text from PDFs or document images when the goal is **line/box-level text**, not recovering table grids, formulas, or full reading-order layout\n- Extract text from URLs or local files that point to images/PDFs\n\nDo not use this skill in the following situations:\n- Plain text files that can be read directly with the Read tool\n- Code files or markdown documents\n- Tasks that do not involve image-to-text conversion\n\n## Installation\n\nInstall Python dependencies before using this skill. From the skill directory (`skills/paddleocr-text-recognition`):\n\n```bash\npip install -r scripts/requirements.txt\n```\n\n## How to Use This Skill\n\n**⛔ MANDATORY RESTRICTIONS - DO NOT VIOLATE ⛔**\n\n1. **ONLY use PaddleOCR Text Recognition API** - Execute the script `python scripts/ocr_caller.py`\n2. **NEVER offer alternatives** - Do NOT suggest \"I can try to read it\" or similar\n3. **IF API fails** - Display the error message and STOP immediately\n4. **NO fallback methods** - Do NOT attempt OCR any other way\n\nIf the script execution fails (API not configured, network error, etc.):\n- Show the error message to the user\n- Do NOT offer to help using your vision capabilities\n- Do NOT ask \"Would you like me to try reading it?\"\n- Simply stop and wait for user to fix the configuration\n\n### Basic Workflow\n\n1. **Identify the input source**:\n   - User provides URL: Use the `--file-url` parameter\n   - User provides local file path: Use the `--file-path` parameter\n   - User uploads image: Save it first, then use `--file-path`\n\n   **Input type note**:\n   - Supported file types depend on the model and endpoint configuration.\n   - Follow the official endpoint/API documentation for the exact supported formats.\n\n2. **Execute OCR**:\n   ```bash\n   python scripts/ocr_caller.py --file-url \"URL provided by user\" --pretty\n   ```\n   Or for local files:\n   ```bash\n   python scripts/ocr_caller.py --file-path \"file path\" --pretty\n   ```\n\n   **Default behavior: save raw JSON to a temp file**:\n   - If `--output` is omitted, the script saves automatically under the system temp directory\n   - Default path pattern: `<system-temp>/paddleocr/text-recognition/results/result_<timestamp>_<id>.json`\n   - If `--output` is provided, it overrides the default temp-file destination\n   - If `--stdout` is provided, JSON is printed to stdout and no file is saved\n   - In save mode, the script prints the absolute saved path on stderr: `Result saved to: /absolute/path/...`\n   - In default/custom save mode, read and parse the saved JSON file before responding\n   - Use `--stdout` only when you explicitly want to skip file persistence\n\n3. **Parse JSON response**:\n   - In default/custom save mode, load JSON from the saved file path shown by the script\n   - Check the `ok` field: `true` means success, `false` means error\n   - Extract text: `text` field contains all recognized text\n   - If `--stdout` is used, parse the stdout JSON directly\n   - Handle errors: If `ok` is false, display `error.message`\n\n4. **Present results to user**:\n   - Display extracted text in a readable format\n   - If the text is empty, the image may contain no text\n   - In save mode, always tell the user the saved file path and that full raw JSON is available there\n\n### IMPORTANT: Complete Output Display\n\n**CRITICAL**: Always display the COMPLETE recognized text to the user. Do NOT truncate or summarize the OCR results.\n\n- The output JSON contains complete output, including full text in `text` field\n- **You MUST display the entire `text` content to the user**, no matter how long it is\n- Do NOT use phrases like \"Here's a summary\" or \"The text begins with...\"\n- Do NOT truncate with \"...\" unless the text truly exceeds reasonable display limits\n- The user expects to see ALL the recognized text, not a preview or excerpt\n\n**Correct approach**:\n```\nI've extracted the text from the image. Here's the complete content:\n\n[Display the entire text here]\n```\n\n**Incorrect approach**:\n```\nI found some text in the image. Here's a preview:\n\"The quick brown fox...\" (truncated)\n```\n\n### Usage Examples\n\n**Example 1: URL OCR**:\n```bash\npython scripts/ocr_caller.py --file-url \"https://example.com/invoice.jpg\" --pretty\n```\n\n**Example 2: Local File OCR**:\n```bash\npython scripts/ocr_caller.py --file-path \"./document.pdf\" --pretty\n```\n\n**Example 3: OCR With Explicit File Type**:\n```bash\npython scripts/ocr_caller.py --file-url \"https://example.com/input\" --file-type 1 --pretty\n```\n\n**Example 4: Print JSON Without Saving**:\n```bash\npython scripts/ocr_caller.py --file-url \"https://example.com/input\" --stdout --pretty\n```\n\n### Understanding the Output\n\nThe output JSON structure is as follows:\n```json\n{\n  \"ok\": true,\n  \"text\": \"All recognized text here...\",\n  \"result\": { ... },\n  \"error\": null\n}\n```\n\n**Key fields**:\n- `ok`: `true` for success, `false` for error\n- `text`: Complete recognized text\n- `result`: Raw API response (for debugging)\n- `error`: Error details if `ok` is false\n\n> Raw result location (default): the temp-file path printed by the script on stderr\n\n### First-Time Configuration\n\n**When API is not configured**:\n\nThe error will show:\n```\nCONFIG_ERROR: PADDLEOCR_OCR_API_URL not configured. Get your API at: https://paddleocr.com\n```\n\n**Configuration workflow**:\n\n1. **Show the exact error message** to the user (including the URL).\n\n2. **Guide the user to configure securely**:\n   - Instruct the user to visit the [PaddleOCR website](https://www.paddleocr.com), click **API**, select the model you need, then copy the `API_URL` and `Token`. They correspond to the API URL (`PADDLEOCR_OCR_API_URL`) and access token (`PADDLEOCR_ACCESS_TOKEN`) used for authentication. Supported model: `PP-OCRv5`.\n   - Optionally, ask the user to configure the request timeout via `PADDLEOCR_OCR_TIMEOUT`.\n   - Recommend configuring through the host application's standard method (e.g., settings file, environment variable UI) rather than pasting credentials in chat. For example, in OpenClaw, environment variables can be set in `~/.openclaw/openclaw.json`.\n\n3. **If the user provides credentials in chat anyway** (accept any reasonable format), for example:\n   - `PADDLEOCR_OCR_API_URL=https://xxx.paddleocr.com/ocr, PADDLEOCR_ACCESS_TOKEN=abc123...`\n   - `Here's my API: https://xxx and token: abc123`\n   - Copy-pasted code format\n   \n   Warn the user that credentials shared in chat may be stored in conversation history. Recommend setting them through the host application's configuration instead when possible.\n\n   Then parse and validate the values:\n   - Extract `PADDLEOCR_OCR_API_URL` (look for URLs with `paddleocr.com` or similar)\n   - Confirm `PADDLEOCR_OCR_API_URL` is a full endpoint ending with `/ocr`\n   - Extract `PADDLEOCR_ACCESS_TOKEN` (long alphanumeric string, usually 40+ chars)\n\n4. **Ask the user to confirm the environment is configured**.\n\n5. **Retry only after confirmation**:\n   - Once the user confirms the environment variables are available, retry the original OCR task\n\n### Error Handling\n\n**Authentication failed**:\n```\nAPI_ERROR: Authentication failed (403). Check your token.\n```\n- Token is invalid, reconfigure with correct credentials\n\n**Quota exceeded**:\n```\nAPI_ERROR: API rate limit exceeded (429)\n```\n- Daily API quota exhausted, inform user to wait or upgrade\n\n**No text detected**:\n- `text` field is empty\n- Image may be blank, corrupted, or contain no text\n\n### Tips for Better Results\n\nIf recognition quality is poor, suggest:\n- Check if the image is clear and contains text\n- Provide a higher resolution image if possible\n\n## Reference Documentation\n\nFor in-depth understanding of the OCR system, refer to:\n- `references/output_schema.md` - Output format specification\n\n> **Note**: Model version, capabilities, and supported file formats are determined by your API endpoint (`PADDLEOCR_OCR_API_URL`) and its official API documentation.\n\n## Testing the Skill\n\nTo verify the skill is working properly:\n```bash\npython scripts/smoke_test.py\n```\n\nThis tests configuration and API connectivity.\n\nFile v1.0.13:_meta.json\n\n{\n  \"ownerId\": \"kn77zppfj1a2fc620aygaf9z9980ewfa\",\n  \"slug\": \"paddleocr-text-recognition\",\n  \"version\": \"1.0.13\",\n  \"publishedAt\": 1774540297025\n}\n\nFile v1.0.13:references/output_schema.md\n\n# PaddleOCR Text Recognition Output Schema\n\nThis document defines the output envelope returned by `ocr_caller.py`.\n\nBy default, `ocr_caller.py` saves the JSON envelope to a unique file under the system temp directory and prints the absolute saved path to `stderr`. Use `--output` when you need a custom destination, or `--stdout` when you want to skip file saving and print JSON directly.\n\n## Output Envelope\n\n`ocr_caller.py` wraps provider response in a stable structure:\n\n```json\n{\n  \"ok\": true,\n  \"text\": \"Extracted text from all pages\",\n  \"result\": { ... },  // raw provider response\n  \"error\": null\n}\n```\n\nOn error:\n\n```json\n{\n  \"ok\": false,\n  \"text\": \"\",\n  \"result\": null,\n  \"error\": {\n    \"code\": \"ERROR_CODE\",\n    \"message\": \"Human-readable message\"\n  }\n}\n```\n\n## Error Codes\n\n| Code | Description |\n|------|-------------|\n| `INPUT_ERROR` | Invalid input (missing file, unsupported format, invalid file type) |\n| `CONFIG_ERROR` | API not configured |\n| `API_ERROR` | API call failed (auth, timeout, service error, or invalid response schema) |\n\n## Raw Result Notes\n\nThe `result` field contains raw provider output.  \nRaw fields may vary by model version and endpoint.\n\n## Raw Result Example\n\n```json\n{\n  \"logId\": \"request-uuid\",\n  \"errorCode\": 0,\n  \"errorMsg\": \"Success\",\n  \"result\": {\n    \"ocrResults\": [\n      {\n        \"prunedResult\": {\n          \"rec_texts\": [\"First line\", \"Second line\"],\n          \"rec_scores\": [0.98, 0.95],\n          \"...\": \"other OCR fields\"\n        },\n        \"ocrImage\": \"https://...\",\n        \"inputImage\": \"https://...\",\n        \"...\": \"other model-specific fields\"\n      }\n    ],\n    \"dataInfo\": {\n      \"numPages\": 1,\n      \"type\": \"pdf\",\n      \"...\": \"other metadata\"\n    },\n    \"...\": \"other top-level fields\"\n  }\n}\n```\n\n## Stable Fields for Downstream Use\n\n- `result[n].prunedResult`  \n  Structured OCR data for page `n`.\n\n- `result[n].prunedResult.rec_texts`  \n  Recognized text lines for page `n`.\n\n- `result[n].prunedResult.rec_scores`  \n  Confidence scores for recognized text lines.\n\n## Text Extraction\n\n`ocr_caller.py` extracts top-level `text` from `result.ocrResults[n].prunedResult.rec_texts`, joins lines with `\\n`, and joins pages with `\\n\\n`.\n\n## Command Examples\n\n```bash\n# OCR from URL (result auto-saves to the system temp directory)\npython scripts/paddleocr-text-recognition/ocr_caller.py --file-url \"URL\" --pretty\n\n# OCR local file (result auto-saves to the system temp directory)\npython scripts/paddleocr-text-recognition/ocr_caller.py --file-path \"doc.pdf\" --pretty\n\n# OCR with explicit file type\npython scripts/paddleocr-text-recognition/ocr_caller.py --file-url \"URL\" --file-type 1 --pretty\n\n# Save result to a custom file path\npython scripts/paddleocr-text-recognition/ocr_caller.py --file-url \"URL\" --output \"./result.json\" --pretty\n\n# Print JSON to stdout without saving a file\npython scripts/paddleocr-text-recognition/ocr_caller.py --file-url \"URL\" --stdout --pretty\n```\n\nFile v1.0.13:scripts/requirements.txt\n\n# PaddleOCR Text Recognition Dependencies\n\nhttpx>=0.24.0","readmeExcerpt":"Skill: PaddleOCR Text Recognition Owner: bobholamovic Summary: Use this skill whenever the user wants text extracted from images, photos, scans, screenshots, or scanned PDFs. Returns exact machine-readable strings with l... Tags: latest:2.0.0 Version history: v2.0.0 | 2026-06-05T10:10:23.964Z | user - Major refactor: removed all local wrapper scripts and sample/reference files, switching to direct CLI usage. - SKILL.","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"paddleocr api \\\n  --model_type ocr \\\n  --file_url \"https://example.com/image.png\""},{"language":"bash","snippet":"paddleocr api \\\n  --model_type ocr \\\n  --file_path \"./document.pdf\""},{"language":"bash","snippet":"# With specific model\npaddleocr api \\\n  --model_type ocr \\\n  --model PP-OCRv5 \\\n  --file_path \"./report.pdf\"\n\n# Disable preprocessing (faster, for flat/well-oriented images)\npaddleocr api \\\n  --model_type ocr \\\n  --file_path \"./document.pdf\" \\\n  --use_doc_unwarping False \\\n  --use_doc_orientation_classify False\n\n# Save result to file\npaddleocr api \\\n  --model_type ocr \\\n  --file_url \"https://...\" \\\n  --output result.json\n\n# Page ranges\npaddleocr api \\\n  --model_type ocr \\\n  --file_path \"./large.pdf\" \\\n  --page_ranges \"1-5,10,15-20\""},{"language":"json","snippet":"{\n  \"jobId\": \"job-xxx\",\n  \"pages\": [\n    {\n      \"prunedResult\": {\n        \"rec_texts\": [\"Line 1\", \"Line 2\"],\n        \"rec_scores\": [0.98, 0.95]\n      },\n      \"ocrImageUrl\": \"https://...\"\n    }\n  ]\n}"},{"language":"bash","snippet":"paddleocr api --model_type ocr --file_path \"./document.pdf\" --use_doc_unwarping False --use_doc_orientation_classify False"},{"language":"bash","snippet":"uv run scripts/ocr_caller.py --help"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: paddleocr-text-recognition\ndescription: >-\n  Use this skill whenever the user wants text extracted from images, photos, scans, screenshots,\n  or scanned PDFs. Returns exact machine-readable strings with line-level text and optional bbox\n  coordinates. Strong accuracy for CJK, small print, and handwritten text.\n  Trigger terms: OCR, 文字识别, 图片转文字, 截图识字, 提取图中文字, 扫描识字, 识字, 纯文字,\n  plain text extraction, 坐标, 检测框, bbox, bounding box, image to text, screenshot, photo scan,\n  recognize text.\nlicense: Apache-2.0\nmetadata:\n  openclaw:\n    requires:\n      env:\n        - PADDLEOCR_ACCESS_TOKEN\n      bins:\n        - paddleocr\n    primaryEnv: PADDLEOCR_ACCESS_TOKEN\n    emoji: \"🔤\"\n    install:\n      - kind: uv\n        package: paddleocr\n        bins: [paddleocr]\n---\n\n# PaddleOCR Text Recognition\n\n## When to Use This Skill\n\n**Use this skill for**:\n\n- Extract text from images (screenshots, photos, scans)\n- Extract text from PDFs or document images when the goal is **line/box-level text**\n- Extract text from URLs or local files that point to images/PDFs\n\n**Do not use for**:\n\n- Documents with tables, formulas, charts, or complex layouts — use Document Parsing instead\n\n## Usage\n\n### Basic OCR\n\nFrom URL:\n\n```bash\npaddleocr api \\\n  --model_type ocr \\\n  --file_url \"https://example.com/image.png\"\n```\n\nFrom local file:\n\n```bash\npaddleocr api \\\n  --model_type ocr \\\n  --file_path \"./document.pdf\"\n```\n\n### Common Options\n\n```bash\n# With specific model\npaddleocr api \\\n  --model_type ocr \\\n  --model PP-OCRv5 \\\n  --file_path \"./report.pdf\"\n\n# Disable preprocessing (faster, for flat/well-oriented images)\npaddleocr api \\\n  --model_type ocr \\\n  --file_path \"./document.pdf\" \\\n  --use_doc_unwarping False \\\n  --use_doc_orientation_classify False\n\n# Save result to file\npaddleocr api \\\n  --model_type ocr \\\n  --file_url \"https://...\" \\\n  --output result.json\n\n# Page ranges\npaddleocr api \\\n  --model_type ocr \\\n  --file_path \"./large.pdf\" \\\n  --page_ranges \"1-5,10,15-20\"\n```\n\n### Output Format\n\n```json\n{\n  \"jobId\": \"job-xxx\",\n  \"pages\": [\n    {\n      \"prunedResult\": {\n        \"rec_texts\": [\"Line 1\", \"Line 2\"],\n        \"rec_scores\": [0.98, 0.95]\n      },\n      \"ocrImageUrl\": \"https://...\"\n    }\n  ]\n}\n```\n\n## Important Notes\n\n**Preprocessing options**: By default, the API enables document preprocessing (unwarping and orientation classification). For flat, well-oriented images (screenshots, properly scanned documents), you can disable preprocessing for faster results:\n\n```bash\npaddleocr api --model_type ocr --file_path \"./document.pdf\" --use_doc_unwarping False --use_doc_orientation_classify False\n```\n\nKeep preprocessing enabled when:\n- The input is a photo of a curved or folded document\n- The document has significant perspective distortion\n- Orientation is uncertain (rotated 90/180/270 degrees)\n\n**Display complete results**: Always show the full extracted content to users. Do not truncate with \"...\" unless content exceeds 10,000 characters. When multiple pages are processed, summa"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn77zppfj1a2fc620aygaf9z9980ewfa\",\n  \"slug\": \"paddleocr-text-recognition\",\n  \"version\": \"2.0.0\",\n  \"publishedAt\": 1780654223964\n}"},{"path":"skill-card.md","content":"## Description:\n\nExtracts machine-readable OCR text from images, screenshots, scans, and scanned PDFs, with line-level text and optional bounding boxes.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[bobholamovic](https://clawhub.ai/user/bobholamovic)\n\n### License/Terms of Use:\n\nApache-2.0\n\n## Use Case:\n\nDevelopers and agent users use this skill to extract plain text from images, screenshots, photos, scans, and scanned PDFs through the PaddleOCR CLI. It is best suited for line-level OCR output and not for tables, formulas, charts, or complex document layouts.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Local images or PDFs may be sent to PaddleOCR's external API without a clear privacy confirmation.\n\nMitigation: Use only approved data with the external OCR service; avoid submitting secrets, regulated records, internal screenshots, or confidential documents unless that service is approved for that data.\n\nRisk: Runtime dependency behavior can change if the PaddleOCR package is installed without controls.\n\nMitigation: Use a pinned dependency version or controlled runtime for higher-risk use.\n\nRisk: OCR output may be incomplete or unsuitable for tables, formulas, charts, and complex document layouts.\n\nMitigation: Use a document parsing workflow for complex layouts and review OCR results before relying on them.\n\n## Reference(s):\n\n- [PaddleOCR Official CLI Documentation](https://www.paddleocr.ai/latest/en/version3.x/inference_deployment/serving/paddleocr_official_api/cli.html)\n- [ClawHub Skill Page](https://clawhub.ai/bobholamovic/skills/paddleocr-text-recognition)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown guidance with shell command examples and JSON OCR result descriptions]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Returns complete extracted content when feasible; examples include line text, recognition scores, optional bounding boxes, page ranges, and saved JSON output.]\n\n## Skill Version(s):\n\n2.0.0 (source: ClawHub release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Use this skill whenever the user wants text extracted from images, photos, scans, screenshots, or scanned PDFs. Returns exact machine-readable strings with l... Skill: PaddleOCR Text Recognition Owner: bobholamovic Summary: Use this skill whenever the user wants text extracted from images, photos, scans, screenshots, or scanned PDFs. Returns exact machine-readable strings with l... Tags: latest:2.0.0 Version history: v2.0.0 | 2026-06-05T10:10:23.964Z | user - Major refactor: removed all local wrapper scripts and sample/reference files, switching to direct CLI usage. - SKILL.","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1303,"uniquenessScore":48,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T05:14:42.977Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T05:14:42.977Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T13:54:14.345Z","emptyReason":null},"items":[{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-10T18:48:31.762Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}