{"id":"57c6bb86-29bd-4c1e-ae54-ea5b2929b0c8","entityType":"agent","slug":"clawhub-ltryee-ocr-locally","name":"OCR Locally","canonicalUrl":"https://www.xpersona.co/agent/clawhub-ltryee-ocr-locally","canonicalPath":"/agent/clawhub-ltryee-ocr-locally","generatedAt":"2026-10-10T21:43:10.016Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T18:41:03.089Z","emptyReason":null},"description":"[macOS only] Use this skill when the user requests OCR (Optical Character Recognition), image/PDF text extraction. Uses macOS native Vision/PDFKit frameworks...","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.3K downloads reported by the source. Last updated 10/10/2026.","installCommand":"clawhub skill install s17f15dn3gc74a13v3v4g132vs856fkf:ocr-locally","sourceUrl":"https://clawhub.ai/ltryee/ocr-locally","homepage":"https://clawhub.ai/ltryee/skills/ocr-locally","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/ltryee/ocr-locally","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/ltryee/skills/ocr-locally","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":62,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"OCR Locally technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-10T18:41:03.089Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T18:41:03.089Z","emptyReason":null},"stars":null,"forks":null,"downloads":1293,"packageName":null,"latestVersion":"1.0.0","tractionLabel":"1.3K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T18:41:03.089Z","emptyReason":null},"lastUpdatedAt":"2026-10-10T18:41:03.089Z","lastCrawledAt":"2026-10-10T18:41:03.089Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-11T18:41:03.089Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.0","createdAt":"2026-05-04T12:14:05.706Z","changelog":"Initial release of local-ocr, providing offline OCR capabilities on macOS. - Supports image and PDF text extraction using macOS native Vision and PDFKit frameworks (macOS 10.15+ required) - Recognizes multiple languages (Chinese, English, Japanese, Korean, and more) - Two output modes: pure text or detailed JSON with confidence and bounding boxes - Handles a variety of image formats (PNG, JPEG, TIFF, BMP) and PDFs, with support for specific page selection - Command-line interface with flexible options for language, output mode, and file paths","fileCount":7,"zipByteSize":19788}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17f15dn3gc74a13v3v4g132vs856fkf:ocr-locally","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s17f15dn3gc74a13v3v4g132vs856fkf:ocr-locally` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/ltryee/ocr-locally before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-ltryee-ocr-locally/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-ltryee-ocr-locally/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-ltryee-ocr-locally/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-ltryee-ocr-locally/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-ltryee-ocr-locally/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-ltryee-ocr-locally/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T21:43:10.015Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-ltryee-ocr-locally/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-ltryee-ocr-locally/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-ltryee-ocr-locally/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-ltryee-ocr-locally/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T18:41:03.089Z","emptyReason":null},"readme":"Skill: OCR Locally\n\nOwner: ltryee\n\nSummary: [macOS only] Use this skill when the user requests OCR (Optical Character Recognition), image/PDF text extraction. Uses macOS native Vision/PDFKit frameworks...\n\nTags: latest:1.0.0\n\nVersion history:\n\nv1.0.0 | 2026-05-04T12:14:05.706Z | user\n\nInitial release of local-ocr, providing offline OCR capabilities on macOS.\n\n- Supports image and PDF text extraction using macOS native Vision and PDFKit frameworks (macOS 10.15+ required)\n- Recognizes multiple languages (Chinese, English, Japanese, Korean, and more)\n- Two output modes: pure text or detailed JSON with confidence and bounding boxes\n- Handles a variety of image formats (PNG, JPEG, TIFF, BMP) and PDFs, with support for specific page selection\n- Command-line interface with flexible options for language, output mode, and file paths\n\nArchive index:\n\nArchive v1.0.0: 7 files, 19788 bytes\n\nFiles: references/usage.md (15727b), scripts/ocr_vision_pro.swift (12488b), scripts/ocr_vision.swift (2851b), scripts/pdf_ocr.swift (12817b), skill-card.md (2174b), SKILL.md (10596b), _meta.json (130b)\n\nFile v1.0.0:SKILL.md\n\n---\nname: local-ocr\ndescription: \"[macOS only] Use this skill when the user requests OCR (Optical Character Recognition), image/PDF text extraction. Uses macOS native Vision/PDFKit frameworks. Triggers: '识别图片', 'OCR', '提取图片文字', '提取PDF文字', '识别PDF', 'extract text from image', 'PDF OCR'.\"\n---\n\n# Local OCR (macOS Only)\n\n## Overview\n\n⚠️ **Platform Requirement**: This skill is **macOS only**. It requires macOS 10.15+ (Catalina or later) and uses macOS native frameworks:\n\n- **Vision framework** - For OCR text recognition\n- **PDFKit framework** - For PDF processing\n- **Core Graphics** - For image rendering\n\nThis skill provides OCR (Optical Character Recognition) capabilities using macOS native Vision framework. It extracts text from images and PDFs without requiring any third-party libraries or internet connection.\n\n## Platform Requirements\n\n⚠️ **macOS Only** - This skill cannot run on Linux, Windows, or other operating systems.\n\n**Required:**\n- macOS 10.15+ (Catalina or later)\n- Vision framework (pre-installed on macOS)\n- PDFKit framework (pre-installed on macOS)\n\n**Why macOS Only?**\n- Uses `Vision` framework for OCR (macOS/iOS only)\n- Uses `PDFKit` framework for PDF processing (macOS/iOS only)\n- Uses `AppKit`/`Core Graphics` for image handling (macOS only)\n\n## When to Use This Skill\n\nTrigger this skill when the user:\n- Requests OCR or image text extraction\n- Mentions extracting text from images, screenshots, PDF files, or scanned documents\n- Uses keywords like: \"识别图片\", \"OCR\", \"提取文字\", \"提取PDF文字\", \"识别PDF\", \"extract text from image\", \"PDF OCR\"\n- Provides an image file or PDF file and asks to read or extract its content\n\n## Core Capabilities\n\n### 1. Text Extraction from Images\n\nUse `scripts/ocr_vision_pro.swift` for comprehensive OCR with the following features:\n- Multi-language support (Chinese, English, Japanese, Korean, and more)\n- **Two output modes** (mutually exclusive):\n  - **Text Mode** (`-t`): Output only extracted text (default)\n  - **JSON Mode** (`-j`): Output complete raw info including text, position, and confidence as JSON\n- Confidence scores for each detected text block\n- Bounding box information (text position in image)\n- Output to console or file\n- Precise or fast recognition modes\n\n**Basic usage:**\n```bash\nswift scripts/ocr_vision_pro.swift <image_path>\n```\n\n**With options:**\n```bash\nswift scripts/ocr_vision_pro.swift <image_path> -l zh-Hans,en -o output.txt -f\n```\n\n### 2. Text Extraction from PDF Files\n\nUse `scripts/pdf_ocr.swift` to extract text from PDF files with the following features:\n- Extract text from specific pages or all pages\n- Support page range specification (e.g., `1-5`, `1,3,5`)\n- **Two output modes** (mutually exclusive):\n  - **Text Mode** (`-t`): Output only extracted text (default)\n  - **JSON Mode** (`-j`): Output complete raw info as JSON\n- Same multi-language support as image OCR\n- Precise or fast recognition modes\n\n**Basic usage (all pages):**\n```bash\nswift scripts/pdf_ocr.swift <pdf_path>\n```\n\n**With page specification:**\n```bash\n# Single page\nswift scripts/pdf_ocr.swift document.pdf -p 1\n\n# Multiple pages\nswift scripts/pdf_ocr.swift document.pdf -p 1,3,5\n\n# Page range\nswift scripts/pdf_ocr.swift document.pdf -p 1-5\n\n# JSON mode\nswift scripts/pdf_ocr.swift document.pdf -p 1 -j\n```\n\n### 3. Output Modes (Mutually Exclusive)\n\nThe script supports two output modes that cannot be used simultaneously:\n\n#### Text Mode (Default, `-t`)\nOutputs only the extracted text:\n- Console output: Pure text\n- File output (`-o` or `-t`): Saves text to file, optionally with separate confidence file\n\n#### JSON Mode (`-j`)\nOutputs complete raw information as JSON:\n- Contains: image path, total blocks, average confidence, and per-block details\n- Per-block info: index, text, confidence, bounding box (x, y, width, height)\n- Outputs to stdout only (no file output options in JSON mode)\n\n**JSON output structure:**\n```json\n{\n  \"imagePath\": \"/path/to/image.png\",\n  \"totalBlocks\": 25,\n  \"averageConfidence\": 0.85,\n  \"blocks\": [\n    {\n      \"index\": 1,\n      \"text\": \"recognized text\",\n      \"confidence\": 0.95,\n      \"boundingBox\": {\n        \"x\": 0.10,\n        \"y\": 0.20,\n        \"width\": 0.30,\n        \"height\": 0.05\n      }\n    }\n  ]\n}\n```\n\n### 4. Supported File Formats\n\n**Image Formats (ocr_vision_pro.swift):**\n- PNG (.png)\n- JPEG (.jpg, .jpeg)\n- TIFF (.tiff, .tif)\n- BMP (.bmp)\n\n**PDF Format (pdf_ocr.swift):**\n- PDF (.pdf) - support single page, multiple pages, or page ranges\n- Specify pages with `-p` option: `1`, `1,3,5`, or `1-5`\n\n### 4. Command-Line Options\n\n| Option | Description |\n|--------|-------------|\n| `-h`, `--help` | Show help information |\n| `-t`, `--text` | Text mode (default, output only extracted text) |\n| `-j`, `--json` | JSON mode (output complete raw info as JSON) |\n| `-l`, `--language <lang>` | Specify recognition language (comma-separated) |\n| `-o`, `--output <file>` | Output text to file, auto-generate confidence file (`<file>_confidence.txt`) |\n| `-t`, `--text <file>` | Output only complete text to specified file (text mode) |\n| `-c`, `--confidence <file>` | Output only confidence details to specified file (text mode) |\n| `-f`, `--fast` | Use fast mode (default: precise mode) |\n\n**Note**: `-t` (text mode) and `-j` (JSON mode) are mutually exclusive. JSON mode outputs to stdout only.\n\n**Supported languages:**\n- `zh-Hans` - Simplified Chinese\n- `zh-Hant` - Traditional Chinese\n- `en` - English\n- `ja` - Japanese\n- `ko` - Korean\n- `fr` - French\n- `de` - German\n- `es` - Spanish\n- `it` - Italian\n- `pt` - Portuguese\n- `ru` - Russian\n\n## Workflow\n\n### Step 1: Identify the Image Path\n\nWhen the user requests OCR:\n1. Ask for the image path if not provided\n2. Accept common path formats: absolute paths, ~/path, or relative paths\n3. Validate that the file exists before proceeding\n\n### Step 2: Determine Recognition Parameters\n\nBased on user request or context:\n1. **Language**: Default to `zh-Hans,en` for Chinese users, or `en` for English users\n2. **Mode**: Use precise mode (default) for accuracy, fast mode (`-f`) for quick preview\n3. **Output**: Ask if user wants results saved to file (`-o` option)\n\n### Step 3: Execute OCR\n\nRun the OCR script with appropriate parameters:\n\n```bash\nswift scripts/ocr_vision_pro.swift \"<image_path>\" -l zh-Hans,en\n```\n\nFor saving to separate files (recommended):\n```bash\nswift scripts/ocr_vision_pro.swift \"<image_path>\" -o \"<output>\"\n```\n\nThis automatically creates two files:\n- `<output>.txt` - Complete extracted text (pure text, no formatting)\n- `<output>_confidence.txt` - Confidence details with statistics and per-block info\n\nFor separate text and confidence files with custom names:\n```bash\nswift scripts/ocr_vision_pro.swift \"<image_path>\" -t \"text.txt\" -c \"confidence.txt\"\n```\n\n### Step 4: Present Results\n\nAfter OCR completes:\n1. Display the extracted text to the user\n2. If saved to file, inform the user of the output file path\n3. Ask if user wants to:\n   - Correct misrecognized characters\n   - Process another image\n   - Save results in a different format\n\n### Step 5: PDF Processing (if PDF file)\n\nWhen processing a PDF file:\n1. **Identify PDF path and pages**:\n   - Ask for PDF path if not provided\n   - Ask which pages to process (default: all pages)\n   - Support formats: `1`, `1,3,5`, or `1-5`\n\n2. **Determine recognition parameters**:\n   - Language: Default to `zh-hans,zh-hant,en`\n   - Mode: Precise (default) or fast (`-f`)\n   - Output: Text mode (default) or JSON mode (`-j`)\n\n3. **Execute PDF OCR**:\n```bash\n# All pages\nswift scripts/pdf_ocr.swift \"<pdf_path>\"\n\n# Specific pages\nswift scripts/pdf_ocr.swift \"<pdf_path>\" -p 1,3,5\n\n# Page range\nswift scripts/pdf_ocr.swift \"<pdf_path>\" -p 1-5\n\n# JSON mode\nswift scripts/pdf_ocr.swift \"<pdf_path>\" -p 1 -j\n```\n\n4. **Present results**:\n   - Text mode: Display text by page\n   - JSON mode: Output complete JSON to stdout\n   - Inform user of output format and options\n\n## Output Format\n\nThe script supports two mutually exclusive output modes:\n\n### Text Mode (Default, `-t`)\n\nOutputs only the extracted text.\n\n#### Console Output (without `-o` or `-t <file>`):\n```\n[Extracted text content]\n```\n\n#### File Output (`-o` option):\nCreates `<output>.txt` with pure text.\n\n#### With Confidence Details (console, when not using `-j`):\n```\n=== 置信度详情 ===\n\n总识别块数: 25\n平均置信度: 0.85\n\n--- 逐块详情 ---\n\n[1] Text content\n    置信度: 0.95\n    位置: x=0.10, y=0.20, w=0.30, h=0.05\n\n--- 低置信度警告 (< 0.8) ---\n[3] \"Some text\" - 置信度: 0.50\n```\n\n#### File Output with Confidence (`-o` option):\n- `<output>.txt` - Complete extracted text\n- `<output>_confidence.txt` - Confidence details\n\n### JSON Mode (`-j`)\n\nOutputs complete raw information as JSON to stdout (no file output in JSON mode).\n\n**JSON Structure:**\n```json\n{\n  \"imagePath\": \"/path/to/image.png\",\n  \"totalBlocks\": 25,\n  \"averageConfidence\": 0.85,\n  \"blocks\": [\n    {\n      \"index\": 1,\n      \"text\": \"recognized text\",\n      \"confidence\": 0.95,\n      \"boundingBox\": {\n        \"x\": 0.10,\n        \"y\": 0.20,\n        \"width\": 0.30,\n        \"height\": 0.05\n      }\n    }\n  ]\n}\n```\n\n**Fields:**\n- `imagePath`: Path to the processed image\n- `totalBlocks`: Total number of recognized text blocks\n- `averageConfidence`: Average confidence score (0.0 - 1.0)\n- `blocks`: Array of recognized text blocks\n  - `index`: Block index (1-based)\n  - `text`: Recognized text content\n  - `confidence`: Confidence score (0.0 - 1.0)\n  - `boundingBox`: Normalized bounding box coordinates (0.0 - 1.0)\n\n## Tips and Best Practices\n\n1. **macOS Only**: This skill requires macOS. It will not work on Linux, Windows, or other operating systems.\n2. **Image Quality**: Higher resolution images produce better results\n3. **Language Specification**: Always specify language for better accuracy\n4. **Precise vs Fast**: Use precise mode for final results, fast mode for testing\n5. **Batch Processing**: For multiple images, use shell loop:\n   ```bash\n   for img in *.png; do\n       swift scripts/ocr_vision_pro.swift \"$img\" -o \"${img%.png}.txt\"\n   done\n   ```\n6. **Confidence Threshold**: Results with confidence < 0.5 may need manual verification\n\n## References\n\nFor detailed usage instructions and examples, load `references/usage.md`.\n\n## Resources\n\n### scripts/\n\n- `ocr_vision_pro.swift` - Enhanced OCR script for images with full feature support\n- `ocr_vision.swift` - Basic OCR script for simple use cases\n- `pdf_ocr.swift` - PDF OCR script for extracting text from PDF files\n\n### references/\n\n- `usage.md` - Comprehensive usage guide with examples\n\nFile v1.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn7azmhecfnjv9kg1t4653y21183e0wp\",\n  \"slug\": \"ocr-locally\",\n  \"version\": \"1.0.0\",\n  \"publishedAt\": 1777896845706\n}\n\nFile v1.0.0:references/usage.md\n\n# Local OCR Skill - Detailed Usage Guide\n\n> ⚠️ **macOS Only** - This skill requires macOS 10.15+ (Catalina or later). It will not work on Linux, Windows, or other operating systems.\n\n## Introduction\n\nThis document provides comprehensive usage instructions for the local-ocr skill, which uses macOS native Vision framework to perform OCR (Optical Character Recognition) on images and PDFKit for PDF processing.\n\n## Quick Start\n\n### Basic OCR\n\nTo extract text from an image:\n\n```bash\nswift scripts/ocr_vision_pro.swift /path/to/image.png\n```\n\n### Save Results to File (Separated Output)\n\n```bash\nswift scripts/ocr_vision_pro.swift /path/to/image.png -o result.txt\n```\n\nThis will automatically create two files:\n- `result.txt` - Complete extracted text\n- `result_confidence.txt` - Confidence details\n\n## Output Modes (Mutually Exclusive)\n\nThe script supports two output modes that cannot be used simultaneously:\n\n### Text Mode (Default, `-t`)\n\nOutputs only the extracted text. Optionally saves to file with separate confidence file.\n\n**Console output:**\n```\n[Extracted text content]\n```\n\n**With `-o` option:**\nCreates two files:\n- `result.txt` - Complete extracted text\n- `result_confidence.txt` - Confidence details\n\n### JSON Mode (`-j`)\n\nOutputs complete raw information as JSON to stdout. No file output options in JSON mode.\n\n**JSON output structure:**\n```json\n{\n  \"imagePath\": \"/path/to/image.png\",\n  \"totalBlocks\": 25,\n  \"averageConfidence\": 0.85,\n  \"blocks\": [\n    {\n      \"index\": 1,\n      \"text\": \"recognized text\",\n      \"confidence\": 0.95,\n      \"boundingBox\": {\n        \"x\": 0.10,\n        \"y\": 0.20,\n        \"width\": 0.30,\n        \"height\": 0.05\n      }\n    }\n  ]\n}\n```\n\n**JSON fields:**\n- `imagePath`: Path to the processed image\n- `totalBlocks`: Total number of recognized text blocks\n- `averageConfidence`: Average confidence score (0.0 - 1.0)\n- `blocks`: Array of recognized text blocks\n  - `index`: Block index (1-based)\n  - `text`: Recognized text content\n  - `confidence`: Confidence score (0.0 - 1.0)\n  - `boundingBox`: Normalized bounding box coordinates (0.0 - 1.0)\n    - `x`, `y`: Top-left corner position\n    - `width`, `height`: Bounding box dimensions\n\n## New Output Format (Text Mode)\n\n## Supported Languages\n\nThe OCR script supports the following languages:\n\n| Language Code | Language |\n|---------------|----------|\n| `zh-Hans` | Simplified Chinese |\n| `zh-Hant` | Traditional Chinese |\n| `en` | English |\n| `ja` | Japanese |\n| `ko` | Korean |\n| `fr` | French |\n| `de` | German |\n| `es` | Spanish |\n| `it` | Italian |\n| `pt` | Portuguese |\n| `ru` | Russian |\n\n**Default languages**: `zh-Hans,zh-Hant,en`\n\n## Command-Line Options\n\n### -h, --help\n\nDisplay help information:\n\n```bash\nswift scripts/ocr_vision_pro.swift -h\n```\n\n### -l, --language <languages>\n\nSpecify recognition language (comma-separated):\n\n```bash\n# Chinese and English\nswift scripts/ocr_vision_pro.swift image.png -l zh-Hans,en\n\n# Multiple languages\nswift scripts/ocr_vision_pro.swift image.png -l zh-Hans,en,ja,ko\n```\n\n### -o, --output <file_path>\n\nOutput complete text to file, and automatically create a separate confidence file:\n\n```bash\nswift scripts/ocr_vision_pro.swift image.png -o result.txt\n```\n\nThis creates:\n- `result.txt` - Complete extracted text\n- `result_confidence.txt` - Confidence details\n\n### -t, --text <file_path>\n\nOutput only the complete text to specified file (no confidence file):\n\n```bash\nswift scripts/ocr_vision_pro.swift image.png -t text_only.txt\n```\n\n### -c, --confidence <file_path>\n\nOutput only the confidence details to specified file:\n\n```bash\nswift scripts/ocr_vision_pro.swift image.png -c confidence.txt\n```\n\n### -f, --fast\n\nUse fast mode (lower accuracy, faster processing):\n\n```bash\nswift scripts/ocr_vision_pro.swift image.png -f\n```\n\n**Default**: Precise mode (higher accuracy, slower processing)\n\n### -j, --json\n\nOutput complete raw information as JSON (mutually exclusive with text mode):\n\n```bash\n# JSON mode outputs to stdout\nswift scripts/ocr_vision_pro.swift image.png -j\n\n# Save JSON output to file\nswift scripts/ocr_vision_pro.swift image.png -j > result.json\n```\n\n**Note**: JSON mode outputs to stdout only. No file output options (`-o`, `-t`, `-c`) are used in JSON mode.\n\n**JSON output example:**\n```json\n{\n  \"imagePath\": \"/path/to/image.png\",\n  \"totalBlocks\": 25,\n  \"averageConfidence\": 0.85,\n  \"blocks\": [\n    {\n      \"index\": 1,\n      \"text\": \"recognized text\",\n      \"confidence\": 0.95,\n      \"boundingBox\": {\n        \"x\": 0.10,\n        \"y\": 0.20,\n        \"width\": 0.30,\n        \"height\": 0.05\n      }\n    }\n  ]\n}\n```\n\n## Complete Examples\n\n### Example 1: Console Output with Separated Sections\n\n```bash\nswift scripts/ocr_vision_pro.swift ~/Desktop/screenshot.png -l zh-Hans,en\n```\n\nOutput:\n```\n=== 完整文本 ===\n\n[Extracted text content here]\n\n=== 置信度详情 ===\n\n总识别块数: 25\n平均置信度: 0.85\n\n--- 逐块详情 ---\n\n[1] Text block\n    置信度: 0.95\n    位置: x=0.10, y=0.20, w=0.30, h=0.05\n\n--- 低置信度警告 (< 0.8) ---\n[3] \"Some text\" - 置信度: 0.50\n```\n\n### Example 2: Save to Separate Files\n\n```bash\nswift scripts/ocr_vision_pro.swift ~/Documents/document.jpg -o ~/Documents/extracted\n```\n\nThis creates:\n- `~/Documents/extracted.txt` - Complete text\n- `~/Documents/extracted_confidence.txt` - Confidence details\n\n### Example 3: Separate Text and Confidence Files\n\n```bash\nswift scripts/ocr_vision_pro.swift image.png -t text.txt -c details.txt\n```\n\n### Example 4: Fast Mode for Quick Preview\n\n```bash\nswift scripts/ocr_vision_pro.swift image.png -f\n```\n\n### Example 5: Batch Processing Multiple Images\n\n```bash\nfor img in *.png; do\n    swift scripts/ocr_vision_pro.swift \"$img\" -o \"${img%.png}_ocr\"\ndone\n```\n\nThis creates for each image:\n- `image_ocr.txt` - Complete text\n- `image_ocr_confidence.txt` - Confidence details\n\n### Example 6: JSON Mode Output\n\n```bash\n# Output JSON to console\nswift scripts/ocr_vision_pro.swift image.png -j\n\n# Save JSON to file\nswift scripts/ocr_vision_pro.swift image.png -j > result.json\n\n# Parse JSON with jq\nswift scripts/ocr_vision_pro.swift image.png -j | jq '.blocks[] | select(.confidence < 0.8)'\n```\n\n**Note**: JSON mode and text mode (`-t`) are mutually exclusive. JSON mode outputs to stdout only.\n\n## Output Format\n\nThe script supports two mutually exclusive output modes:\n\n### Text Mode (Default, `-t`)\n\n#### Console Output (without `-o` or `-t <file>`)\n\nOutputs only the extracted text:\n```\n[Extracted text content]\n```\n\nWhen not using `-j`, confidence details are also displayed:\n```\n=== 置信度详情 ===\n\n总识别块数: 25\n平均置信度: 0.85\n\n--- 逐块详情 ---\n\n[1] Text content\n    置信度: 0.95\n    位置: x=0.10, y=0.20, w=0.30, h=0.05\n\n--- 低置信度警告 (< 0.8) ---\n[3] \"Some text\" - 置信度: 0.50\n```\n\n#### File Output (`-o` option)\n\nCreates two separate files:\n- `<output>.txt` - Complete extracted text (pure text, no formatting)\n- `<output>_confidence.txt` - Confidence details\n\n#### File Output (`-t` and `-c` options)\n\nCreates separate files for text and confidence:\n```bash\nswift scripts/ocr_vision_pro.swift image.png -t text.txt -c details.txt\n```\n\n### JSON Mode (`-j`)\n\nOutputs complete raw information as JSON to stdout (no file output in JSON mode).\n\n**JSON Structure:**\n```json\n{\n  \"imagePath\": \"/path/to/image.png\",\n  \"totalBlocks\": 25,\n  \"averageConfidence\": 0.85,\n  \"blocks\": [\n    {\n      \"index\": 1,\n      \"text\": \"recognized text\",\n      \"confidence\": 0.95,\n      \"boundingBox\": {\n        \"x\": 0.10,\n        \"y\": 0.20,\n        \"width\": 0.30,\n        \"height\": 0.05\n      }\n    }\n  ]\n}\n```\n\n**Note**: JSON mode and text mode are mutually exclusive. JSON mode outputs to stdout only.\n\n## Understanding Confidence Scores\n\n| Confidence Range | Interpretation |\n|------------------|----------------|\n| 0.9 - 1.0 | Very high confidence |\n| 0.7 - 0.9 | High confidence |\n| 0.5 - 0.7 | Medium confidence (may need verification) |\n| < 0.5 | Low confidence (likely misrecognized) |\n\n**Note**: \n- Bounding box coordinates are normalized (0.0 - 1.0) relative to image dimensions\n- Low confidence blocks (< 0.8) are highlighted in the warning section\n\n## Supported Image Formats\n\n- **PNG** (.png)\n- **JPEG** (.jpg, .jpeg)\n- **TIFF** (.tiff, .tif)\n- **BMP** (.bmp)\n- **PDF** (.pdf) - first page only\n\n## Troubleshooting\n\n### Issue: \"文件不存在\" (File does not exist)\n\n**Cause**: Incorrect file path\n**Solution**: \n- Use absolute path: `/Users/vector/Desktop/image.png`\n- Or expand tilde: `~/Desktop/image.png`\n- Verify file exists: `ls -la <path>`\n\n### Issue: \"无法加载图片\" (Cannot load image)\n\n**Cause**: Unsupported image format or corrupted file\n**Solution**:\n- Convert to supported format (PNG or JPEG recommended)\n- Verify file is not corrupted\n\n### Issue: Low Recognition Accuracy\n\n**Cause**: Incorrect language setting or low image quality\n**Solution**:\n- Specify correct language: `-l zh-Hans,en`\n- Use precise mode (default, do not use `-f`)\n- Improve image quality (higher resolution, better lighting)\n- Ensure text is not rotated or skewed\n- Check low confidence warnings in output\n\n### Issue: Missing Text in Results\n\n**Cause**: Text may be too small, blurry, or in unsupported language\n**Solution**:\n- Increase image resolution\n- Crop image to focus on text area\n- Verify language is supported and specified correctly\n\n## Performance Considerations\n\n| Image Size | Precise Mode | Fast Mode |\n|------------|--------------|-----------|\n| < 1 MB | ~2-5 seconds | ~1-2 seconds |\n| 1-5 MB | ~5-10 seconds | ~2-5 seconds |\n| > 5 MB | ~10-30 seconds | ~5-10 seconds |\n\n**Tips**:\n- Resize large images before OCR for faster processing\n- Use fast mode (`-f`) for quick preview\n- Process images in batch during off-hours\n\n## System Requirements\n\n> ⚠️ **Platform**: macOS only (not compatible with Linux, Windows, or other OS)\n\n- **Operating System**: macOS 10.15 (Catalina) or later\n- **Framework**: Vision + PDFKit + Core Graphics (pre-installed on macOS)\n- **Swift**: Pre-installed on macOS (no need to install separately)\n- **Disk Space**: Minimal (scripts are < 10 KB each)\n\n## Notes\n\n1. **Privacy**: All processing is done locally on your Mac. No data is sent to the internet.\n2. **Language Download**: Some languages may require downloading language data on first use (system will prompt).\n3. **Handwriting**: The Vision framework is optimized for printed text. Handwriting recognition may have lower accuracy.\n4. **Complex Layouts**: Documents with multiple columns, tables, or unusual layouts may require post-processing to reorder text correctly.\n5. **Separated Output**: The new output format separates pure text from metadata, making it easier to use the extracted text directly.\n\n## References\n\n- [Apple Vision Framework Documentation](https://developer.apple.com/documentation/vision)\n- [VNRecognizeTextRequest](https://developer.apple.com/documentation/vision/vnrecognizetextrequest)\n\n## PDF OCR\n\n> ⚠️ **macOS Only** - PDF OCR also requires macOS 10.15+ with PDFKit framework.\n\nThis skill also supports extracting text from PDF files using the `pdf_ocr.swift` script, which uses PDFKit to render PDF pages to images and then applies Vision framework OCR.\n\n### Quick Start\n\n**Basic usage (all pages):**\n```bash\nswift scripts/pdf_ocr.swift document.pdf\n```\n\n**Specify pages:**\n```bash\n# Single page\nswift scripts/pdf_ocr.swift document.pdf -p 1\n\n# Multiple pages\nswift scripts/pdf_ocr.swift document.pdf -p 1,3,5\n\n# Page range\nswift scripts/pdf_ocr.swift document.pdf -p 1-5\n```\n\n**JSON mode:**\n```bash\nswift scripts/pdf_ocr.swift document.pdf -p 1 -j\n```\n\n### Page Specification Formats\n\nThe `-p` option supports multiple formats:\n- **Single page**: `1`\n- **Multiple pages**: `1,3,5`\n- **Page range**: `1-5`\n- **Combination**: `1,3-5,8-10`\n\n### Output Modes (Mutually Exclusive)\n\n#### Text Mode (Default, `-t`)\n\nOutputs extracted text by page:\n```\n=== 第 1 页 ===\n\n[Extracted text content]\n\n=== 第 2 页 ===\n\n[Extracted text content]\n```\n\n#### JSON Mode (`-j`)\n\nOutputs complete raw information as JSON to stdout:\n```json\n{\n  \"pdfPath\": \"/path/to/document.pdf\",\n  \"totalPages\": 23,\n  \"processedPages\": 2,\n  \"pages\": [\n    {\n      \"pageNumber\": 1,\n      \"blockCount\": 5,\n      \"averageConfidence\": 0.56,\n      \"text\": \"page text content\",\n      \"blocks\": [...]\n    }\n  ]\n}\n```\n\n**PDF JSON Fields:**\n- `pdfPath`: Path to the PDF file\n- `totalPages`: Total number of pages in PDF\n- `processedPages`: Number of pages processed\n- `pages`: Array of page results\n  - `pageNumber`: Page number (1-based)\n  - `blockCount`: Number of text blocks recognized\n  - `averageConfidence`: Average confidence score\n  - `text`: Full text of the page\n  - `blocks`: Array of text blocks with index, text, confidence, boundingBox\n\n### Command-Line Options (PDF OCR)\n\n| Option | Description |\n|--------|-------------|\n| `-h`, `--help` | Show help information |\n| `-t`, `--text` | Text mode (default, output only extracted text) |\n| `-j`, `--json` | JSON mode (output complete raw info as JSON) |\n| `-p`, `--pages <pages>` | Specify pages to process (default: all pages) |\n| `-l`, `--language <lang>` | Specify recognition language (comma-separated) |\n| `-f`, `--fast` | Use fast mode (default: precise mode) |\n\n**Note**: `-t` (text mode) and `-j` (JSON mode) are mutually exclusive. JSON mode outputs to stdout only.\n\n### Supported Languages\n\nSame as image OCR: `zh-hans`, `zh-hant`, `en`, `ja`, `ko`, `fr`, `de`, `es`, `it`, `pt`, `ru`\n\n**Default**: `zh-hans,zh-hant,en`\n\n### Examples\n\n#### Example 1: Extract All Pages\n```bash\nswift scripts/pdf_ocr.swift document.pdf\n```\n\n#### Example 2: Extract Specific Pages\n```bash\nswift scripts/pdf_ocr.swift document.pdf -p 1,3,5\n```\n\n#### Example 3: Extract Page Range\n```bash\nswift scripts/pdf_ocr.swift document.pdf -p 1-5\n```\n\n#### Example 4: JSON Mode Output\n```bash\n# Output JSON to console\nswift scripts/pdf_ocr.swift document.pdf -p 1 -j\n\n# Save JSON to file\nswift scripts/pdf_ocr.swift document.pdf -j > result.json\n```\n\n#### Example 5: Fast Mode for Quick Preview\n```bash\nswift scripts/pdf_ocr.swift document.pdf -p 1 -f\n```\n\n### Troubleshooting (PDF OCR)\n\n#### Issue: \"无法加载 PDF 文件\"\n**Cause**: Incorrect file path or corrupted PDF file\n**Solution**:\n- Use absolute path: `/Users/vector/Documents/document.pdf`\n- Verify file exists: `ls -la <path>`\n- Check if PDF is password-protected\n\n#### Issue: Poor Recognition Accuracy on PDF\n**Cause**: Low resolution PDF or scanned PDF with poor quality\n**Solution**:\n- Use precise mode (default, do not use `-f`)\n- Specify correct language: `-l zh-hans,en`\n- For scanned PDFs, try increasing resolution before OCR\n\n#### Issue: Incorrect Page Numbers\n**Cause**: PDF page numbering may not start from 1\n**Solution**:\n- The script uses actual page indices (1-based)\n- Check PDF page count first: `mdls -name kMDItemNumberOfPages document.pdf`\n\n### System Requirements (PDF OCR)\n\n> ⚠️ **Platform**: macOS only\n\n- **Operating System**: macOS 10.15+ (Catalina or later)\n- **Framework**: PDFKit + Vision (pre-installed on macOS)\n- **Swift**: Pre-installed on macOS\n\n### Notes (PDF OCR)\n\n1. **Platform**: macOS only - requires macOS 10.15+ with PDFKit and Vision frameworks\n2. **Privacy**: All processing is done locally on your Mac. No data is sent to the internet.\n3. **Page Rendering**: PDF pages are rendered to images at their native resolution before OCR.\n4. **Large PDFs**: Processing many pages may take time. Use `-f` (fast mode) for quick preview.\n5. **Password-Protected PDFs**: Not currently supported. Remove password protection before OCR.\n\n## References\n\n- [Apple PDFKit Documentation](https://developer.apple.com/documentation/pdfkit)\n- [Apple Vision Framework Documentation](https://developer.apple.com/documentation/vision)\n\nFile v1.0.0:skill-card.md\n\n## Description:\n\nOCR Locally helps an agent extract text from images and PDFs on macOS using local Vision and PDFKit-based Swift scripts.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[ltryee](https://clawhub.ai/user/ltryee)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and agents use this skill to perform offline OCR on user-selected images, screenshots, scanned documents, and PDFs on macOS. It supports plain-text extraction, optional confidence details, JSON output, language selection, and PDF page selection.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill reads local image or PDF files and may write OCR output files.\n\nMitigation: Use only intended input files, choose explicit output paths, and avoid reusing filenames that contain important content.\n\nRisk: Invalid PDF page ranges may crash or hang an OCR run.\n\nMitigation: Prefer simple validated page ranges such as 1-5 and use the dedicated PDF script for PDF files.\n\nRisk: OCR output can be inaccurate, especially for low-confidence blocks or complex layouts.\n\nMitigation: Review confidence details and manually verify low-confidence or business-critical extracted text.\n\n## Reference(s):\n\n- [Usage Guide](references/usage.md)\n- [Apple Vision Framework Documentation](https://developer.apple.com/documentation/vision)\n- [VNRecognizeTextRequest Documentation](https://developer.apple.com/documentation/vision/vnrecognizetextrequest)\n\n## Skill Output:\n\n**Output Type(s):** [Text, JSON, Files, Shell commands, Guidance]\n\n**Output Format:** [Markdown guidance with inline shell commands; OCR results are plain text, confidence text files, or JSON.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Runs locally on macOS 10.15+ and may write user-specified OCR output files.]\n\n## Skill Version(s):\n\n1.0.0 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.","readmeExcerpt":"Skill: OCR Locally Owner: ltryee Summary: [macOS only] Use this skill when the user requests OCR (Optical Character Recognition), image/PDF text extraction. Uses macOS native Vision/PDFKit frameworks... Tags: latest:1.0.0 Version history: v1.0.0 | 2026-05-04T12:14:05.706Z | user Initial release of local-ocr, providing offline OCR capabilities on macOS. - Supports image and PDF text extraction using macOS native Visio","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"swift scripts/ocr_vision_pro.swift <image_path>"},{"language":"bash","snippet":"swift scripts/ocr_vision_pro.swift <image_path> -l zh-Hans,en -o output.txt -f"},{"language":"bash","snippet":"swift scripts/pdf_ocr.swift <pdf_path>"},{"language":"bash","snippet":"# Single page\nswift scripts/pdf_ocr.swift document.pdf -p 1\n\n# Multiple pages\nswift scripts/pdf_ocr.swift document.pdf -p 1,3,5\n\n# Page range\nswift scripts/pdf_ocr.swift document.pdf -p 1-5\n\n# JSON mode\nswift scripts/pdf_ocr.swift document.pdf -p 1 -j"},{"language":"json","snippet":"{\n  \"imagePath\": \"/path/to/image.png\",\n  \"totalBlocks\": 25,\n  \"averageConfidence\": 0.85,\n  \"blocks\": [\n    {\n      \"index\": 1,\n      \"text\": \"recognized text\",\n      \"confidence\": 0.95,\n      \"boundingBox\": {\n        \"x\": 0.10,\n        \"y\": 0.20,\n        \"width\": 0.30,\n        \"height\": 0.05\n      }\n    }\n  ]\n}"},{"language":"bash","snippet":"swift scripts/ocr_vision_pro.swift \"<image_path>\" -l zh-Hans,en"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: local-ocr\ndescription: \"[macOS only] Use this skill when the user requests OCR (Optical Character Recognition), image/PDF text extraction. Uses macOS native Vision/PDFKit frameworks. Triggers: '识别图片', 'OCR', '提取图片文字', '提取PDF文字', '识别PDF', 'extract text from image', 'PDF OCR'.\"\n---\n\n# Local OCR (macOS Only)\n\n## Overview\n\n⚠️ **Platform Requirement**: This skill is **macOS only**. It requires macOS 10.15+ (Catalina or later) and uses macOS native frameworks:\n\n- **Vision framework** - For OCR text recognition\n- **PDFKit framework** - For PDF processing\n- **Core Graphics** - For image rendering\n\nThis skill provides OCR (Optical Character Recognition) capabilities using macOS native Vision framework. It extracts text from images and PDFs without requiring any third-party libraries or internet connection.\n\n## Platform Requirements\n\n⚠️ **macOS Only** - This skill cannot run on Linux, Windows, or other operating systems.\n\n**Required:**\n- macOS 10.15+ (Catalina or later)\n- Vision framework (pre-installed on macOS)\n- PDFKit framework (pre-installed on macOS)\n\n**Why macOS Only?**\n- Uses `Vision` framework for OCR (macOS/iOS only)\n- Uses `PDFKit` framework for PDF processing (macOS/iOS only)\n- Uses `AppKit`/`Core Graphics` for image handling (macOS only)\n\n## When to Use This Skill\n\nTrigger this skill when the user:\n- Requests OCR or image text extraction\n- Mentions extracting text from images, screenshots, PDF files, or scanned documents\n- Uses keywords like: \"识别图片\", \"OCR\", \"提取文字\", \"提取PDF文字\", \"识别PDF\", \"extract text from image\", \"PDF OCR\"\n- Provides an image file or PDF file and asks to read or extract its content\n\n## Core Capabilities\n\n### 1. Text Extraction from Images\n\nUse `scripts/ocr_vision_pro.swift` for comprehensive OCR with the following features:\n- Multi-language support (Chinese, English, Japanese, Korean, and more)\n- **Two output modes** (mutually exclusive):\n  - **Text Mode** (`-t`): Output only extracted text (default)\n  - **JSON Mode** (`-j`): Output complete raw info including text, position, and confidence as JSON\n- Confidence scores for each detected text block\n- Bounding box information (text position in image)\n- Output to console or file\n- Precise or fast recognition modes\n\n**Basic usage:**\n```bash\nswift scripts/ocr_vision_pro.swift <image_path>\n```\n\n**With options:**\n```bash\nswift scripts/ocr_vision_pro.swift <image_path> -l zh-Hans,en -o output.txt -f\n```\n\n### 2. Text Extraction from PDF Files\n\nUse `scripts/pdf_ocr.swift` to extract text from PDF files with the following features:\n- Extract text from specific pages or all pages\n- Support page range specification (e.g., `1-5`, `1,3,5`)\n- **Two output modes** (mutually exclusive):\n  - **Text Mode** (`-t`): Output only extracted text (default)\n  - **JSON Mode** (`-j`): Output complete raw info as JSON\n- Same multi-language support as image OCR\n- Precise or fast recognition modes\n\n**Basic usage (all pages):**\n```bash\nswift scripts/pdf_ocr.swift <pdf_path>\n```\n\n**With page specificati"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7azmhecfnjv9kg1t4653y21183e0wp\",\n  \"slug\": \"ocr-locally\",\n  \"version\": \"1.0.0\",\n  \"publishedAt\": 1777896845706\n}"},{"path":"references/usage.md","content":"# Local OCR Skill - Detailed Usage Guide\n\n> ⚠️ **macOS Only** - This skill requires macOS 10.15+ (Catalina or later). It will not work on Linux, Windows, or other operating systems.\n\n## Introduction\n\nThis document provides comprehensive usage instructions for the local-ocr skill, which uses macOS native Vision framework to perform OCR (Optical Character Recognition) on images and PDFKit for PDF processing.\n\n## Quick Start\n\n### Basic OCR\n\nTo extract text from an image:\n\n```bash\nswift scripts/ocr_vision_pro.swift /path/to/image.png\n```\n\n### Save Results to File (Separated Output)\n\n```bash\nswift scripts/ocr_vision_pro.swift /path/to/image.png -o result.txt\n```\n\nThis will automatically create two files:\n- `result.txt` - Complete extracted text\n- `result_confidence.txt` - Confidence details\n\n## Output Modes (Mutually Exclusive)\n\nThe script supports two output modes that cannot be used simultaneously:\n\n### Text Mode (Default, `-t`)\n\nOutputs only the extracted text. Optionally saves to file with separate confidence file.\n\n**Console output:**\n```\n[Extracted text content]\n```\n\n**With `-o` option:**\nCreates two files:\n- `result.txt` - Complete extracted text\n- `result_confidence.txt` - Confidence details\n\n### JSON Mode (`-j`)\n\nOutputs complete raw information as JSON to stdout. No file output options in JSON mode.\n\n**JSON output structure:**\n```json\n{\n  \"imagePath\": \"/path/to/image.png\",\n  \"totalBlocks\": 25,\n  \"averageConfidence\": 0.85,\n  \"blocks\": [\n    {\n      \"index\": 1,\n      \"text\": \"recognized text\",\n      \"confidence\": 0.95,\n      \"boundingBox\": {\n        \"x\": 0.10,\n        \"y\": 0.20,\n        \"width\": 0.30,\n        \"height\": 0.05\n      }\n    }\n  ]\n}\n```\n\n**JSON fields:**\n- `imagePath`: Path to the processed image\n- `totalBlocks`: Total number of recognized text blocks\n- `averageConfidence`: Average confidence score (0.0 - 1.0)\n- `blocks`: Array of recognized text blocks\n  - `index`: Block index (1-based)\n  - `text`: Recognized text content\n  - `confidence`: Confidence score (0.0 - 1.0)\n  - `boundingBox`: Normalized bounding box coordinates (0.0 - 1.0)\n    - `x`, `y`: Top-left corner position\n    - `width`, `height`: Bounding box dimensions\n\n## New Output Format (Text Mode)\n\n## Supported Languages\n\nThe OCR script supports the following languages:\n\n| Language Code | Language |\n|---------------|----------|\n| `zh-Hans` | Simplified Chinese |\n| `zh-Hant` | Traditional Chinese |\n| `en` | English |\n| `ja` | Japanese |\n| `ko` | Korean |\n| `fr` | French |\n| `de` | German |\n| `es` | Spanish |\n| `it` | Italian |\n| `pt` | Portuguese |\n| `ru` | Russian |\n\n**Default languages**: `zh-Hans,zh-Hant,en`\n\n## Command-Line Options\n\n### -h, --help\n\nDisplay help information:\n\n```bash\nswift scripts/ocr_vision_pro.swift -h\n```\n\n### -l, --language <languages>\n\nSpecify recognition language (comma-separated):\n\n```bash\n# Chinese and English\nswift scripts/ocr_vision_pro.swift image.png -l zh-Hans,en\n\n# Multiple languages\nswift scripts/ocr_vision_pro.swift image.png -l zh-Hans,en"},{"path":"skill-card.md","content":"## Description:\n\nOCR Locally helps an agent extract text from images and PDFs on macOS using local Vision and PDFKit-based Swift scripts.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[ltryee](https://clawhub.ai/user/ltryee)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and agents use this skill to perform offline OCR on user-selected images, screenshots, scanned documents, and PDFs on macOS. It supports plain-text extraction, optional confidence details, JSON output, language selection, and PDF page selection.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill reads local image or PDF files and may write OCR output files.\n\nMitigation: Use only intended input files, choose explicit output paths, and avoid reusing filenames that contain important content.\n\nRisk: Invalid PDF page ranges may crash or hang an OCR run.\n\nMitigation: Prefer simple validated page ranges such as 1-5 and use the dedicated PDF script for PDF files.\n\nRisk: OCR output can be inaccurate, especially for low-confidence blocks or complex layouts.\n\nMitigation: Review confidence details and manually verify low-confidence or business-critical extracted text.\n\n## Reference(s):\n\n- [Usage Guide](references/usage.md)\n- [Apple Vision Framework Documentation](https://developer.apple.com/documentation/vision)\n- [VNRecognizeTextRequest Documentation](https://developer.apple.com/documentation/vision/vnrecognizetextrequest)\n\n## Skill Output:\n\n**Output Type(s):** [Text, JSON, Files, Shell commands, Guidance]\n\n**Output Format:** [Markdown guidance with inline shell commands; OCR results are plain text, confidence text files, or JSON.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Runs locally on macOS 10.15+ and may write user-specified OCR output files.]\n\n## Skill Version(s):\n\n1.0.0 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":1430,"uniquenessScore":42,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-10T18:41:03.089Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-10T18:41:03.089Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T21:43:10.016Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}