{"id":"ca472997-88ca-4cef-afdf-7c24427b821f","entityType":"agent","slug":"clawhub-openlark-tesseract-image-ocr","name":"Tesseract OCR Image Text Extraction","canonicalUrl":"https://www.xpersona.co/agent/clawhub-openlark-tesseract-image-ocr","canonicalPath":"/agent/clawhub-openlark-tesseract-image-ocr","generatedAt":"2026-10-10T14:45:00.215Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T11:47:45.407Z","emptyReason":null},"description":"Extract text from images using Tesseract.js (OCR). Supports multi-language recognition including Chinese and English, region recognition, character whitelist...","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.5K downloads reported by the source. Last updated 10/10/2026.","installCommand":"clawhub skill install s1727wv2g20pc729snzcm4nf8183hy72:tesseract-image-ocr","sourceUrl":"https://clawhub.ai/openlark/tesseract-image-ocr","homepage":"https://clawhub.ai/openlark/skills/tesseract-image-ocr","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/openlark/tesseract-image-ocr","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/openlark/skills/tesseract-image-ocr","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":63,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Tesseract OCR Image Text Extraction technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-10T11:47:45.407Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T11:47:45.407Z","emptyReason":null},"stars":null,"forks":null,"downloads":1454,"packageName":null,"latestVersion":"1.0.0","tractionLabel":"1.5K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T11:47:45.407Z","emptyReason":null},"lastUpdatedAt":"2026-10-10T11:47:45.407Z","lastCrawledAt":"2026-10-10T11:47:45.407Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-11T11:47:45.407Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.0","createdAt":"2026-05-05T01:42:38.362Z","changelog":"- Initial release of tesseract-image-ocr. - Extract text from images using Tesseract.js with support for over 100 languages, including Chinese and English. - Features include region recognition, character whitelist filtering, text orientation detection, and adjustable output formats (text, hocr, blocks, tsv). - Supports multiple recognition parameters: language selection, page segmentation modes, OCR engine modes, and DPI specification. - Requires Node.js environment and does not support PDF files.","fileCount":5,"zipByteSize":8217}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s1727wv2g20pc729snzcm4nf8183hy72:tesseract-image-ocr","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s1727wv2g20pc729snzcm4nf8183hy72:tesseract-image-ocr` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/openlark/tesseract-image-ocr before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-openlark-tesseract-image-ocr/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-openlark-tesseract-image-ocr/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-openlark-tesseract-image-ocr/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-openlark-tesseract-image-ocr/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-openlark-tesseract-image-ocr/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-openlark-tesseract-image-ocr/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T14:45:00.215Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-openlark-tesseract-image-ocr/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-openlark-tesseract-image-ocr/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-openlark-tesseract-image-ocr/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-openlark-tesseract-image-ocr/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T11:47:45.407Z","emptyReason":null},"readme":"Skill: Tesseract OCR Image Text Extraction\n\nOwner: openlark\n\nSummary: Extract text from images using Tesseract.js (OCR). Supports multi-language recognition including Chinese and English, region recognition, character whitelist...\n\nTags: latest:1.0.0\n\nVersion history:\n\nv1.0.0 | 2026-05-05T01:42:38.362Z | user\n\n- Initial release of tesseract-image-ocr.\n- Extract text from images using Tesseract.js with support for over 100 languages, including Chinese and English.\n- Features include region recognition, character whitelist filtering, text orientation detection, and adjustable output formats (text, hocr, blocks, tsv).\n- Supports multiple recognition parameters: language selection, page segmentation modes, OCR engine modes, and DPI specification.\n- Requires Node.js environment and does not support PDF files.\n\nArchive index:\n\nArchive v1.0.0: 5 files, 8217 bytes\n\nFiles: references/api.md (5756b), scripts/ocr.js (4348b), skill-card.md (2247b), SKILL.md (5922b), _meta.json (138b)\n\nFile v1.0.0:SKILL.md\n\n---\nname: tesseract-image-ocr\ndescription: Extract text from images using Tesseract.js (OCR). Supports multi-language recognition including Chinese and English, region recognition, character whitelist filtering, text orientation detection, and can run in a Node.js environment.\n---\n\n# Tesseract OCR Image Text Extraction\n\nExtract text content from images based on Tesseract.js (the WebAssembly port of the Tesseract OCR engine).\n\n## Use Cases\n\nUse when users need \"image to text,\" \"OCR recognition,\" \"extract text from images,\" \"screenshot character recognition,\" \"scan to text,\" or \"image text orientation detection.\"\n\n## Core Capabilities\n\n- Recognize text from local images or image URLs\n- Support for 100+ languages, with the ability to specify multiple languages simultaneously (e.g., `['eng', 'chi_sim']`)\n- Support for specifying recognition regions (`--rectangle`), character whitelists (`--whitelist`)\n- Support for text orientation and script detection (`--detect`)\n- Support for switching page segmentation modes (`--psm`) and OCR engine modes (`--oem`)\n- Output formats: `text` (default), `hocr`, `blocks` (JSON), `tsv`\n\n## Limitations\n\n- Does not support PDF files\n- Does not modify the Tesseract recognition model to improve accuracy\n- Requires a Node.js environment (this Skill uses Node.js scripts)\n\n---\n\n## Workflow\n\n### 1. Confirm Environment\n\n```shell\nnode -v && npm ls tesseract.js 2>/dev/null || echo \"tesseract.js not installed\"\n```\n\nIf not installed:\n\n```shell\ncd /root/.openclaw/workspace/skills/tesseract-ocr && npm init -y > /dev/null 2>&1 && npm install tesseract.js\n```\n\n### 2. Execute Recognition\n\nBasic command:\n\n```shell\nnode scripts/ocr.js <image-path-or-url> [--options]\n```\n\n**Parameter Descriptions:**\n\n| Parameter | Type | Default | Description |\n|-----------|------|---------|-------------|\n| `<image>` | Required | — | Local path or HTTPS URL |\n| `--lang` | string | `eng` | Language code(s), multiple joined with `+`, e.g., `eng+chi_sim` |\n| `--psm` | number | — | Page segmentation mode (see PSM table below) |\n| `--oem` | number | — | OCR engine mode (see OEM table below) |\n| `--whitelist` | string | — | Character whitelist, e.g., `0123456789` to recognize only digits |\n| `--rectangle` | string | — | Recognition region, format `top,left,width,height` |\n| `--output` | string | `text` | Output format: `text` / `hocr` / `blocks` / `tsv` |\n| `--detect` | flag | — | Detect text orientation and script (does not perform OCR) |\n| `--dpi` | number | — | Manually specify image DPI |\n\n**Common Examples:**\n\n```shell\n# Basic mixed Chinese-English recognition\nnode scripts/ocr.js photo.jpg --lang chi_sim+eng\n\n# Recognize digits only (license plates, CAPTCHAs, etc.)\nnode scripts/ocr.js captcha.png --whitelist 0123456789\n\n# Column-based recognition (suitable for vertical Chinese text)\nnode scripts/ocr.js scroll.jpg --lang chi_sim --psm 4\n\n# Specify a region for recognition\nnode scripts/ocr.js receipt.png --rectangle 50,100,400,200\n\n# Detect image text orientation\nnode scripts/ocr.js rotated.jpg --detect\n\n# Output structured data\nnode scripts/ocr.js doc.png --output blocks\n\n# Manually specify DPI (avoids \"Invalid resolution 0 dpi\" warning)\nnode scripts/ocr.js scan.png --dpi 300\n```\n\n### 3. Batch Recognition of Multiple Images\n\nTo reuse a Worker, the AI should write an inline script:\n\n```javascript\nconst { createWorker } = require('tesseract.js');\n(async () => {\n  const worker = await createWorker('eng');\n  for (const img of ['a.png', 'b.png', 'c.png']) {\n    const { data: { text } } = await worker.recognize(img);\n    console.log(img, '→', text);\n  }\n  await worker.terminate();\n})();\n```\n\n---\n\n## PSM — Page Segmentation Modes\n\nThe `--psm` parameter controls how Tesseract analyzes page layout:\n\n| PSM | Name | Description |\n|-----|------|-------------|\n| 0 | OSD_ONLY | Orientation and script detection only |\n| 1 | AUTO_OSD | Automatic page segmentation + orientation detection |\n| 2 | AUTO_ONLY | Automatic page segmentation, no orientation detection |\n| 3 | AUTO | Fully automatic page segmentation (default) |\n| 4 | SINGLE_COLUMN | Single column of variable size text |\n| 5 | SINGLE_BLOCK_VERT_TEXT | Single block of vertical text |\n| 6 | SINGLE_BLOCK | Single block of text |\n| 7 | SINGLE_LINE | Single line of text |\n| 8 | SINGLE_WORD | Single word |\n| 9 | CIRCLE_WORD | Single word in a circular arrangement |\n| 10 | SINGLE_CHAR | Single character |\n| 11 | SPARSE_TEXT | Sparse text (find as much as possible) |\n| 12 | SPARSE_TEXT_OSD | Sparse text + orientation detection |\n| 13 | RAW_LINE | Raw line (treated as a single line) |\n\n## OEM — OCR Engine Modes\n\n| OEM | Description |\n|-----|-------------|\n| 0 | Legacy engine |\n| 1 | LSTM neural network engine (default) |\n| 2 | Legacy + LSTM |\n| 3 | Default (automatically selected based on current configuration) |\n\n## Language Code Quick Reference\n\n| Language | Code |\n|----------|------|\n| English | `eng` |\n| Simplified Chinese | `chi_sim` |\n| Traditional Chinese | `chi_tra` |\n| Japanese | `jpn` |\n| Korean | `kor` |\n| French | `fra` |\n| German | `deu` |\n| Spanish | `spa` |\n| Russian | `rus` |\n| Arabic | `ara` |\n| Hindi | `hin` |\n\nFull list: [tesseract_lang_list.md](https://github.com/naptha/tesseract.js/blob/master/docs/tesseract_lang_list.md)\n\n## Advanced Usage\n\nThe following scenarios require the AI to write inline scripts directly rather than using `scripts/ocr.js`:\n\n- **Reusing a Worker after switching languages**: Use `worker.reinitialize(langs, oem)`\n- **Setting Tesseract parameters**: Use `worker.setParameters({ tessedit_pageseg_mode: ... })`\n- **Detecting text orientation** (requires Legacy engine): Call `worker.detect(image)` after `createWorker('eng', 0, { legacyCore: true, legacyLang: true })`\n- **Processing large numbers of images in parallel**: Use `createScheduler()` + multiple Workers\n\nFor complete API reference, see [references/api.md](references/api.md).\n\nFile v1.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn75qrtb885pznwsjwsh18dvf1813bv0\",\n  \"slug\": \"tesseract-image-ocr\",\n  \"version\": \"1.0.0\",\n  \"publishedAt\": 1777945358362\n}\n\nFile v1.0.0:references/api.md\n\n# Tesseract.js Full API Reference\n\n> Based on the [Tesseract.js Official API Documentation](https://github.com/naptha/tesseract.js/blob/master/docs/api.md)\n\n---\n\n## createWorker(langs, oem, options): Worker\n\nCreates a Tesseract.js Worker instance. A Worker manages a single Tesseract instance within a Web Worker (browser) or Worker Thread (Node.js).\n\n**Parameters:**\n\n| Parameter | Type | Description |\n|-----------|------|-------------|\n| `langs` | string\\|string[] | Language code(s), e.g., `'eng'` or `['eng', 'chi_sim']` |\n| `oem` | number | OCR engine mode (see OEM table) |\n| `options` | object | Custom options |\n\n**options Fields:**\n\n| Field | Type | Default | Description |\n|-------|------|---------|-------------|\n| `corePath` | string | — | Path to the tesseract.js-core directory (**Note: points to a directory, not a single .js file**) |\n| `langPath` | string | — | Download path for training language data (no trailing `/`) |\n| `workerPath` | string | — | Download path for worker script |\n| `cachePath` | string | — | Cache path; more commonly used in Node.js |\n| `cacheMethod` | string | `'write'` | Cache strategy: `write`/`readOnly`/`refresh`/`none` |\n| `legacyCore` | boolean | `false` | Whether to download Legacy model support code |\n| `legacyLang` | boolean | `false` | Whether to download Legacy language data |\n| `workerBlobURL` | boolean | `true` | Whether to load the Worker using a Blob URL |\n| `gzip` | boolean | `true` | Whether remote training data is gzip compressed |\n| `logger` | function | — | Progress callback, e.g., `m => console.log(m)` |\n| `errorHandler` | function | — | Worker error callback |\n\n---\n\n## worker.recognize(image, options, output, jobId): Promise\n\nCore OCR recognition method.\n\n**Parameters:**\n\n| Parameter | Type | Description |\n|-----------|------|-------------|\n| `image` | string\\|Buffer\\|... | Image (see supported image formats) |\n| `options` | object | Recognition options |\n| `options.rectangle` | object | Restrict recognition region `{top, left, width, height}` |\n| `output` | object | Output format toggles, e.g., `{ hocr: true, blocks: true }` |\n| `jobId` | string | Optional job ID |\n\n**Returns:** `{ jobId, data: { text, hocr?, blocks?, tsv?, ... } }`\n\nThe returned `data.text` is the plain text result. An empty result is returned even if no text is detected; an exception will not be thrown.\n\n---\n\n## worker.setParameters(params, jobId): Promise\n\nSets Tesseract engine parameters (calls SetVariable).\n\n**Common Parameters:**\n\n| Parameter Name | Type | Default | Description |\n|----------------|------|---------|-------------|\n| `tessedit_pageseg_mode` | string | `'3'` | Page segmentation mode |\n| `tessedit_char_whitelist` | string | `''` | Character whitelist |\n| `preserve_interword_spaces` | string | `'0'` | Preserve inter-word spaces |\n| `user_defined_dpi` | string | `''` | Manually specify DPI |\n\n> `setParameters` cannot modify `oem`; to change the OEM, use `worker.reinitialize()`.\n\n---\n\n## worker.reinitialize(langs, oem, config, jobId): Promise\n\nReinitialize an existing Worker with different languages/OEM.\n\n```javascript\nawait worker.reinitialize('chi_sim', 1);\n```\n\nTo switch to the Legacy engine (OEM=0), the Worker must have been created with `legacyCore: true, legacyLang: true` set in `createWorker`.\n\n---\n\n## worker.detect(image, jobId): Promise\n\nPerforms OSD (Orientation and Script Detection) without performing OCR.\n\n**Prerequisite:** The Worker must be created using `legacyCore: true, legacyLang: true`.\n\n```javascript\nconst worker = await createWorker('eng', 0, { legacyCore: true, legacyLang: true });\nconst { data } = await worker.detect('rotated.jpg');\n// data: { orientation_confidence, orientation_degrees, script, script_confidence }\n```\n\n---\n\n## worker.terminate(jobId): Promise\n\nTerminate the Worker and clean up resources.\n\n```javascript\nawait worker.terminate();\n```\n\n---\n\n## createScheduler(): Scheduler\n\nCreate a scheduler for parallel processing.\n\n### Scheduler Methods\n\n| Method | Description |\n|--------|-------------|\n| `scheduler.addWorker(worker)` | Add a Worker to the scheduling pool |\n| `scheduler.addJob({ recognize(image, options, output) })` | Add a recognition job |\n| `scheduler.getQueueLen()` | Get the current queue length |\n| `scheduler.getNumWorkers()` | Get the number of Workers |\n\n```javascript\nconst scheduler = createScheduler();\nconst w1 = await createWorker('eng');\nconst w2 = await createWorker('eng');\nscheduler.addWorker(w1);\nscheduler.addWorker(w2);\n\nconst results = await Promise.all(images.map(img =>\n  scheduler.addJob('recognize', img)\n));\nawait Promise.all([w1.terminate(), w2.terminate()]);\n```\n\n---\n\n## setLogging(flag: boolean)\n\nEnable/disable log output.\n\n---\n\n## PSM — Page Segmentation Modes\n\n| Value | Name | Description |\n|-------|------|-------------|\n| 0 | OSD_ONLY | Orientation and script detection only |\n| 1 | AUTO_OSD | Automatic page segmentation + orientation detection |\n| 2 | AUTO_ONLY | Automatic page segmentation, no orientation detection |\n| 3 | AUTO | Fully automatic page segmentation (default) |\n| 4 | SINGLE_COLUMN | Single column of variable size text |\n| 5 | SINGLE_BLOCK_VERT_TEXT | Single block of vertical text |\n| 6 | SINGLE_BLOCK | Single block of text |\n| 7 | SINGLE_LINE | Single line of text |\n| 8 | SINGLE_WORD | Single word |\n| 9 | CIRCLE_WORD | Single word in a circular arrangement |\n| 10 | SINGLE_CHAR | Single character |\n| 11 | SPARSE_TEXT | Sparse text |\n| 12 | SPARSE_TEXT_OSD | Sparse text + orientation detection |\n| 13 | RAW_LINE | Raw line |\n\n## OEM — OCR Engine Modes\n\n| Value | Description |\n|-------|-------------|\n| 0 | Legacy engine |\n| 1 | LSTM neural network engine (default) |\n| 2 | Legacy + LSTM |\n| 3 | Default (selected automatically) |\n\nFile v1.0.0:skill-card.md\n\n## Description:\n\nExtract text from images using Tesseract.js (OCR). Supports multi-language recognition including Chinese and English, region recognition, character whitelist filtering, text orientation detection, and can run in a Node.js environment.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[openlark](https://clawhub.ai/user/openlark)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and agents use this skill to extract text from local or trusted remote images, including screenshots and scanned images, with language, region, whitelist, orientation, and output-format controls.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: npm dependency downloads may introduce unreviewed package or supply-chain changes.\n\nMitigation: Install only in environments where dependency downloads are allowed, and pin tesseract.js with a reviewed lockfile before operational use.\n\nRisk: Remote image URLs can expose private content to untrusted hosts or fetch content from untrusted locations.\n\nMitigation: Prefer local image files for private or regulated data, and use remote URLs only when the host is trusted.\n\n## Reference(s):\n\n- [Tesseract.js API Reference](references/api.md)\n- [Tesseract.js Official API Documentation](https://github.com/naptha/tesseract.js/blob/master/docs/api.md)\n- [Tesseract.js Language List](https://github.com/naptha/tesseract.js/blob/master/docs/tesseract_lang_list.md)\n\n## Skill Output:\n\n**Output Type(s):** [text, code, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown guidance with shell and JavaScript examples; OCR execution can return plain text, JSON, hOCR, or TSV.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Supports language selection, region selection, character whitelists, page segmentation modes, OCR engine modes, DPI overrides, and orientation detection.]\n\n## Skill Version(s):\n\n1.0.0 (source: server release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.","readmeExcerpt":"Skill: Tesseract OCR Image Text Extraction Owner: openlark Summary: Extract text from images using Tesseract.js (OCR). Supports multi-language recognition including Chinese and English, region recognition, character whitelist... Tags: latest:1.0.0 Version history: v1.0.0 | 2026-05-05T01:42:38.362Z | user - Initial release of tesseract-image-ocr. - Extract text from images using Tesseract.js with support for over 100 ","codeSnippets":[],"executableExamples":[{"language":"shell","snippet":"node -v && npm ls tesseract.js 2>/dev/null || echo \"tesseract.js not installed\""},{"language":"shell","snippet":"cd /root/.openclaw/workspace/skills/tesseract-ocr && npm init -y > /dev/null 2>&1 && npm install tesseract.js"},{"language":"shell","snippet":"node scripts/ocr.js <image-path-or-url> [--options]"},{"language":"shell","snippet":"# Basic mixed Chinese-English recognition\nnode scripts/ocr.js photo.jpg --lang chi_sim+eng\n\n# Recognize digits only (license plates, CAPTCHAs, etc.)\nnode scripts/ocr.js captcha.png --whitelist 0123456789\n\n# Column-based recognition (suitable for vertical Chinese text)\nnode scripts/ocr.js scroll.jpg --lang chi_sim --psm 4\n\n# Specify a region for recognition\nnode scripts/ocr.js receipt.png --rectangle 50,100,400,200\n\n# Detect image text orientation\nnode scripts/ocr.js rotated.jpg --detect\n\n# Output structured data\nnode scripts/ocr.js doc.png --output blocks\n\n# Manually specify DPI (avoids \"Invalid resolution 0 dpi\" warning)\nnode scripts/ocr.js scan.png --dpi 300"},{"language":"javascript","snippet":"const { createWorker } = require('tesseract.js');\n(async () => {\n  const worker = await createWorker('eng');\n  for (const img of ['a.png', 'b.png', 'c.png']) {\n    const { data: { text } } = await worker.recognize(img);\n    console.log(img, '→', text);\n  }\n  await worker.terminate();\n})();"},{"language":"javascript","snippet":"await worker.reinitialize('chi_sim', 1);"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: tesseract-image-ocr\ndescription: Extract text from images using Tesseract.js (OCR). Supports multi-language recognition including Chinese and English, region recognition, character whitelist filtering, text orientation detection, and can run in a Node.js environment.\n---\n\n# Tesseract OCR Image Text Extraction\n\nExtract text content from images based on Tesseract.js (the WebAssembly port of the Tesseract OCR engine).\n\n## Use Cases\n\nUse when users need \"image to text,\" \"OCR recognition,\" \"extract text from images,\" \"screenshot character recognition,\" \"scan to text,\" or \"image text orientation detection.\"\n\n## Core Capabilities\n\n- Recognize text from local images or image URLs\n- Support for 100+ languages, with the ability to specify multiple languages simultaneously (e.g., `['eng', 'chi_sim']`)\n- Support for specifying recognition regions (`--rectangle`), character whitelists (`--whitelist`)\n- Support for text orientation and script detection (`--detect`)\n- Support for switching page segmentation modes (`--psm`) and OCR engine modes (`--oem`)\n- Output formats: `text` (default), `hocr`, `blocks` (JSON), `tsv`\n\n## Limitations\n\n- Does not support PDF files\n- Does not modify the Tesseract recognition model to improve accuracy\n- Requires a Node.js environment (this Skill uses Node.js scripts)\n\n---\n\n## Workflow\n\n### 1. Confirm Environment\n\n```shell\nnode -v && npm ls tesseract.js 2>/dev/null || echo \"tesseract.js not installed\"\n```\n\nIf not installed:\n\n```shell\ncd /root/.openclaw/workspace/skills/tesseract-ocr && npm init -y > /dev/null 2>&1 && npm install tesseract.js\n```\n\n### 2. Execute Recognition\n\nBasic command:\n\n```shell\nnode scripts/ocr.js <image-path-or-url> [--options]\n```\n\n**Parameter Descriptions:**\n\n| Parameter | Type | Default | Description |\n|-----------|------|---------|-------------|\n| `<image>` | Required | — | Local path or HTTPS URL |\n| `--lang` | string | `eng` | Language code(s), multiple joined with `+`, e.g., `eng+chi_sim` |\n| `--psm` | number | — | Page segmentation mode (see PSM table below) |\n| `--oem` | number | — | OCR engine mode (see OEM table below) |\n| `--whitelist` | string | — | Character whitelist, e.g., `0123456789` to recognize only digits |\n| `--rectangle` | string | — | Recognition region, format `top,left,width,height` |\n| `--output` | string | `text` | Output format: `text` / `hocr` / `blocks` / `tsv` |\n| `--detect` | flag | — | Detect text orientation and script (does not perform OCR) |\n| `--dpi` | number | — | Manually specify image DPI |\n\n**Common Examples:**\n\n```shell\n# Basic mixed Chinese-English recognition\nnode scripts/ocr.js photo.jpg --lang chi_sim+eng\n\n# Recognize digits only (license plates, CAPTCHAs, etc.)\nnode scripts/ocr.js captcha.png --whitelist 0123456789\n\n# Column-based recognition (suitable for vertical Chinese text)\nnode scripts/ocr.js scroll.jpg --lang chi_sim --psm 4\n\n# Specify a region for recognition\nnode scripts/ocr.js receipt.png --rectangle 50,100,400,200\n\n# Detect image text orient"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn75qrtb885pznwsjwsh18dvf1813bv0\",\n  \"slug\": \"tesseract-image-ocr\",\n  \"version\": \"1.0.0\",\n  \"publishedAt\": 1777945358362\n}"},{"path":"references/api.md","content":"# Tesseract.js Full API Reference\n\n> Based on the [Tesseract.js Official API Documentation](https://github.com/naptha/tesseract.js/blob/master/docs/api.md)\n\n---\n\n## createWorker(langs, oem, options): Worker\n\nCreates a Tesseract.js Worker instance. A Worker manages a single Tesseract instance within a Web Worker (browser) or Worker Thread (Node.js).\n\n**Parameters:**\n\n| Parameter | Type | Description |\n|-----------|------|-------------|\n| `langs` | string\\|string[] | Language code(s), e.g., `'eng'` or `['eng', 'chi_sim']` |\n| `oem` | number | OCR engine mode (see OEM table) |\n| `options` | object | Custom options |\n\n**options Fields:**\n\n| Field | Type | Default | Description |\n|-------|------|---------|-------------|\n| `corePath` | string | — | Path to the tesseract.js-core directory (**Note: points to a directory, not a single .js file**) |\n| `langPath` | string | — | Download path for training language data (no trailing `/`) |\n| `workerPath` | string | — | Download path for worker script |\n| `cachePath` | string | — | Cache path; more commonly used in Node.js |\n| `cacheMethod` | string | `'write'` | Cache strategy: `write`/`readOnly`/`refresh`/`none` |\n| `legacyCore` | boolean | `false` | Whether to download Legacy model support code |\n| `legacyLang` | boolean | `false` | Whether to download Legacy language data |\n| `workerBlobURL` | boolean | `true` | Whether to load the Worker using a Blob URL |\n| `gzip` | boolean | `true` | Whether remote training data is gzip compressed |\n| `logger` | function | — | Progress callback, e.g., `m => console.log(m)` |\n| `errorHandler` | function | — | Worker error callback |\n\n---\n\n## worker.recognize(image, options, output, jobId): Promise\n\nCore OCR recognition method.\n\n**Parameters:**\n\n| Parameter | Type | Description |\n|-----------|------|-------------|\n| `image` | string\\|Buffer\\|... | Image (see supported image formats) |\n| `options` | object | Recognition options |\n| `options.rectangle` | object | Restrict recognition region `{top, left, width, height}` |\n| `output` | object | Output format toggles, e.g., `{ hocr: true, blocks: true }` |\n| `jobId` | string | Optional job ID |\n\n**Returns:** `{ jobId, data: { text, hocr?, blocks?, tsv?, ... } }`\n\nThe returned `data.text` is the plain text result. An empty result is returned even if no text is detected; an exception will not be thrown.\n\n---\n\n## worker.setParameters(params, jobId): Promise\n\nSets Tesseract engine parameters (calls SetVariable).\n\n**Common Parameters:**\n\n| Parameter Name | Type | Default | Description |\n|----------------|------|---------|-------------|\n| `tessedit_pageseg_mode` | string | `'3'` | Page segmentation mode |\n| `tessedit_char_whitelist` | string | `''` | Character whitelist |\n| `preserve_interword_spaces` | string | `'0'` | Preserve inter-word spaces |\n| `user_defined_dpi` | string | `''` | Manually specify DPI |\n\n> `setParameters` cannot modify `oem`; to change the OEM, use `worker.reinitialize()`.\n\n---\n\n## worker.reinitialize(langs, o"},{"path":"skill-card.md","content":"## Description:\n\nExtract text from images using Tesseract.js (OCR). Supports multi-language recognition including Chinese and English, region recognition, character whitelist filtering, text orientation detection, and can run in a Node.js environment.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[openlark](https://clawhub.ai/user/openlark)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and agents use this skill to extract text from local or trusted remote images, including screenshots and scanned images, with language, region, whitelist, orientation, and output-format controls.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: npm dependency downloads may introduce unreviewed package or supply-chain changes.\n\nMitigation: Install only in environments where dependency downloads are allowed, and pin tesseract.js with a reviewed lockfile before operational use.\n\nRisk: Remote image URLs can expose private content to untrusted hosts or fetch content from untrusted locations.\n\nMitigation: Prefer local image files for private or regulated data, and use remote URLs only when the host is trusted.\n\n## Reference(s):\n\n- [Tesseract.js API Reference](references/api.md)\n- [Tesseract.js Official API Documentation](https://github.com/naptha/tesseract.js/blob/master/docs/api.md)\n- [Tesseract.js Language List](https://github.com/naptha/tesseract.js/blob/master/docs/tesseract_lang_list.md)\n\n## Skill Output:\n\n**Output Type(s):** [text, code, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown guidance with shell and JavaScript examples; OCR execution can return plain text, JSON, hOCR, or TSV.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Supports language selection, region selection, character whitelists, page segmentation modes, OCR engine modes, DPI overrides, and orientation detection.]\n\n## Skill Version(s):\n\n1.0.0 (source: server release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":1412,"uniquenessScore":43,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-10T11:47:45.407Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-10T11:47:45.407Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T14:45:00.215Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}