{"id":"68a6e65f-4fbe-4a66-9dfb-6b9ee75f7414","entityType":"agent","slug":"clawhub-compdf-youna-pdf-to-word-docx","name":"PDF to Word Converter","canonicalUrl":"https://www.xpersona.co/agent/clawhub-compdf-youna-pdf-to-word-docx","canonicalPath":"/agent/clawhub-compdf-youna-pdf-to-word-docx","generatedAt":"2026-10-10T07:42:11.847Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T23:47:57.848Z","emptyReason":null},"description":"PDF to Word converts PDF to editable Word/DOCX with AI-powered layout analysis and table recognition, built on ComPDF Conversion SDK to better preserve table...","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.9K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s178t897hzzp9pvkwbav4e2kfs83g980:pdf-to-word-docx","sourceUrl":"https://clawhub.ai/compdf-youna/pdf-to-word-docx","homepage":"https://clawhub.ai/compdf-youna/skills/pdf-to-word-docx","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/compdf-youna/pdf-to-word-docx","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/compdf-youna/skills/pdf-to-word-docx","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":65,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"PDF to Word Converter technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T23:47:57.848Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T23:47:57.848Z","emptyReason":null},"stars":null,"forks":null,"downloads":1870,"packageName":null,"latestVersion":"1.2.0","tractionLabel":"1.9K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T23:47:57.848Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T23:47:57.848Z","lastCrawledAt":"2026-10-09T23:47:57.848Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T23:47:57.848Z","lastVerifiedAt":null,"highlights":[{"version":"1.2.0","createdAt":"2026-06-24T03:46:44.581Z","changelog":"pdf-to-word-docx 1.2.0 changelog - Updated homepage URL to include UTM parameters for improved source tracking. - Removed the redundant skill-card.md file for a cleaner project structure.","fileCount":5,"zipByteSize":16604},{"version":"1.1.0","createdAt":"2026-04-29T12:02:01.232Z","changelog":"**PDF to Word Converter 1.1.0** - Improved skill description with more specific use cases and enhanced clarity for typical PDF to Word and document layout preservation scenarios. - Updated summary emphasizes AI-powered layout analysis, robust table, multi-column, and image handling, and real-world editing use cases. - No code or script changes were made; documentation was updated for clearer discoverability and guidance. - All conversion features and usage patterns remain compatible and unchanged.","fileCount":5,"zipByteSize":16217},{"version":"1.0.0","createdAt":"2026-03-19T11:06:23.211Z","changelog":"pdf to word (pdf-to-word-docx) — A reusable Agent Skill wrapping the ComPDFKit Conversion SDK for local PDF and image format conversion with AI layout analysis and OCR support. Supported Conversions - Input: PDF, JPG/JPEG, PNG, BMP, TIFF/TIF, WEBP, JPEG2000 - Output: Word, Excel, PPT, HTML, Image, TXT, JSON, Markdown, RTF, CSV (10 formats) Features - AI-powered layout analysis for accurate document structure preservation - OCR support for scanned documents and images - Cross-platform support: Windows and macOS","fileCount":4,"zipByteSize":14395}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s178t897hzzp9pvkwbav4e2kfs83g980:pdf-to-word-docx","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s178t897hzzp9pvkwbav4e2kfs83g980:pdf-to-word-docx` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/compdf-youna/pdf-to-word-docx before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-compdf-youna-pdf-to-word-docx/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-compdf-youna-pdf-to-word-docx/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-compdf-youna-pdf-to-word-docx/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-compdf-youna-pdf-to-word-docx/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-compdf-youna-pdf-to-word-docx/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-compdf-youna-pdf-to-word-docx/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T07:42:11.846Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-compdf-youna-pdf-to-word-docx/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-compdf-youna-pdf-to-word-docx/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-compdf-youna-pdf-to-word-docx/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-compdf-youna-pdf-to-word-docx/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T23:47:57.848Z","emptyReason":null},"readme":"Skill: PDF to Word Converter\n\nOwner: compdf-youna\n\nSummary: PDF to Word converts PDF to editable Word/DOCX with AI-powered layout analysis and table recognition, built on ComPDF Conversion SDK to better preserve table...\n\nTags: latest:1.2.0\n\nVersion history:\n\nv1.2.0 | 2026-06-24T03:46:44.581Z | user\n\npdf-to-word-docx 1.2.0 changelog\n\n- Updated homepage URL to include UTM parameters for improved source tracking.\n- Removed the redundant skill-card.md file for a cleaner project structure.\n\nv1.1.0 | 2026-04-29T12:02:01.232Z | user\n\n**PDF to Word Converter 1.1.0**\n\n- Improved skill description with more specific use cases and enhanced clarity for typical PDF to Word and document layout preservation scenarios.\n- Updated summary emphasizes AI-powered layout analysis, robust table, multi-column, and image handling, and real-world editing use cases.\n- No code or script changes were made; documentation was updated for clearer discoverability and guidance.\n- All conversion features and usage patterns remain compatible and unchanged.\n\nv1.0.0 | 2026-03-19T11:06:23.211Z | user\n\npdf to word (pdf-to-word-docx) — A reusable Agent Skill wrapping the ComPDFKit Conversion SDK for local PDF and image format conversion with AI layout analysis and OCR support.\n\nSupported Conversions\n- Input: PDF, JPG/JPEG, PNG, BMP, TIFF/TIF, WEBP, JPEG2000\n- Output: Word, Excel, PPT, HTML, Image, TXT, JSON, Markdown, RTF, CSV (10 formats)\n\nFeatures\n- AI-powered layout analysis for accurate document structure preservation\n- OCR support for scanned documents and images\n- Cross-platform support: Windows and macOS\n\nArchive index:\n\nArchive v1.2.0: 5 files, 16604 bytes\n\nFiles: License.txt (3060b), scripts/pdf-to-word-docx.py (20921b), skill-card.md (3363b), SKILL.md (24546b), _meta.json (135b)\n\nFile v1.2.0:SKILL.md\n\n---\nname: pdf-to-word-docx\nversion: 1.2.0\ndescription: PDF to Word converts PDF to editable Word/DOCX with AI-powered layout analysis and table recognition, built on ComPDF Conversion SDK to better preserve tables, multi-column layouts, lists, and images for downstream editing. It fits requests such as “pdf to word,” “convert pdf to docx,” “pdf to editable word,” “pdf to office,” “keep layout in word,” and “convert report to docx.” Example queries include “Convert this PDF contract to editable Word while keeping the tables intact,” “Turn this report into DOCX and preserve the multi-column layout,” and “Export this PDF to Word for further editing.”\nhomepage: https://www.compdf.com/?utm_source=clawhub&utm_medium=skillhub&utm_campaign=compdf_pdf_skill_to_word&ref_platform_id=clawhub_skills\nmetadata:\n  clawdbot:\n    emoji: \"📑\"\n    requires:\n      env: []\n    files: [\"scripts/*\"]\ncompatibility: Requires Windows or macOS. Python with ComPDFKitConversion package (pip install ComPDFKitConversion). AI model (~525MB) auto-downloaded on first run.\n---\n\n# PDF to Word Converter\n\n## Purpose\n- Wraps the `ComPDFKitConversion` Python SDK into a reusable local conversion workflow, supporting PDF / image to Word, PPT, Excel, HTML, RTF, Image, TXT, JSON, Markdown, and CSV (10 output formats in total).\n\n## Agent Skills Standard Compatibility\n- This Skill uses an Anthropic Agent Skills-compatible directory structure: `pdf-to-word-docx/`.\n- The entry point is `SKILL.md`; helper scripts are placed in `scripts/`.\n- The document uses `$ARGUMENTS` and `${CLAUDE_SKILL_DIR}` conventions for distribution and execution in Claude Code / Agent Skills-compatible environments.\n\n## Input / Output\n- Input: The target format (`word`/`excel`/`ppt`/`html`/`rtf`/`image`/`txt`/`json`/`markdown`/`csv`), the PDF or image path, and the output path are passed via Skill arguments or the command line. An optional PDF password and conversion parameters may also be provided.\n- Supported input file types:\n  - PDF files (`.pdf`)\n  - Image files (`.jpg`/`.jpeg`/`.png`/`.bmp`/`.tif`/`.tiff`/`.webp`/`.jp2`/`.gif`/`.tga`)\n- Output: A file in the corresponding format (`.docx`, `.pptx`, `.xlsx`, `.html`, `.rtf`, image, `.txt`, `.json`, `.md`, `.csv`), or a clear error message.\n\n## Prerequisites\n- Supports Windows and macOS.\n- The conversion SDK must be installed first:\n  ```bash\n  pip install ComPDFKitConversion\n  ```\n- On first run, the script automatically downloads `license.xml` from the ComPDF server and caches it in the `scripts/` directory:\n  ```text\n  https://download.compdf.com/skills/license/license.xml\n  ```\n- The script reads the `<key>...</key>` field from `license.xml` and uses that key for `LibraryManager.license_verify(...)` authentication — it does not pass the XML file path directly to the SDK.\n- To use a custom license, place your own `license.xml` in the `scripts/` directory; the script will use it directly without downloading.\n- During SDK initialization, the `resource` directory is always set to the directory containing `pdf-to-word-docx.py`, i.e., the `scripts/` directory itself.\n- When `--enable-ocr` or `--enable-ai-layout` (enabled by default) is used, the Skill also requires `scripts/documentai.model`. If the file does not exist, the script will automatically download it from:\n  ```text\n  https://download.compdf.com/skills/model/documentai.model\n  ```\n- To reuse an existing model file, you can override the default model path via an environment variable:\n  ```bash\n  export COMPDF_DOCUMENT_AI_MODEL=\"/path/to/documentai.model\"\n  ```\n\n## Workflow\n1. Confirm the Python package is installed:\n   ```bash\n   python -m pip show ComPDFKitConversion\n   ```\n2. The script automatically downloads `license.xml` on first run; the `scripts/` directory is used directly as the SDK `resource` path.\n3. In Agent Skills / Claude Code environments, prefer using the Skill's built-in script path variable:\n   ```bash\n   python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.pdf output.docx\n   python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" ppt input.pdf output.pptx\n   python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" excel input.pdf output.xlsx\n   ```\n4. For more control, append common parameters:\n   ```bash\n   python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" excel input.pdf output.xlsx --page-ranges \"1-3,5\" --excel-all-content --excel-worksheet-option for-page\n   python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.pdf output.docx --enable-ocr --page-layout-mode flow\n   ```\n5. On startup, the script ensures `scripts/license.xml` exists (downloading it automatically from the ComPDF server if missing), reads the `<key>` field for SDK authentication, and uses the `scripts/` directory as the `resource` path.\n6. If `--enable-ocr` or `--enable-ai-layout` (enabled by default) is active, the script checks whether `scripts/documentai.model` exists; if not, it downloads the file automatically before initializing the Document AI model.\n7. Check the return code; if it is not `SUCCESS`, handle license, password, resource, model, or input file issues according to the error name.\n\n## documentai.model Download Optimization\n- The script preferentially uses the model file pointed to by `COMPDF_DOCUMENT_AI_MODEL`.\n- The default model path is `scripts/documentai.model`.\n- During automatic download, the file is first written to `documentai.model.part` and then atomically renamed to the final file upon success, preventing partial file corruption.\n- On download failure, the script retries automatically with back-off intervals of `2s / 5s / 10s`.\n\n## Invoking Directly as a Skill\n- In environments that support Agent Skills, the Skill can be called directly:\n  ```text\n  /pdf-to-word-docx word input.pdf output.docx\n  /pdf-to-word-docx excel input.pdf output.xlsx --excel-worksheet-option for-page\n  ```\n- When the Skill receives arguments, it passes them through to the script as-is:\n  ```bash\n  python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" $ARGUMENTS\n  ```\n- If the environment does not support direct Skill invocation, fall back to a regular command-line call.\n\n## Supported Output Formats\n- `word` → calls `CPDFConversion.start_pdf_to_word`\n- `excel` → calls `CPDFConversion.start_pdf_to_excel`\n- `ppt` → calls `CPDFConversion.start_pdf_to_ppt`\n- `html` → calls `CPDFConversion.start_pdf_to_html`\n- `rtf` → calls `CPDFConversion.start_pdf_to_rtf`\n- `image` → calls `CPDFConversion.start_pdf_to_image`\n- `txt` → calls `CPDFConversion.start_pdf_to_txt`\n- `json` → calls `CPDFConversion.start_pdf_to_json`\n- `markdown` → calls `CPDFConversion.start_pdf_to_markdown`\n- `csv` → reuses `CPDFConversion.start_pdf_to_excel` with table/Excel parameters to produce CSV-friendly output\n\n## Input Source Types\n- The script supports **PDF and image** as input sources. The SDK's `start_pdf_to_*` interfaces natively accept image files with no pre-processing required.\n- By default, the script auto-detects the input type from the file extension:\n  - `.pdf` → `pdf`\n  - `.png/.jpg/.jpeg/.bmp/.tif/.tiff/.gif/.webp/.tga` → `image`\n- You can also specify the source type explicitly:\n  ```bash\n  python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.png output.docx --source-type image\n  ```\n- `image -> *` and `pdf -> *` share the same set of `CPDFConversion.start_pdf_to_*` interfaces; only the input file type differs.\n\n## Smart Defaults\nThe script automatically adjusts certain parameters based on the input source and output format to reduce manual configuration:\n\n| Trigger | Automatic Behavior | User-Overridable | Description |\n|----------|----------|-------------|------|\n| Input source is an **image** (auto-detected or explicit `--source-type image`) | Automatically enables `--enable-ocr` | No (`--enable-ocr` uses `store_true`; there is no `--no-enable-ocr`) | Text in images must be extracted via OCR; without OCR, output will contain only images and no text |\n| Output format is **HTML** (`format = html`) | Automatically sets `--page-layout-mode` to `box` (box layout) | Yes — passing `--page-layout-mode flow` explicitly overrides this | Box layout better preserves the original formatting in HTML; specify `flow` explicitly if flow layout is needed |\n\nWhen triggered, the script prints a notice to `stderr`, for example:\n```text\nAuto-enabled OCR for image input.\nAuto-set page layout mode to BOX for HTML output.\n```\n\n## All Parameters\n\n### Positional Parameters\n| Parameter | Description |\n|------|------|\n| `format` | Target format: `word`/`excel`/`ppt`/`html`/`rtf`/`image`/`txt`/`json`/`markdown`/`csv` |\n| `input_pdf` | Input file path (PDF or image) |\n| `output_path` | Output file path |\n\n### General Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--source-type` | Option | `auto` | Input source type: `auto`/`pdf`/`image` |\n| `--password` | String | `\"\"` | PDF open password |\n| `--page-ranges` | String | None | Page range, e.g. `1-3,5` |\n| `--font-name` | String | `\"\"` | Output font name |\n\n### Layout Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--enable-ai-layout` | Boolean | **True** | AI layout analysis (disable with `--no-enable-ai-layout`) |\n| `--page-layout-mode` | Option | SDK default `flow` (auto-switched to `box` for HTML output) | Page layout: `box` (box layout) / `flow` (flow layout) |\n\n### Content Retention Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--contain-image` | Boolean | **True** | Retain images (disable with `--no-contain-image`) |\n| `--contain-annotation` | Boolean | **True** | Retain annotations (disable with `--no-contain-annotation`) |\n| `--contain-page-background-image` | Boolean | **True** | Retain page background images (disable with `--no-contain-page-background-image`) |\n| `--formula-to-image` | Boolean | False | Convert formulas to image output |\n| `--transparent-text` | Boolean | False | Preserve transparent text |\n\n### Output Control Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--output-document-per-page` | Boolean | False | Split output into one document per page |\n| `--auto-create-folder` | Boolean | **True** | Automatically create output directory (disable with `--no-auto-create-folder`) |\n\n### OCR Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--enable-ocr` | Boolean | False (auto-enabled for image input) | Enable OCR |\n| `--ocr-option` | Option | SDK default `all` | OCR scope: `invalid-character`/`scan-page`/`invalid-character-and-scan-page`/`all` |\n| `--ocr-language` | Multi-select | `auto` | OCR language(s); multiple languages can be specified simultaneously. Options: `auto`/`chinese`/`chinese-tra`/`english`/`korean`/`japanese`/`latin`/`devanagari`/`cyrillic`/`arabic`/`tamil`/`telugu`/`kannada`/`thai`/`greek`/`eslav` |\n\n### Excel-Specific Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--excel-all-content` | Boolean | False | Include all content in Excel output |\n| `--excel-csv-format` | Boolean | False | Output Excel result in CSV format |\n| `--excel-worksheet-option` | Option | SDK default `for-table` | Worksheet split strategy: `for-table`/`for-page`/`for-document` |\n\n### JSON-Specific Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--json-contain-table` | Boolean | **True** | Include table data in JSON output (disable with `--no-json-contain-table`) |\n\n### TXT-Specific Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--txt-table-format` | Boolean | **True** | Enable table formatting in TXT output (disable with `--no-txt-table-format`) |\n\n### HTML-Specific Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--html-option` | Option | SDK default `single-page` | HTML output mode: `single-page`/`single-page-with-bookmark`/`multiple-page`/`multiple-page-with-bookmark` |\n\n### Image-Specific Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--image-type` | Option | SDK default `jpg` | Image output format: `jpg`/`jpeg`/`jpeg2000`/`png`/`bmp`/`tiff`/`tga`/`gif`/`webp` |\n| `--image-color-mode` | Option | SDK default `color` | Image color mode: `color`/`gray`/`binary` |\n| `--image-scaling` | Float | `1.0` | Image scaling factor |\n| `--image-path-enhance` | Boolean | False | Enable image path enhancement |\n\n### Parameter Default Value Rules\n- **Parameters that default to True** (`--enable-ai-layout`/`--contain-image`/`--contain-annotation`/`--contain-page-background-image`/`--auto-create-folder`/`--json-contain-table`/`--txt-table-format`) use `BooleanOptionalAction`; pass `--no-xxx` to disable.\n- **Parameters that default to False** (`--enable-ocr`/`--formula-to-image`/`--transparent-text`/`--output-document-per-page`/`--excel-all-content`/`--excel-csv-format`/`--image-path-enhance`) use `store_true`; passing the flag enables them.\n- **All CLI parameter defaults are fully consistent with the SDK's `ConvertOptions()` defaults** — omitting a parameter is equivalent to using the SDK's original default value.\n\n## Recommended Command Examples\n\n### PDF to Word (default parameters, AI layout analysis enabled)\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.pdf output.docx\n```\n\n### PDF to Word, box layout, no images, no AI layout analysis\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.pdf output.docx --no-enable-ai-layout --no-contain-image --page-layout-mode box\n```\n\n### PDF to Word, retain annotations and background images, one document per page\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.pdf output.docx --output-document-per-page\n```\n\n### PDF to Excel, include all content and split worksheets by page\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" excel input.pdf output.xlsx --excel-all-content --excel-worksheet-option for-page\n```\n\n### PDF to TXT, with table formatting enabled\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" txt input.pdf output.txt\n```\n\n### PDF to HTML, multi-page with bookmarks mode\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" html input.pdf output_dir --html-option multiple-page-with-bookmark\n```\n\n### PDF to Image, PNG format, grayscale, 2x scaling\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" image input.pdf output.png --image-type png --image-color-mode gray --image-scaling 2.0\n```\n\n### Image to Word (OCR auto-enabled, specify Chinese language)\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.png output.docx --ocr-language chinese\n```\n> Note: For image input, the script automatically enables OCR — there is no need to pass `--enable-ocr` manually. To specify an OCR language, `--ocr-language` can still be used.\n\n### PDF with OCR enabled (multiple languages)\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.pdf output.docx --enable-ocr --ocr-language chinese english japanese\n```\n\n## Trial License and Usage Limits\n- The `scripts/license.xml` auto-downloaded from the ComPDF server is a **Trial License**, allowing a maximum of **200 conversions**.\n- The script uses a SHA-256 fingerprint to detect whether the current License is the default trial key; **no usage limit applies when using any other License**.\n- After each successful conversion using the trial License, the script prints the current used/remaining count to `stderr`, for example:\n  ```text\n  Trial license: 5/200 conversions used, 195 remaining.\n  ```\n- When the trial limit is reached (200 conversions), the script refuses to convert and prompts the user to purchase a full License:\n  ```text\n  Error: Trial license usage limit reached (200 conversions). Please purchase a license at: https://www.compdf.com/contact-sales?utm_source=clawhub&utm_medium=skillhub&utm_campaign=pdf_skill_to_word&ref_platform_id=clawhub_skills\n  ```\n- When the trial License has expired (SDK authentication fails), the error message also includes a purchase link.\n- **After purchasing a full License**, place a custom `license.xml` containing the new `<key>` in `scripts/` (overwriting the auto-downloaded trial file) — no script modifications or counter file cleanup are required.\n\n## Confirmed Facts\n- `ComPDFKitConversion` has been successfully installed on the local machine.\n- The installed package provides 10 conversion methods including `CPDFConversion.start_pdf_to_word/start_pdf_to_ppt/start_pdf_to_excel`.\n- `LibraryManager` provides `initialize`, `license_verify`, `release`, `set_document_ai_model`, and `set_ocr_language`.\n- Official documentation confirms support for PDF to Word / Excel / PPT / HTML / RTF / Image / TXT / JSON / Markdown.\n- The SDK's `start_pdf_to_*` interfaces natively accept image file input (PNG → Word has been verified successfully).\n- `enable_ai_layout` defaults to `True` in the SDK; `set_document_ai_model()` must be called first to load the model before use, otherwise a 0xC0000005 crash will occur.\n- `--ocr-language` supports specifying multiple languages simultaneously (e.g. `--ocr-language chinese english`).\n\n## Risks / Notes\n- The official requirements page states Python `>=3.6`, while the demo page states `<3.11`, but PyPI currently provides a `cp314` wheel in practice; treat the locally installable wheel as the source of truth, but always verify installation in a new environment first.\n- If the script cannot download `license.xml` from the server (network issue) and no manual file exists in `scripts/`, or the `<key>` field is empty, the script cannot complete SDK authentication and cannot perform any real conversions.\n- `documentai.model` is a large file (approximately 525 MB); there will be a noticeable download delay the first time OCR / AI layout is enabled. Because `--enable-ai-layout` defaults to True, **the model download will be triggered on the very first run**.\n- If the runtime environment cannot access `https://download.compdf.com/skills/model/documentai.model`, place `documentai.model` in the `scripts/` directory in advance.\n- Do not directly apply the initialization patterns from ComPDF SDKs for other languages to the Python package; this Skill is based on the locally verified `LibraryManager` / `CPDFConversion` API.\n\n## Resource Navigation\n- License file: `License.txt`\n- Script: `scripts/pdf-to-word-docx.py`\n- SDK authentication file: `scripts/license.xml` (auto-downloaded from `https://download.compdf.com/skills/license/license.xml` if missing)\n- SDK authentication source: the `<key>` field in `license.xml`\n- SDK resource path: `scripts/`\n- OCR / AI layout model: `scripts/documentai.model` (auto-downloaded if missing)\n- Purchase a full License: `https://www.compdf.com/contact-sales?utm_source=clawhub&utm_medium=skillhub&utm_campaign=pdf_skill_to_word&ref_platform_id=clawhub_skills`\n- Official documentation:\n  - `https://www.compdf.com/guides/conversion-sdk/python/overview`\n  - `https://www.compdf.com/guides/conversion-sdk/python/pdf-to-word`\n  - `https://www.compdf.com/guides/conversion-sdk/python/pdf-to-excel`\n  - `https://www.compdf.com/guides/conversion-sdk/python/pdf-to-ppt`\n  - `https://www.compdf.com/guides/conversion-sdk/python/apply-license`\n\n## Acceptance Checklist\n- [ ] `python -m pip show ComPDFKitConversion` shows the installed package\n- [ ] Running `python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" --help` or an equivalent local command produces normal output\n- [ ] The script auto-downloads `scripts/license.xml` if missing, then extracts the license key from the `<key>` field for authentication\n- [ ] The script uses the `scripts/` directory as the SDK resource path\n- [ ] The script recognizes all 10 target formats: `word`/`excel`/`ppt`/`html`/`rtf`/`image`/`txt`/`json`/`markdown`/`csv`\n- [ ] The script accepts both PDF and image files (`.png`/`.jpg`/`.jpeg`/`.bmp`/`.tif`/`.tiff`/`.gif`/`.webp`/`.tga`) as input\n- [ ] When `--enable-ocr` or `--enable-ai-layout` (enabled by default) is active and `documentai.model` is missing, the script auto-downloads the model\n- [ ] When `license.xml` cannot be obtained (download fails and no manual file exists) or authentication fails, a clear error is output rather than a silent failure\n- [ ] The 7 parameters that default to True can be disabled with `--no-xxx`\n- [ ] `--ocr-language` supports specifying multiple languages simultaneously\n- [ ] After a conversion using the trial License, the usage count increments\n- [ ] When the trial License reaches 200 conversions, the script refuses to convert and outputs a purchase link\n- [ ] When using a non-trial License, no usage limit applies\n- [ ] For image input, even if `--enable-ocr` is not passed, the script automatically enables OCR and prints a notice to `stderr`\n- [ ] For HTML output, even if `--page-layout-mode` is not passed, the script automatically uses `box` (box layout) and prints a notice to `stderr`\n- [ ] For HTML output, explicitly passing `--page-layout-mode flow` overrides the automatic box layout behavior\n\n## Distribution Notes\n- This Skill does not depend on any machine-specific absolute paths.\n- When distributing to other users, the following directory structure is sufficient:\n  ```text\n  pdf-to-word-docx/\n  ├── SKILL.md\n  ├── License.txt\n  └── scripts/\n      └── pdf-to-word-docx.py\n  ```\n- Users place this directory under their own skills root directory and the Skill is ready to use.\n- `license.xml` is auto-downloaded at runtime; no need to include it in the distribution package.\n\n## Common Pitfalls\n- `scripts/license.xml` is missing and cannot be auto-downloaded (network unavailable or server error): the script will error out before authentication. If you are in an offline environment, place `license.xml` manually in the `scripts/` directory.\n- `scripts/license.xml` is missing the `<key>` field or its value is empty: the script will error out before authentication.\n- SDK resource files required by the SDK are absent from the `scripts/` directory: conversion may fail after `LibraryManager.initialize()`.\n- A password-protected PDF is provided without `--password`: this will trigger `PDF_PASSWORD_ERROR`.\n- OCR / AI layout is enabled but `documentai.model` is not present locally and the network is unavailable: the model download will fail; place the file in the `scripts/` directory manually in advance.\n- When the Excel output strategy is unclear, prefer passing `--excel-worksheet-option` explicitly to avoid unexpected result structures.\n- When converting images to other formats, the script already enables OCR automatically; if the output still contains no text, check whether `documentai.model` is complete and whether the OCR language matches.\n- Once the trial License usage limit is exhausted, a full License must be purchased to continue; purchase link: `https://www.compdf.com/contact-sales?utm_source=clawhub&utm_medium=skillhub&utm_campaign=pdf_skill_to_word&ref_platform_id=clawhub_skills`.\n\n## Copyright\n\nThis Skill is built on top of the [ComPDFKit Conversion SDK](https://www.compdf.com/pdf-sdk?utm_source=clawhub&utm_medium=skillhub&utm_campaign=compdf_pdf_skill&ref_platform_id=clawhub_skills).\n\n```\n© 2014-2026 PDF Technologies, Inc., a KDAN Company. All Rights Reserved.\n```\n\n- **SDK Name**: ComPDFKitConversion\n- **SDK Author**: PDF Technologies, Inc.\n- **License Type**: Commercial License (Commercial / Proprietary) — non-exclusive, non-transferable, non-sublicensable, revocable\n- **Official Website**: https://www.compdf.com/?utm_source=clawhub&utm_medium=skillhub&utm_campaign=compdf_pdf_skill_to_word&ref_platform_id=clawhub_skills\n- **Contact**: support@compdf.com\n- **Terms of Service**: https://www.compdf.com/terms-of-service/?utm_source=clawhub&utm_medium=skillhub&utm_campaign=compdf_pdf_skill_to_word&ref_platform_id=clawhub_skills\n- **Privacy Policy**: https://www.compdf.com/privacy-policy/?utm_source=clawhub&utm_medium=skillhub&utm_campaign=compdf_pdf_skill_to_word&ref_platform_id=clawhub_skills\n\n> **Important**: Under the ComPDFKit Terms of Service, distributing the documentation, sample code, or source code of the ComPDFKit Conversion SDK to third parties is prohibited. Please ensure you have obtained a valid ComPDFKit License before using this Skill.\n\nFile v1.2.0:_meta.json\n\n{\n  \"ownerId\": \"kn75g40xmv3tgkb21e079d055n82v5h8\",\n  \"slug\": \"pdf-to-word-docx\",\n  \"version\": \"1.2.0\",\n  \"publishedAt\": 1782272804581\n}\n\nFile v1.2.0:skill-card.md\n\n## Description:\n\nPDF to Word converts PDF to editable Word/DOCX with AI-powered layout analysis and table recognition, built on ComPDF Conversion SDK to better preserve tables, multi-column layouts, lists, and images for downstream editing.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[compdf-youna](https://clawhub.ai/user/compdf-youna)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nExternal users, developers, and document automation agents use this skill to convert PDFs or image files into editable office and text formats while preserving layout, tables, lists, and images. It is suited for document editing, extraction, and format migration workflows that can run on Windows or macOS with the ComPDFKitConversion package.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill depends on ComPDF's proprietary SDK and requires license verification before conversion.\n\nMitigation: Review the ComPDFKitConversion package and license terms before use, and provide an approved license.xml for production workflows.\n\nRisk: First-run execution can download license.xml and a large documentai.model file from ComPDF when local copies are missing.\n\nMitigation: Run in a restricted workspace, pre-stage verified license and model files where possible, and allow network access only to expected ComPDF download locations.\n\nRisk: The trial license counter is stored in the user's home directory and may affect repeat conversions.\n\nMitigation: Use a full license for production or shared environments, and account for the local trial counter when testing.\n\nRisk: Sensitive documents are processed by a local SDK and may produce editable output files containing the original document content.\n\nMitigation: Convert sensitive files only in approved local workspaces with appropriate file access controls and output cleanup procedures.\n\n## Reference(s):\n\n- [ClawHub Skill Page](https://clawhub.ai/compdf-youna/skills/pdf-to-word-docx)\n- [ComPDF Homepage](https://www.compdf.com/?utm_source=clawhub&utm_medium=skillhub&utm_campaign=compdf_pdf_skill_to_word&ref_platform_id=clawhub_skills)\n- [ComPDF Conversion SDK for Python Overview](https://www.compdf.com/guides/conversion-sdk/python/overview)\n- [ComPDF PDF to Word Guide](https://www.compdf.com/guides/conversion-sdk/python/pdf-to-word)\n- [ComPDF PDF to Excel Guide](https://www.compdf.com/guides/conversion-sdk/python/pdf-to-excel)\n- [ComPDF PDF to PPT Guide](https://www.compdf.com/guides/conversion-sdk/python/pdf-to-ppt)\n- [ComPDF Apply License Guide](https://www.compdf.com/guides/conversion-sdk/python/apply-license)\n\n## Skill Output:\n\n**Output Type(s):** [Files, Text, Markdown, Shell commands, Configuration instructions]\n\n**Output Format:** [Converted document files with command-line status text and error messages]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Supports DOCX, PPTX, XLSX, HTML, RTF, image, TXT, JSON, Markdown, and CSV outputs from PDF or image inputs.]\n\n## Skill Version(s):\n\n1.2.0 (source: frontmatter and server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.2.0:License.txt\n\npdf to word - License Notice\n============================\n\nCopyright (c) 2014-2026 PDF Technologies, Inc., a KDAN Company.\nAll Rights Reserved.\n\nThis tool is built on the ComPDFKit Conversion SDK, a commercial product\nof PDF Technologies, Inc.\n\n------------------------------------------------------------------------\n1. LICENSE TYPE\n------------------------------------------------------------------------\n\nThe ComPDFKit Conversion SDK is provided under a COMMERCIAL / PROPRIETARY\nlicense. The license is non-exclusive, non-transferable, non-sublicensable,\nand revocable.\n\nYou must obtain a valid license key from PDF Technologies, Inc. before\nusing this software. A trial license is included for evaluation purposes\nonly, limited to 200 conversions.\n\n------------------------------------------------------------------------\n2. USAGE RESTRICTIONS\n------------------------------------------------------------------------\n\nBy using this software, you agree to the ComPDFKit Terms of Service:\nhttps://www.compdf.com/terms-of-service/?utm_source=clawhub&utm_medium=skillhub&utm_campaign=compdf_pdf_skill_to_word&ref_platform_id=clawhub_skills\n\nIn particular:\n\n  - You may NOT distribute, share, or disclose any documents, sample code,\n    or source code of the ComPDFKit Conversion SDK to any third party.\n\n  - You may NOT reverse-engineer, decompile, or disassemble the SDK\n    binaries.\n\n  - You may NOT use the SDK for any unlawful purpose or in violation of\n    applicable laws.\n\n  - The trial license is for evaluation only and must not be used in\n    production environments.\n\n------------------------------------------------------------------------\n3. DISCLAIMER OF WARRANTIES\n------------------------------------------------------------------------\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS\nOR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,\nFITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.\n\nIN NO EVENT SHALL PDF TECHNOLOGIES, INC. OR ITS AFFILIATES BE LIABLE FOR\nANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT,\nTORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE\nOR THE USE OR OTHER DEALINGS IN THE SOFTWARE.\n\n------------------------------------------------------------------------\n4. CONTACT & PURCHASE\n------------------------------------------------------------------------\n\n  Website: https://www.compdf.com/pdf-sdk?utm_source=clawhub&utm_medium=skillhub&utm_campaign=compdf_pdf_skill&ref_platform_id=clawhub_skills\n\n  Purchase License: https://www.compdf.com/contact-sales?utm_source=clawhub&utm_medium=skillhub&utm_campaign=pdf_skill_to_word&ref_platform_id=clawhub_skills\n\n  Email: support@compdf.com\n\n  Terms of Service: https://www.compdf.com/terms-of-service/?utm_source=clawhub&utm_medium=skillhub&utm_campaign=compdf_pdf_skill_to_word&ref_platform_id=clawhub_skills\n\n  Privacy Policy: https://www.compdf.com/privacy-policy/?utm_source=clawhub&utm_medium=skillhub&utm_campaign=compdf_pdf_skill_to_word&ref_platform_id=clawhub_skills\n\nArchive v1.1.0: 5 files, 16217 bytes\n\nFiles: License.txt (2543b), scripts/pdf-to-word-docx.py (20602b), skill-card.md (3061b), SKILL.md (23699b), _meta.json (135b)\n\nFile v1.1.0:SKILL.md\n\n---\nname: pdf-to-word-docx\nversion: 1.1.0\ndescription: PDF to Word converts PDF to editable Word/DOCX with AI-powered layout analysis and table recognition, built on ComPDF Conversion SDK to better preserve tables, multi-column layouts, lists, and images for downstream editing. It fits requests such as “pdf to word,” “convert pdf to docx,” “pdf to editable word,” “pdf to office,” “keep layout in word,” and “convert report to docx.” Example queries include “Convert this PDF contract to editable Word while keeping the tables intact,” “Turn this report into DOCX and preserve the multi-column layout,” and “Export this PDF to Word for further editing.”\nhomepage: https://www.compdf.com\nmetadata:\n  clawdbot:\n    emoji: \"📑\"\n    requires:\n      env: []\n    files: [\"scripts/*\"]\ncompatibility: Requires Windows or macOS. Python with ComPDFKitConversion package (pip install ComPDFKitConversion). AI model (~525MB) auto-downloaded on first run.\n---\n\n# PDF to Word Converter\n\n## Purpose\n- Wraps the `ComPDFKitConversion` Python SDK into a reusable local conversion workflow, supporting PDF / image to Word, PPT, Excel, HTML, RTF, Image, TXT, JSON, Markdown, and CSV (10 output formats in total).\n\n## Agent Skills Standard Compatibility\n- This Skill uses an Anthropic Agent Skills-compatible directory structure: `pdf-to-word-docx/`.\n- The entry point is `SKILL.md`; helper scripts are placed in `scripts/`.\n- The document uses `$ARGUMENTS` and `${CLAUDE_SKILL_DIR}` conventions for distribution and execution in Claude Code / Agent Skills-compatible environments.\n\n## Input / Output\n- Input: The target format (`word`/`excel`/`ppt`/`html`/`rtf`/`image`/`txt`/`json`/`markdown`/`csv`), the PDF or image path, and the output path are passed via Skill arguments or the command line. An optional PDF password and conversion parameters may also be provided.\n- Supported input file types:\n  - PDF files (`.pdf`)\n  - Image files (`.jpg`/`.jpeg`/`.png`/`.bmp`/`.tif`/`.tiff`/`.webp`/`.jp2`/`.gif`/`.tga`)\n- Output: A file in the corresponding format (`.docx`, `.pptx`, `.xlsx`, `.html`, `.rtf`, image, `.txt`, `.json`, `.md`, `.csv`), or a clear error message.\n\n## Prerequisites\n- Supports Windows and macOS.\n- The conversion SDK must be installed first:\n  ```bash\n  pip install ComPDFKitConversion\n  ```\n- On first run, the script automatically downloads `license.xml` from the ComPDF server and caches it in the `scripts/` directory:\n  ```text\n  https://download.compdf.com/skills/license/license.xml\n  ```\n- The script reads the `<key>...</key>` field from `license.xml` and uses that key for `LibraryManager.license_verify(...)` authentication — it does not pass the XML file path directly to the SDK.\n- To use a custom license, place your own `license.xml` in the `scripts/` directory; the script will use it directly without downloading.\n- During SDK initialization, the `resource` directory is always set to the directory containing `pdf-to-word-docx.py`, i.e., the `scripts/` directory itself.\n- When `--enable-ocr` or `--enable-ai-layout` (enabled by default) is used, the Skill also requires `scripts/documentai.model`. If the file does not exist, the script will automatically download it from:\n  ```text\n  https://download.compdf.com/skills/model/documentai.model\n  ```\n- To reuse an existing model file, you can override the default model path via an environment variable:\n  ```bash\n  export COMPDF_DOCUMENT_AI_MODEL=\"/path/to/documentai.model\"\n  ```\n\n## Workflow\n1. Confirm the Python package is installed:\n   ```bash\n   python -m pip show ComPDFKitConversion\n   ```\n2. The script automatically downloads `license.xml` on first run; the `scripts/` directory is used directly as the SDK `resource` path.\n3. In Agent Skills / Claude Code environments, prefer using the Skill's built-in script path variable:\n   ```bash\n   python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.pdf output.docx\n   python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" ppt input.pdf output.pptx\n   python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" excel input.pdf output.xlsx\n   ```\n4. For more control, append common parameters:\n   ```bash\n   python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" excel input.pdf output.xlsx --page-ranges \"1-3,5\" --excel-all-content --excel-worksheet-option for-page\n   python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.pdf output.docx --enable-ocr --page-layout-mode flow\n   ```\n5. On startup, the script ensures `scripts/license.xml` exists (downloading it automatically from the ComPDF server if missing), reads the `<key>` field for SDK authentication, and uses the `scripts/` directory as the `resource` path.\n6. If `--enable-ocr` or `--enable-ai-layout` (enabled by default) is active, the script checks whether `scripts/documentai.model` exists; if not, it downloads the file automatically before initializing the Document AI model.\n7. Check the return code; if it is not `SUCCESS`, handle license, password, resource, model, or input file issues according to the error name.\n\n## documentai.model Download Optimization\n- The script preferentially uses the model file pointed to by `COMPDF_DOCUMENT_AI_MODEL`.\n- The default model path is `scripts/documentai.model`.\n- During automatic download, the file is first written to `documentai.model.part` and then atomically renamed to the final file upon success, preventing partial file corruption.\n- On download failure, the script retries automatically with back-off intervals of `2s / 5s / 10s`.\n\n## Invoking Directly as a Skill\n- In environments that support Agent Skills, the Skill can be called directly:\n  ```text\n  /pdf-to-word-docx word input.pdf output.docx\n  /pdf-to-word-docx excel input.pdf output.xlsx --excel-worksheet-option for-page\n  ```\n- When the Skill receives arguments, it passes them through to the script as-is:\n  ```bash\n  python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" $ARGUMENTS\n  ```\n- If the environment does not support direct Skill invocation, fall back to a regular command-line call.\n\n## Supported Output Formats\n- `word` → calls `CPDFConversion.start_pdf_to_word`\n- `excel` → calls `CPDFConversion.start_pdf_to_excel`\n- `ppt` → calls `CPDFConversion.start_pdf_to_ppt`\n- `html` → calls `CPDFConversion.start_pdf_to_html`\n- `rtf` → calls `CPDFConversion.start_pdf_to_rtf`\n- `image` → calls `CPDFConversion.start_pdf_to_image`\n- `txt` → calls `CPDFConversion.start_pdf_to_txt`\n- `json` → calls `CPDFConversion.start_pdf_to_json`\n- `markdown` → calls `CPDFConversion.start_pdf_to_markdown`\n- `csv` → reuses `CPDFConversion.start_pdf_to_excel` with table/Excel parameters to produce CSV-friendly output\n\n## Input Source Types\n- The script supports **PDF and image** as input sources. The SDK's `start_pdf_to_*` interfaces natively accept image files with no pre-processing required.\n- By default, the script auto-detects the input type from the file extension:\n  - `.pdf` → `pdf`\n  - `.png/.jpg/.jpeg/.bmp/.tif/.tiff/.gif/.webp/.tga` → `image`\n- You can also specify the source type explicitly:\n  ```bash\n  python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.png output.docx --source-type image\n  ```\n- `image -> *` and `pdf -> *` share the same set of `CPDFConversion.start_pdf_to_*` interfaces; only the input file type differs.\n\n## Smart Defaults\nThe script automatically adjusts certain parameters based on the input source and output format to reduce manual configuration:\n\n| Trigger | Automatic Behavior | User-Overridable | Description |\n|----------|----------|-------------|------|\n| Input source is an **image** (auto-detected or explicit `--source-type image`) | Automatically enables `--enable-ocr` | No (`--enable-ocr` uses `store_true`; there is no `--no-enable-ocr`) | Text in images must be extracted via OCR; without OCR, output will contain only images and no text |\n| Output format is **HTML** (`format = html`) | Automatically sets `--page-layout-mode` to `box` (box layout) | Yes — passing `--page-layout-mode flow` explicitly overrides this | Box layout better preserves the original formatting in HTML; specify `flow` explicitly if flow layout is needed |\n\nWhen triggered, the script prints a notice to `stderr`, for example:\n```text\nAuto-enabled OCR for image input.\nAuto-set page layout mode to BOX for HTML output.\n```\n\n## All Parameters\n\n### Positional Parameters\n| Parameter | Description |\n|------|------|\n| `format` | Target format: `word`/`excel`/`ppt`/`html`/`rtf`/`image`/`txt`/`json`/`markdown`/`csv` |\n| `input_pdf` | Input file path (PDF or image) |\n| `output_path` | Output file path |\n\n### General Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--source-type` | Option | `auto` | Input source type: `auto`/`pdf`/`image` |\n| `--password` | String | `\"\"` | PDF open password |\n| `--page-ranges` | String | None | Page range, e.g. `1-3,5` |\n| `--font-name` | String | `\"\"` | Output font name |\n\n### Layout Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--enable-ai-layout` | Boolean | **True** | AI layout analysis (disable with `--no-enable-ai-layout`) |\n| `--page-layout-mode` | Option | SDK default `flow` (auto-switched to `box` for HTML output) | Page layout: `box` (box layout) / `flow` (flow layout) |\n\n### Content Retention Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--contain-image` | Boolean | **True** | Retain images (disable with `--no-contain-image`) |\n| `--contain-annotation` | Boolean | **True** | Retain annotations (disable with `--no-contain-annotation`) |\n| `--contain-page-background-image` | Boolean | **True** | Retain page background images (disable with `--no-contain-page-background-image`) |\n| `--formula-to-image` | Boolean | False | Convert formulas to image output |\n| `--transparent-text` | Boolean | False | Preserve transparent text |\n\n### Output Control Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--output-document-per-page` | Boolean | False | Split output into one document per page |\n| `--auto-create-folder` | Boolean | **True** | Automatically create output directory (disable with `--no-auto-create-folder`) |\n\n### OCR Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--enable-ocr` | Boolean | False (auto-enabled for image input) | Enable OCR |\n| `--ocr-option` | Option | SDK default `all` | OCR scope: `invalid-character`/`scan-page`/`invalid-character-and-scan-page`/`all` |\n| `--ocr-language` | Multi-select | `auto` | OCR language(s); multiple languages can be specified simultaneously. Options: `auto`/`chinese`/`chinese-tra`/`english`/`korean`/`japanese`/`latin`/`devanagari`/`cyrillic`/`arabic`/`tamil`/`telugu`/`kannada`/`thai`/`greek`/`eslav` |\n\n### Excel-Specific Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--excel-all-content` | Boolean | False | Include all content in Excel output |\n| `--excel-csv-format` | Boolean | False | Output Excel result in CSV format |\n| `--excel-worksheet-option` | Option | SDK default `for-table` | Worksheet split strategy: `for-table`/`for-page`/`for-document` |\n\n### JSON-Specific Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--json-contain-table` | Boolean | **True** | Include table data in JSON output (disable with `--no-json-contain-table`) |\n\n### TXT-Specific Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--txt-table-format` | Boolean | **True** | Enable table formatting in TXT output (disable with `--no-txt-table-format`) |\n\n### HTML-Specific Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--html-option` | Option | SDK default `single-page` | HTML output mode: `single-page`/`single-page-with-bookmark`/`multiple-page`/`multiple-page-with-bookmark` |\n\n### Image-Specific Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--image-type` | Option | SDK default `jpg` | Image output format: `jpg`/`jpeg`/`jpeg2000`/`png`/`bmp`/`tiff`/`tga`/`gif`/`webp` |\n| `--image-color-mode` | Option | SDK default `color` | Image color mode: `color`/`gray`/`binary` |\n| `--image-scaling` | Float | `1.0` | Image scaling factor |\n| `--image-path-enhance` | Boolean | False | Enable image path enhancement |\n\n### Parameter Default Value Rules\n- **Parameters that default to True** (`--enable-ai-layout`/`--contain-image`/`--contain-annotation`/`--contain-page-background-image`/`--auto-create-folder`/`--json-contain-table`/`--txt-table-format`) use `BooleanOptionalAction`; pass `--no-xxx` to disable.\n- **Parameters that default to False** (`--enable-ocr`/`--formula-to-image`/`--transparent-text`/`--output-document-per-page`/`--excel-all-content`/`--excel-csv-format`/`--image-path-enhance`) use `store_true`; passing the flag enables them.\n- **All CLI parameter defaults are fully consistent with the SDK's `ConvertOptions()` defaults** — omitting a parameter is equivalent to using the SDK's original default value.\n\n## Recommended Command Examples\n\n### PDF to Word (default parameters, AI layout analysis enabled)\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.pdf output.docx\n```\n\n### PDF to Word, box layout, no images, no AI layout analysis\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.pdf output.docx --no-enable-ai-layout --no-contain-image --page-layout-mode box\n```\n\n### PDF to Word, retain annotations and background images, one document per page\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.pdf output.docx --output-document-per-page\n```\n\n### PDF to Excel, include all content and split worksheets by page\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" excel input.pdf output.xlsx --excel-all-content --excel-worksheet-option for-page\n```\n\n### PDF to TXT, with table formatting enabled\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" txt input.pdf output.txt\n```\n\n### PDF to HTML, multi-page with bookmarks mode\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" html input.pdf output_dir --html-option multiple-page-with-bookmark\n```\n\n### PDF to Image, PNG format, grayscale, 2x scaling\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" image input.pdf output.png --image-type png --image-color-mode gray --image-scaling 2.0\n```\n\n### Image to Word (OCR auto-enabled, specify Chinese language)\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.png output.docx --ocr-language chinese\n```\n> Note: For image input, the script automatically enables OCR — there is no need to pass `--enable-ocr` manually. To specify an OCR language, `--ocr-language` can still be used.\n\n### PDF with OCR enabled (multiple languages)\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.pdf output.docx --enable-ocr --ocr-language chinese english japanese\n```\n\n## Trial License and Usage Limits\n- The `scripts/license.xml` auto-downloaded from the ComPDF server is a **Trial License**, allowing a maximum of **200 conversions**.\n- The script uses a SHA-256 fingerprint to detect whether the current License is the default trial key; **no usage limit applies when using any other License**.\n- After each successful conversion using the trial License, the script prints the current used/remaining count to `stderr`, for example:\n  ```text\n  Trial license: 5/200 conversions used, 195 remaining.\n  ```\n- When the trial limit is reached (200 conversions), the script refuses to convert and prompts the user to purchase a full License:\n  ```text\n  Error: Trial license usage limit reached (200 conversions). Please purchase a license at: https://www.compdf.com/contact-sales\n  ```\n- When the trial License has expired (SDK authentication fails), the error message also includes a purchase link.\n- **After purchasing a full License**, place a custom `license.xml` containing the new `<key>` in `scripts/` (overwriting the auto-downloaded trial file) — no script modifications or counter file cleanup are required.\n\n## Confirmed Facts\n- `ComPDFKitConversion` has been successfully installed on the local machine.\n- The installed package provides 10 conversion methods including `CPDFConversion.start_pdf_to_word/start_pdf_to_ppt/start_pdf_to_excel`.\n- `LibraryManager` provides `initialize`, `license_verify`, `release`, `set_document_ai_model`, and `set_ocr_language`.\n- Official documentation confirms support for PDF to Word / Excel / PPT / HTML / RTF / Image / TXT / JSON / Markdown.\n- The SDK's `start_pdf_to_*` interfaces natively accept image file input (PNG → Word has been verified successfully).\n- `enable_ai_layout` defaults to `True` in the SDK; `set_document_ai_model()` must be called first to load the model before use, otherwise a 0xC0000005 crash will occur.\n- `--ocr-language` supports specifying multiple languages simultaneously (e.g. `--ocr-language chinese english`).\n\n## Risks / Notes\n- The official requirements page states Python `>=3.6`, while the demo page states `<3.11`, but PyPI currently provides a `cp314` wheel in practice; treat the locally installable wheel as the source of truth, but always verify installation in a new environment first.\n- If the script cannot download `license.xml` from the server (network issue) and no manual file exists in `scripts/`, or the `<key>` field is empty, the script cannot complete SDK authentication and cannot perform any real conversions.\n- `documentai.model` is a large file (approximately 525 MB); there will be a noticeable download delay the first time OCR / AI layout is enabled. Because `--enable-ai-layout` defaults to True, **the model download will be triggered on the very first run**.\n- If the runtime environment cannot access `https://download.compdf.com/skills/model/documentai.model`, place `documentai.model` in the `scripts/` directory in advance.\n- Do not directly apply the initialization patterns from ComPDF SDKs for other languages to the Python package; this Skill is based on the locally verified `LibraryManager` / `CPDFConversion` API.\n\n## Resource Navigation\n- License file: `License.txt`\n- Script: `scripts/pdf-to-word-docx.py`\n- SDK authentication file: `scripts/license.xml` (auto-downloaded from `https://download.compdf.com/skills/license/license.xml` if missing)\n- SDK authentication source: the `<key>` field in `license.xml`\n- SDK resource path: `scripts/`\n- OCR / AI layout model: `scripts/documentai.model` (auto-downloaded if missing)\n- Purchase a full License: `https://www.compdf.com/contact-sales`\n- Official documentation:\n  - `https://www.compdf.com/guides/conversion-sdk/python/overview`\n  - `https://www.compdf.com/guides/conversion-sdk/python/pdf-to-word`\n  - `https://www.compdf.com/guides/conversion-sdk/python/pdf-to-excel`\n  - `https://www.compdf.com/guides/conversion-sdk/python/pdf-to-ppt`\n  - `https://www.compdf.com/guides/conversion-sdk/python/apply-license`\n\n## Acceptance Checklist\n- [ ] `python -m pip show ComPDFKitConversion` shows the installed package\n- [ ] Running `python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" --help` or an equivalent local command produces normal output\n- [ ] The script auto-downloads `scripts/license.xml` if missing, then extracts the license key from the `<key>` field for authentication\n- [ ] The script uses the `scripts/` directory as the SDK resource path\n- [ ] The script recognizes all 10 target formats: `word`/`excel`/`ppt`/`html`/`rtf`/`image`/`txt`/`json`/`markdown`/`csv`\n- [ ] The script accepts both PDF and image files (`.png`/`.jpg`/`.jpeg`/`.bmp`/`.tif`/`.tiff`/`.gif`/`.webp`/`.tga`) as input\n- [ ] When `--enable-ocr` or `--enable-ai-layout` (enabled by default) is active and `documentai.model` is missing, the script auto-downloads the model\n- [ ] When `license.xml` cannot be obtained (download fails and no manual file exists) or authentication fails, a clear error is output rather than a silent failure\n- [ ] The 7 parameters that default to True can be disabled with `--no-xxx`\n- [ ] `--ocr-language` supports specifying multiple languages simultaneously\n- [ ] After a conversion using the trial License, the usage count increments\n- [ ] When the trial License reaches 200 conversions, the script refuses to convert and outputs a purchase link\n- [ ] When using a non-trial License, no usage limit applies\n- [ ] For image input, even if `--enable-ocr` is not passed, the script automatically enables OCR and prints a notice to `stderr`\n- [ ] For HTML output, even if `--page-layout-mode` is not passed, the script automatically uses `box` (box layout) and prints a notice to `stderr`\n- [ ] For HTML output, explicitly passing `--page-layout-mode flow` overrides the automatic box layout behavior\n\n## Distribution Notes\n- This Skill does not depend on any machine-specific absolute paths.\n- When distributing to other users, the following directory structure is sufficient:\n  ```text\n  pdf-to-word-docx/\n  ├── SKILL.md\n  ├── License.txt\n  └── scripts/\n      └── pdf-to-word-docx.py\n  ```\n- Users place this directory under their own skills root directory and the Skill is ready to use.\n- `license.xml` is auto-downloaded at runtime; no need to include it in the distribution package.\n\n## Common Pitfalls\n- `scripts/license.xml` is missing and cannot be auto-downloaded (network unavailable or server error): the script will error out before authentication. If you are in an offline environment, place `license.xml` manually in the `scripts/` directory.\n- `scripts/license.xml` is missing the `<key>` field or its value is empty: the script will error out before authentication.\n- SDK resource files required by the SDK are absent from the `scripts/` directory: conversion may fail after `LibraryManager.initialize()`.\n- A password-protected PDF is provided without `--password`: this will trigger `PDF_PASSWORD_ERROR`.\n- OCR / AI layout is enabled but `documentai.model` is not present locally and the network is unavailable: the model download will fail; place the file in the `scripts/` directory manually in advance.\n- When the Excel output strategy is unclear, prefer passing `--excel-worksheet-option` explicitly to avoid unexpected result structures.\n- When converting images to other formats, the script already enables OCR automatically; if the output still contains no text, check whether `documentai.model` is complete and whether the OCR language matches.\n- Once the trial License usage limit is exhausted, a full License must be purchased to continue; purchase link: `https://www.compdf.com/contact-sales`.\n\n## Copyright\n\nThis Skill is built on top of the [ComPDFKit Conversion SDK](https://www.compdf.com).\n\n```\n© 2014-2026 PDF Technologies, Inc., a KDAN Company. All Rights Reserved.\n```\n\n- **SDK Name**: ComPDFKitConversion\n- **SDK Author**: PDF Technologies, Inc.\n- **License Type**: Commercial License (Commercial / Proprietary) — non-exclusive, non-transferable, non-sublicensable, revocable\n- **Official Website**: https://www.compdf.com\n- **Contact**: support@compdf.com\n- **Terms of Service**: https://www.compdf.com/terms-of-service\n- **Privacy Policy**: https://www.compdf.com/privacy-policy\n\n> **Important**: Under the ComPDFKit Terms of Service, distributing the documentation, sample code, or source code of the ComPDFKit Conversion SDK to third parties is prohibited. Please ensure you have obtained a valid ComPDFKit License before using this Skill.\n\nFile v1.1.0:_meta.json\n\n{\n  \"ownerId\": \"kn75g40xmv3tgkb21e079d055n82v5h8\",\n  \"slug\": \"pdf-to-word-docx\",\n  \"version\": \"1.1.0\",\n  \"publishedAt\": 1777464121232\n}\n\nFile v1.1.0:skill-card.md\n\n## Description: <br>\nPDF to Word Converter converts PDFs and image files into editable Word/DOCX and other document formats using the ComPDFKit Conversion SDK, with AI layout analysis, OCR, and table recognition to help preserve document structure. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[compdf-youna](https://clawhub.ai/user/compdf-youna) <br>\n\n### License/Terms of Use: <br>\nMIT-0 for the skill release; ComPDFKit Conversion SDK commercial/proprietary terms apply <br>\n\n\n## Use Case: <br>\nDevelopers, document operations teams, and agent users use this skill to convert PDFs or images into editable DOCX and related office/document formats while preserving tables, images, annotations, and page layout where possible. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill depends on the proprietary ComPDFKitConversion SDK and may download license.xml and a large documentai.model file from compdf.com on first run. <br>\nMitigation: Install only in environments where proprietary SDK use and first-run downloads from compdf.com are approved; pre-stage the license or model files when network access is restricted. <br>\nRisk: Protected-PDF passwords passed on the command line may appear in shared shell history, logs, or agent transcripts. <br>\nMitigation: Run password-protected conversions in a controlled session and avoid exposing passwords in shared transcripts or retained shell history. <br>\nRisk: The converter writes local output files and can create output folders automatically. <br>\nMitigation: Confirm input and output paths before execution and review generated files before using them downstream. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill listing](https://clawhub.ai/compdf-youna/pdf-to-word-docx) <br>\n- [ComPDF homepage](https://www.compdf.com) <br>\n- [ComPDF Conversion SDK Python overview](https://www.compdf.com/guides/conversion-sdk/python/overview) <br>\n- [ComPDF PDF to Word Python guide](https://www.compdf.com/guides/conversion-sdk/python/pdf-to-word) <br>\n- [ComPDF apply license guide](https://www.compdf.com/guides/conversion-sdk/python/apply-license) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, code, shell commands, configuration, guidance, files] <br>\n**Output Format:** [Markdown guidance and shell commands that produce converted files such as DOCX, XLSX, PPTX, HTML, RTF, images, TXT, JSON, Markdown, or CSV.] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Supports source format, output path, password, page ranges, OCR, AI layout, language, and format-specific conversion options.] <br>\n\n## Skill Version(s): <br>\n1.1.0 (source: SKILL.md frontmatter and server release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.1.0:License.txt\n\npdf to word - License Notice\n============================\n\nCopyright (c) 2014-2026 PDF Technologies, Inc., a KDAN Company.\nAll Rights Reserved.\n\nThis tool is built on the ComPDFKit Conversion SDK, a commercial product\nof PDF Technologies, Inc.\n\n------------------------------------------------------------------------\n1. LICENSE TYPE\n------------------------------------------------------------------------\n\nThe ComPDFKit Conversion SDK is provided under a COMMERCIAL / PROPRIETARY\nlicense. The license is non-exclusive, non-transferable, non-sublicensable,\nand revocable.\n\nYou must obtain a valid license key from PDF Technologies, Inc. before\nusing this software. A trial license is included for evaluation purposes\nonly, limited to 200 conversions.\n\n------------------------------------------------------------------------\n2. USAGE RESTRICTIONS\n------------------------------------------------------------------------\n\nBy using this software, you agree to the ComPDFKit Terms of Service:\nhttps://www.compdf.com/terms-of-service\n\nIn particular:\n\n  - You may NOT distribute, share, or disclose any documents, sample code,\n    or source code of the ComPDFKit Conversion SDK to any third party.\n\n  - You may NOT reverse-engineer, decompile, or disassemble the SDK\n    binaries.\n\n  - You may NOT use the SDK for any unlawful purpose or in violation of\n    applicable laws.\n\n  - The trial license is for evaluation only and must not be used in\n    production environments.\n\n------------------------------------------------------------------------\n3. DISCLAIMER OF WARRANTIES\n------------------------------------------------------------------------\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS\nOR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,\nFITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.\n\nIN NO EVENT SHALL PDF TECHNOLOGIES, INC. OR ITS AFFILIATES BE LIABLE FOR\nANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT,\nTORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE\nOR THE USE OR OTHER DEALINGS IN THE SOFTWARE.\n\n------------------------------------------------------------------------\n4. CONTACT & PURCHASE\n------------------------------------------------------------------------\n\n  Website:          https://www.compdf.com\n  Purchase License: https://www.compdf.com/contact-sales\n  Email:            support@compdf.com\n  Terms of Service: https://www.compdf.com/terms-of-service\n  Privacy Policy:   https://www.compdf.com/privacy-policy\n\nArchive v1.0.0: 4 files, 14395 bytes\n\nFiles: License.txt (2543b), scripts/pdf-to-word-docx.py (19881b), SKILL.md (23221b), _meta.json (135b)\n\nFile v1.0.0:SKILL.md\n\n---\nname: pdf-to-word-docx\nversion: 1.0.0\ndescription: PDF conversion toolkit featuring AI layout analysis and OCR. Converts PDFs to Word/Docx, Markdown, JSON, PPT, CSV, HTML, and XML for seamless LLM data processing.\nhomepage: https://www.compdf.com\nmetadata:\n  clawdbot:\n    emoji: \"📑\"\n    requires:\n      env: []\n    files: [\"scripts/*\"]\ncompatibility: Requires Windows or macOS. Python with ComPDFKitConversion package (pip install ComPDFKitConversion). AI model (~525MB) auto-downloaded on first run.\n---\n\n# pdf to word\n\n## Purpose\n- Wraps the `ComPDFKitConversion` Python SDK into a reusable local conversion workflow, supporting PDF / image to Word, PPT, Excel, HTML, RTF, Image, TXT, JSON, Markdown, and CSV (10 output formats in total).\n\n## Agent Skills Standard Compatibility\n- This Skill uses an Anthropic Agent Skills-compatible directory structure: `pdf-to-word-docx/`.\n- The entry point is `SKILL.md`; helper scripts are placed in `scripts/`.\n- The document uses `$ARGUMENTS` and `${CLAUDE_SKILL_DIR}` conventions for distribution and execution in Claude Code / Agent Skills-compatible environments.\n\n## Input / Output\n- Input: The target format (`word`/`excel`/`ppt`/`html`/`rtf`/`image`/`txt`/`json`/`markdown`/`csv`), the PDF or image path, and the output path are passed via Skill arguments or the command line. An optional PDF password and conversion parameters may also be provided.\n- Supported input file types:\n  - PDF files (`.pdf`)\n  - Image files (`.jpg`/`.jpeg`/`.png`/`.bmp`/`.tif`/`.tiff`/`.webp`/`.jp2`/`.gif`/`.tga`)\n- Output: A file in the corresponding format (`.docx`, `.pptx`, `.xlsx`, `.html`, `.rtf`, image, `.txt`, `.json`, `.md`, `.csv`), or a clear error message.\n\n## Prerequisites\n- Supports Windows and macOS.\n- The conversion SDK must be installed first:\n  ```bash\n  pip install ComPDFKitConversion\n  ```\n- On first run, the script automatically downloads `license.xml` from the ComPDF server and caches it in the `scripts/` directory:\n  ```text\n  https://download.compdf.com/skills/license/license.xml\n  ```\n- The script reads the `<key>...</key>` field from `license.xml` and uses that key for `LibraryManager.license_verify(...)` authentication — it does not pass the XML file path directly to the SDK.\n- To use a custom license, place your own `license.xml` in the `scripts/` directory; the script will use it directly without downloading.\n- During SDK initialization, the `resource` directory is always set to the directory containing `pdf-to-word-docx.py`, i.e., the `scripts/` directory itself.\n- When `--enable-ocr` or `--enable-ai-layout` (enabled by default) is used, the Skill also requires `scripts/documentai.model`. If the file does not exist, the script will automatically download it from:\n  ```text\n  https://download.compdf.com/skills/model/documentai.model\n  ```\n- To reuse an existing model file, you can override the default model path via an environment variable:\n  ```bash\n  export COMPDF_DOCUMENT_AI_MODEL=\"/path/to/documentai.model\"\n  ```\n\n## Workflow\n1. Confirm the Python package is installed:\n   ```bash\n   python -m pip show ComPDFKitConversion\n   ```\n2. The script automatically downloads `license.xml` on first run; the `scripts/` directory is used directly as the SDK `resource` path.\n3. In Agent Skills / Claude Code environments, prefer using the Skill's built-in script path variable:\n   ```bash\n   python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.pdf output.docx\n   python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" ppt input.pdf output.pptx\n   python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" excel input.pdf output.xlsx\n   ```\n4. For more control, append common parameters:\n   ```bash\n   python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" excel input.pdf output.xlsx --page-ranges \"1-3,5\" --excel-all-content --excel-worksheet-option for-page\n   python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.pdf output.docx --enable-ocr --page-layout-mode flow\n   ```\n5. On startup, the script ensures `scripts/license.xml` exists (downloading it automatically from the ComPDF server if missing), reads the `<key>` field for SDK authentication, and uses the `scripts/` directory as the `resource` path.\n6. If `--enable-ocr` or `--enable-ai-layout` (enabled by default) is active, the script checks whether `scripts/documentai.model` exists; if not, it downloads the file automatically before initializing the Document AI model.\n7. Check the return code; if it is not `SUCCESS`, handle license, password, resource, model, or input file issues according to the error name.\n\n## documentai.model Download Optimization\n- The script preferentially uses the model file pointed to by `COMPDF_DOCUMENT_AI_MODEL`.\n- The default model path is `scripts/documentai.model`.\n- During automatic download, the file is first written to `documentai.model.part` and then atomically renamed to the final file upon success, preventing partial file corruption.\n- On download failure, the script retries automatically with back-off intervals of `2s / 5s / 10s`.\n\n## Invoking Directly as a Skill\n- In environments that support Agent Skills, the Skill can be called directly:\n  ```text\n  /pdf-to-word-docx word input.pdf output.docx\n  /pdf-to-word-docx excel input.pdf output.xlsx --excel-worksheet-option for-page\n  ```\n- When the Skill receives arguments, it passes them through to the script as-is:\n  ```bash\n  python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" $ARGUMENTS\n  ```\n- If the environment does not support direct Skill invocation, fall back to a regular command-line call.\n\n## Supported Output Formats\n- `word` → calls `CPDFConversion.start_pdf_to_word`\n- `excel` → calls `CPDFConversion.start_pdf_to_excel`\n- `ppt` → calls `CPDFConversion.start_pdf_to_ppt`\n- `html` → calls `CPDFConversion.start_pdf_to_html`\n- `rtf` → calls `CPDFConversion.start_pdf_to_rtf`\n- `image` → calls `CPDFConversion.start_pdf_to_image`\n- `txt` → calls `CPDFConversion.start_pdf_to_txt`\n- `json` → calls `CPDFConversion.start_pdf_to_json`\n- `markdown` → calls `CPDFConversion.start_pdf_to_markdown`\n- `csv` → reuses `CPDFConversion.start_pdf_to_excel` with table/Excel parameters to produce CSV-friendly output\n\n## Input Source Types\n- The script supports **PDF and image** as input sources. The SDK's `start_pdf_to_*` interfaces natively accept image files with no pre-processing required.\n- By default, the script auto-detects the input type from the file extension:\n  - `.pdf` → `pdf`\n  - `.png/.jpg/.jpeg/.bmp/.tif/.tiff/.gif/.webp/.tga` → `image`\n- You can also specify the source type explicitly:\n  ```bash\n  python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.png output.docx --source-type image\n  ```\n- `image -> *` and `pdf -> *` share the same set of `CPDFConversion.start_pdf_to_*` interfaces; only the input file type differs.\n\n## Smart Defaults\nThe script automatically adjusts certain parameters based on the input source and output format to reduce manual configuration:\n\n| Trigger | Automatic Behavior | User-Overridable | Description |\n|----------|----------|-------------|------|\n| Input source is an **image** (auto-detected or explicit `--source-type image`) | Automatically enables `--enable-ocr` | No (`--enable-ocr` uses `store_true`; there is no `--no-enable-ocr`) | Text in images must be extracted via OCR; without OCR, output will contain only images and no text |\n| Output format is **HTML** (`format = html`) | Automatically sets `--page-layout-mode` to `box` (box layout) | Yes — passing `--page-layout-mode flow` explicitly overrides this | Box layout better preserves the original formatting in HTML; specify `flow` explicitly if flow layout is needed |\n\nWhen triggered, the script prints a notice to `stderr`, for example:\n```text\nAuto-enabled OCR for image input.\nAuto-set page layout mode to BOX for HTML output.\n```\n\n## All Parameters\n\n### Positional Parameters\n| Parameter | Description |\n|------|------|\n| `format` | Target format: `word`/`excel`/`ppt`/`html`/`rtf`/`image`/`txt`/`json`/`markdown`/`csv` |\n| `input_pdf` | Input file path (PDF or image) |\n| `output_path` | Output file path |\n\n### General Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--source-type` | Option | `auto` | Input source type: `auto`/`pdf`/`image` |\n| `--password` | String | `\"\"` | PDF open password |\n| `--page-ranges` | String | None | Page range, e.g. `1-3,5` |\n| `--font-name` | String | `\"\"` | Output font name |\n\n### Layout Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--enable-ai-layout` | Boolean | **True** | AI layout analysis (disable with `--no-enable-ai-layout`) |\n| `--page-layout-mode` | Option | SDK default `flow` (auto-switched to `box` for HTML output) | Page layout: `box` (box layout) / `flow` (flow layout) |\n\n### Content Retention Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--contain-image` | Boolean | **True** | Retain images (disable with `--no-contain-image`) |\n| `--contain-annotation` | Boolean | **True** | Retain annotations (disable with `--no-contain-annotation`) |\n| `--contain-page-background-image` | Boolean | **True** | Retain page background images (disable with `--no-contain-page-background-image`) |\n| `--formula-to-image` | Boolean | False | Convert formulas to image output |\n| `--transparent-text` | Boolean | False | Preserve transparent text |\n\n### Output Control Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--output-document-per-page` | Boolean | False | Split output into one document per page |\n| `--auto-create-folder` | Boolean | **True** | Automatically create output directory (disable with `--no-auto-create-folder`) |\n\n### OCR Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--enable-ocr` | Boolean | False (auto-enabled for image input) | Enable OCR |\n| `--ocr-option` | Option | SDK default `all` | OCR scope: `invalid-character`/`scan-page`/`invalid-character-and-scan-page`/`all` |\n| `--ocr-language` | Multi-select | `auto` | OCR language(s); multiple languages can be specified simultaneously. Options: `auto`/`chinese`/`chinese-tra`/`english`/`korean`/`japanese`/`latin`/`devanagari`/`cyrillic`/`arabic`/`tamil`/`telugu`/`kannada`/`thai`/`greek`/`eslav` |\n\n### Excel-Specific Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--excel-all-content` | Boolean | False | Include all content in Excel output |\n| `--excel-csv-format` | Boolean | False | Output Excel result in CSV format |\n| `--excel-worksheet-option` | Option | SDK default `for-table` | Worksheet split strategy: `for-table`/`for-page`/`for-document` |\n\n### JSON-Specific Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--json-contain-table` | Boolean | **True** | Include table data in JSON output (disable with `--no-json-contain-table`) |\n\n### TXT-Specific Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--txt-table-format` | Boolean | **True** | Enable table formatting in TXT output (disable with `--no-txt-table-format`) |\n\n### HTML-Specific Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--html-option` | Option | SDK default `single-page` | HTML output mode: `single-page`/`single-page-with-bookmark`/`multiple-page`/`multiple-page-with-bookmark` |\n\n### Image-Specific Parameters\n| Parameter | Type | Default | Description |\n|------|------|--------|------|\n| `--image-type` | Option | SDK default `jpg` | Image output format: `jpg`/`jpeg`/`jpeg2000`/`png`/`bmp`/`tiff`/`tga`/`gif`/`webp` |\n| `--image-color-mode` | Option | SDK default `color` | Image color mode: `color`/`gray`/`binary` |\n| `--image-scaling` | Float | `1.0` | Image scaling factor |\n| `--image-path-enhance` | Boolean | False | Enable image path enhancement |\n\n### Parameter Default Value Rules\n- **Parameters that default to True** (`--enable-ai-layout`/`--contain-image`/`--contain-annotation`/`--contain-page-background-image`/`--auto-create-folder`/`--json-contain-table`/`--txt-table-format`) use `BooleanOptionalAction`; pass `--no-xxx` to disable.\n- **Parameters that default to False** (`--enable-ocr`/`--formula-to-image`/`--transparent-text`/`--output-document-per-page`/`--excel-all-content`/`--excel-csv-format`/`--image-path-enhance`) use `store_true`; passing the flag enables them.\n- **All CLI parameter defaults are fully consistent with the SDK's `ConvertOptions()` defaults** — omitting a parameter is equivalent to using the SDK's original default value.\n\n## Recommended Command Examples\n\n### PDF to Word (default parameters, AI layout analysis enabled)\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.pdf output.docx\n```\n\n### PDF to Word, box layout, no images, no AI layout analysis\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.pdf output.docx --no-enable-ai-layout --no-contain-image --page-layout-mode box\n```\n\n### PDF to Word, retain annotations and background images, one document per page\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.pdf output.docx --output-document-per-page\n```\n\n### PDF to Excel, include all content and split worksheets by page\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" excel input.pdf output.xlsx --excel-all-content --excel-worksheet-option for-page\n```\n\n### PDF to TXT, with table formatting enabled\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" txt input.pdf output.txt\n```\n\n### PDF to HTML, multi-page with bookmarks mode\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" html input.pdf output_dir --html-option multiple-page-with-bookmark\n```\n\n### PDF to Image, PNG format, grayscale, 2x scaling\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" image input.pdf output.png --image-type png --image-color-mode gray --image-scaling 2.0\n```\n\n### Image to Word (OCR auto-enabled, specify Chinese language)\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.png output.docx --ocr-language chinese\n```\n> Note: For image input, the script automatically enables OCR — there is no need to pass `--enable-ocr` manually. To specify an OCR language, `--ocr-language` can still be used.\n\n### PDF with OCR enabled (multiple languages)\n```bash\npython \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.pdf output.docx --enable-ocr --ocr-language chinese english japanese\n```\n\n## Trial License and Usage Limits\n- The `scripts/license.xml` auto-downloaded from the ComPDF server is a **Trial License**, allowing a maximum of **200 conversions**.\n- The script uses a SHA-256 fingerprint to detect whether the current License is the default trial key; **no usage limit applies when using any other License**.\n- After each successful conversion using the trial License, the script prints the current used/remaining count to `stderr`, for example:\n  ```text\n  Trial license: 5/200 conversions used, 195 remaining.\n  ```\n- When the trial limit is reached (200 conversions), the script refuses to convert and prompts the user to purchase a full License:\n  ```text\n  Error: Trial license usage limit reached (200 conversions). Please purchase a license at: https://www.compdf.com/contact-sales\n  ```\n- When the trial License has expired (SDK authentication fails), the error message also includes a purchase link.\n- **After purchasing a full License**, place a custom `license.xml` containing the new `<key>` in `scripts/` (overwriting the auto-downloaded trial file) — no script modifications or counter file cleanup are required.\n\n## Confirmed Facts\n- `ComPDFKitConversion 3.9.0` has been successfully installed on the local machine.\n- The installed package provides 10 conversion methods including `CPDFConversion.start_pdf_to_word/start_pdf_to_ppt/start_pdf_to_excel`.\n- `LibraryManager` provides `initialize`, `license_verify`, `release`, `set_document_ai_model`, and `set_ocr_language`.\n- Official documentation confirms support for PDF to Word / Excel / PPT / HTML / RTF / Image / TXT / JSON / Markdown.\n- The SDK's `start_pdf_to_*` interfaces natively accept image file input (PNG → Word has been verified successfully).\n- `enable_ai_layout` defaults to `True` in the SDK; `set_document_ai_model()` must be called first to load the model before use, otherwise a 0xC0000005 crash will occur.\n- `--ocr-language` supports specifying multiple languages simultaneously (e.g. `--ocr-language chinese english`).\n\n## Risks / Notes\n- The official requirements page states Python `>=3.6`, while the demo page states `<3.11`, but PyPI currently provides a `cp314` wheel in practice; treat the locally installable wheel as the source of truth, but always verify installation in a new environment first.\n- If the script cannot download `license.xml` from the server (network issue) and no manual file exists in `scripts/`, or the `<key>` field is empty, the script cannot complete SDK authentication and cannot perform any real conversions.\n- `documentai.model` is a large file (approximately 525 MB); there will be a noticeable download delay the first time OCR / AI layout is enabled. Because `--enable-ai-layout` defaults to True, **the model download will be triggered on the very first run**.\n- If the runtime environment cannot access `https://download.compdf.com/skills/model/documentai.model`, place `documentai.model` in the `scripts/` directory in advance.\n- Do not directly apply the initialization patterns from ComPDF SDKs for other languages to the Python package; this Skill is based on the locally verified `LibraryManager` / `CPDFConversion` API.\n\n## Resource Navigation\n- License file: `License.txt`\n- Script: `scripts/pdf-to-word-docx.py`\n- SDK authentication file: `scripts/license.xml` (auto-downloaded from `https://download.compdf.com/skills/license/license.xml` if missing)\n- SDK authentication source: the `<key>` field in `license.xml`\n- SDK resource path: `scripts/`\n- OCR / AI layout model: `scripts/documentai.model` (auto-downloaded if missing)\n- Purchase a full License: `https://www.compdf.com/contact-sales`\n- Official documentation:\n  - `https://www.compdf.com/guides/conversion-sdk/python/overview`\n  - `https://www.compdf.com/guides/conversion-sdk/python/pdf-to-word`\n  - `https://www.compdf.com/guides/conversion-sdk/python/pdf-to-excel`\n  - `https://www.compdf.com/guides/conversion-sdk/python/pdf-to-ppt`\n  - `https://www.compdf.com/guides/conversion-sdk/python/apply-license`\n\n## Acceptance Checklist\n- [ ] `python -m pip show ComPDFKitConversion` shows the installed package\n- [ ] Running `python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" --help` or an equivalent local command produces normal output\n- [ ] The script auto-downloads `scripts/license.xml` if missing, then extracts the license key from the `<key>` field for authentication\n- [ ] The script uses the `scripts/` directory as the SDK resource path\n- [ ] The script recognizes all 10 target formats: `word`/`excel`/`ppt`/`html`/`rtf`/`image`/`txt`/`json`/`markdown`/`csv`\n- [ ] The script accepts both PDF and image files (`.png`/`.jpg`/`.jpeg`/`.bmp`/`.tif`/`.tiff`/`.gif`/`.webp`/`.tga`) as input\n- [ ] When `--enable-ocr` or `--enable-ai-layout` (enabled by default) is active and `documentai.model` is missing, the script auto-downloads the model\n- [ ] When `license.xml` cannot be obtained (download fails and no manual file exists) or authentication fails, a clear error is output rather than a silent failure\n- [ ] The 7 parameters that default to True can be disabled with `--no-xxx`\n- [ ] `--ocr-language` supports specifying multiple languages simultaneously\n- [ ] After a conversion using the trial License, the usage count increments\n- [ ] When the trial License reaches 200 conversions, the script refuses to convert and outputs a purchase link\n- [ ] When using a non-trial License, no usage limit applies\n- [ ] For image input, even if `--enable-ocr` is not passed, the script automatically enables OCR and prints a notice to `stderr`\n- [ ] For HTML output, even if `--page-layout-mode` is not passed, the script automatically uses `box` (box layout) and prints a notice to `stderr`\n- [ ] For HTML output, explicitly passing `--page-layout-mode flow` overrides the automatic box layout behavior\n\n## Distribution Notes\n- This Skill does not depend on any machine-specific absolute paths.\n- When distributing to other users, the following directory structure is sufficient:\n  ```text\n  pdf-to-word-docx/\n  ├── SKILL.md\n  ├── License.txt\n  └── scripts/\n      └── pdf-to-word-docx.py\n  ```\n- Users place this directory under their own skills root directory and the Skill is ready to use.\n- `license.xml` is auto-downloaded at runtime; no need to include it in the distribution package.\n\n## Common Pitfalls\n- `scripts/license.xml` is missing and cannot be auto-downloaded (network unavailable or server error): the script will error out before authentication. If you are in an offline environment, place `license.xml` manually in the `scripts/` directory.\n- `scripts/license.xml` is missing the `<key>` field or its value is empty: the script will error out before authentication.\n- SDK resource files required by the SDK are absent from the `scripts/` directory: conversion may fail after `LibraryManager.initialize()`.\n- A password-protected PDF is provided without `--password`: this will trigger `PDF_PASSWORD_ERROR`.\n- OCR / AI layout is enabled but `documentai.model` is not present locally and the network is unavailable: the model download will fail; place the file in the `scripts/` directory manually in advance.\n- When the Excel output strategy is unclear, prefer passing `--excel-worksheet-option` explicitly to avoid unexpected result structures.\n- When converting images to other formats, the script already enables OCR automatically; if the output still contains no text, check whether `documentai.model` is complete and whether the OCR language matches.\n- Once the trial License usage limit is exhausted, a full License must be purchased to continue; purchase link: `https://www.compdf.com/contact-sales`.\n\n## Copyright\n\nThis Skill is built on top of the [ComPDFKit Conversion SDK](https://www.compdf.com).\n\n```\n© 2014-2026 PDF Technologies, Inc., a KDAN Company. All Rights Reserved.\n```\n\n- **SDK Name**: ComPDFKitConversion\n- **SDK Author**: PDF Technologies, Inc.\n- **License Type**: Commercial License (Commercial / Proprietary) — non-exclusive, non-transferable, non-sublicensable, revocable\n- **Official Website**: https://www.compdf.com\n- **Contact**: support@compdf.com\n- **Terms of Service**: https://www.compdf.com/terms-of-service\n- **Privacy Policy**: https://www.compdf.com/privacy-policy\n\n> **Important**: Under the ComPDFKit Terms of Service, distributing the documentation, sample code, or source code of the ComPDFKit Conversion SDK to third parties is prohibited. Please ensure you have obtained a valid ComPDFKit License before using this Skill.\n\nFile v1.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn75g40xmv3tgkb21e079d055n82v5h8\",\n  \"slug\": \"pdf-to-word-docx\",\n  \"version\": \"1.0.0\",\n  \"publishedAt\": 1773918383211\n}\n\nFile v1.0.0:License.txt\n\npdf to word - License Notice\n============================\n\nCopyright (c) 2014-2026 PDF Technologies, Inc., a KDAN Company.\nAll Rights Reserved.\n\nThis tool is built on the ComPDFKit Conversion SDK, a commercial product\nof PDF Technologies, Inc.\n\n------------------------------------------------------------------------\n1. LICENSE TYPE\n------------------------------------------------------------------------\n\nThe ComPDFKit Conversion SDK is provided under a COMMERCIAL / PROPRIETARY\nlicense. The license is non-exclusive, non-transferable, non-sublicensable,\nand revocable.\n\nYou must obtain a valid license key from PDF Technologies, Inc. before\nusing this software. A trial license is included for evaluation purposes\nonly, limited to 200 conversions.\n\n------------------------------------------------------------------------\n2. USAGE RESTRICTIONS\n------------------------------------------------------------------------\n\nBy using this software, you agree to the ComPDFKit Terms of Service:\nhttps://www.compdf.com/terms-of-service\n\nIn particular:\n\n  - You may NOT distribute, share, or disclose any documents, sample code,\n    or source code of the ComPDFKit Conversion SDK to any third party.\n\n  - You may NOT reverse-engineer, decompile, or disassemble the SDK\n    binaries.\n\n  - You may NOT use the SDK for any unlawful purpose or in violation of\n    applicable laws.\n\n  - The trial license is for evaluation only and must not be used in\n    production environments.\n\n------------------------------------------------------------------------\n3. DISCLAIMER OF WARRANTIES\n------------------------------------------------------------------------\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS\nOR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,\nFITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.\n\nIN NO EVENT SHALL PDF TECHNOLOGIES, INC. OR ITS AFFILIATES BE LIABLE FOR\nANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT,\nTORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE\nOR THE USE OR OTHER DEALINGS IN THE SOFTWARE.\n\n------------------------------------------------------------------------\n4. CONTACT & PURCHASE\n------------------------------------------------------------------------\n\n  Website:          https://www.compdf.com\n  Purchase License: https://www.compdf.com/contact-sales\n  Email:            support@compdf.com\n  Terms of Service: https://www.compdf.com/terms-of-service\n  Privacy Policy:   https://www.compdf.com/privacy-policy","readmeExcerpt":"Skill: PDF to Word Converter Owner: compdf-youna Summary: PDF to Word converts PDF to editable Word/DOCX with AI-powered layout analysis and table recognition, built on ComPDF Conversion SDK to better preserve table... Tags: latest:1.2.0 Version history: v1.2.0 | 2026-06-24T03:46:44.581Z | user pdf-to-word-docx 1.2.0 changelog - Updated homepage URL to include UTM parameters for improved source tracking. - Removed th","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"pip install ComPDFKitConversion"},{"language":"text","snippet":"https://download.compdf.com/skills/license/license.xml"},{"language":"text","snippet":"https://download.compdf.com/skills/model/documentai.model"},{"language":"bash","snippet":"export COMPDF_DOCUMENT_AI_MODEL=\"/path/to/documentai.model\""},{"language":"bash","snippet":"python -m pip show ComPDFKitConversion"},{"language":"bash","snippet":"python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" word input.pdf output.docx\n   python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" ppt input.pdf output.pptx\n   python \"${CLAUDE_SKILL_DIR}/scripts/pdf-to-word-docx.py\" excel input.pdf output.xlsx"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: pdf-to-word-docx\nversion: 1.2.0\ndescription: PDF to Word converts PDF to editable Word/DOCX with AI-powered layout analysis and table recognition, built on ComPDF Conversion SDK to better preserve tables, multi-column layouts, lists, and images for downstream editing. It fits requests such as “pdf to word,” “convert pdf to docx,” “pdf to editable word,” “pdf to office,” “keep layout in word,” and “convert report to docx.” Example queries include “Convert this PDF contract to editable Word while keeping the tables intact,” “Turn this report into DOCX and preserve the multi-column layout,” and “Export this PDF to Word for further editing.”\nhomepage: https://www.compdf.com/?utm_source=clawhub&utm_medium=skillhub&utm_campaign=compdf_pdf_skill_to_word&ref_platform_id=clawhub_skills\nmetadata:\n  clawdbot:\n    emoji: \"📑\"\n    requires:\n      env: []\n    files: [\"scripts/*\"]\ncompatibility: Requires Windows or macOS. Python with ComPDFKitConversion package (pip install ComPDFKitConversion). AI model (~525MB) auto-downloaded on first run.\n---\n\n# PDF to Word Converter\n\n## Purpose\n- Wraps the `ComPDFKitConversion` Python SDK into a reusable local conversion workflow, supporting PDF / image to Word, PPT, Excel, HTML, RTF, Image, TXT, JSON, Markdown, and CSV (10 output formats in total).\n\n## Agent Skills Standard Compatibility\n- This Skill uses an Anthropic Agent Skills-compatible directory structure: `pdf-to-word-docx/`.\n- The entry point is `SKILL.md`; helper scripts are placed in `scripts/`.\n- The document uses `$ARGUMENTS` and `${CLAUDE_SKILL_DIR}` conventions for distribution and execution in Claude Code / Agent Skills-compatible environments.\n\n## Input / Output\n- Input: The target format (`word`/`excel`/`ppt`/`html`/`rtf`/`image`/`txt`/`json`/`markdown`/`csv`), the PDF or image path, and the output path are passed via Skill arguments or the command line. An optional PDF password and conversion parameters may also be provided.\n- Supported input file types:\n  - PDF files (`.pdf`)\n  - Image files (`.jpg`/`.jpeg`/`.png`/`.bmp`/`.tif`/`.tiff`/`.webp`/`.jp2`/`.gif`/`.tga`)\n- Output: A file in the corresponding format (`.docx`, `.pptx`, `.xlsx`, `.html`, `.rtf`, image, `.txt`, `.json`, `.md`, `.csv`), or a clear error message.\n\n## Prerequisites\n- Supports Windows and macOS.\n- The conversion SDK must be installed first:\n  ```bash\n  pip install ComPDFKitConversion\n  ```\n- On first run, the script automatically downloads `license.xml` from the ComPDF server and caches it in the `scripts/` directory:\n  ```text\n  https://download.compdf.com/skills/license/license.xml\n  ```\n- The script reads the `<key>...</key>` field from `license.xml` and uses that key for `LibraryManager.license_verify(...)` authentication — it does not pass the XML file path directly to the SDK.\n- To use a custom license, place your own `license.xml` in the `scripts/` directory; the script will use it directly without downloading.\n- During SDK initialization, the `resource` directory is"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn75g40xmv3tgkb21e079d055n82v5h8\",\n  \"slug\": \"pdf-to-word-docx\",\n  \"version\": \"1.2.0\",\n  \"publishedAt\": 1782272804581\n}"},{"path":"skill-card.md","content":"## Description:\n\nPDF to Word converts PDF to editable Word/DOCX with AI-powered layout analysis and table recognition, built on ComPDF Conversion SDK to better preserve tables, multi-column layouts, lists, and images for downstream editing.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[compdf-youna](https://clawhub.ai/user/compdf-youna)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nExternal users, developers, and document automation agents use this skill to convert PDFs or image files into editable office and text formats while preserving layout, tables, lists, and images. It is suited for document editing, extraction, and format migration workflows that can run on Windows or macOS with the ComPDFKitConversion package.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill depends on ComPDF's proprietary SDK and requires license verification before conversion.\n\nMitigation: Review the ComPDFKitConversion package and license terms before use, and provide an approved license.xml for production workflows.\n\nRisk: First-run execution can download license.xml and a large documentai.model file from ComPDF when local copies are missing.\n\nMitigation: Run in a restricted workspace, pre-stage verified license and model files where possible, and allow network access only to expected ComPDF download locations.\n\nRisk: The trial license counter is stored in the user's home directory and may affect repeat conversions.\n\nMitigation: Use a full license for production or shared environments, and account for the local trial counter when testing.\n\nRisk: Sensitive documents are processed by a local SDK and may produce editable output files containing the original document content.\n\nMitigation: Convert sensitive files only in approved local workspaces with appropriate file access controls and output cleanup procedures.\n\n## Reference(s):\n\n- [ClawHub Skill Page](https://clawhub.ai/compdf-youna/skills/pdf-to-word-docx)\n- [ComPDF Homepage](https://www.compdf.com/?utm_source=clawhub&utm_medium=skillhub&utm_campaign=compdf_pdf_skill_to_word&ref_platform_id=clawhub_skills)\n- [ComPDF Conversion SDK for Python Overview](https://www.compdf.com/guides/conversion-sdk/python/overview)\n- [ComPDF PDF to Word Guide](https://www.compdf.com/guides/conversion-sdk/python/pdf-to-word)\n- [ComPDF PDF to Excel Guide](https://www.compdf.com/guides/conversion-sdk/python/pdf-to-excel)\n- [ComPDF PDF to PPT Guide](https://www.compdf.com/guides/conversion-sdk/python/pdf-to-ppt)\n- [ComPDF Apply License Guide](https://www.compdf.com/guides/conversion-sdk/python/apply-license)\n\n## Skill Output:\n\n**Output Type(s):** [Files, Text, Markdown, Shell commands, Configuration instructions]\n\n**Output Format:** [Converted document files with command-line status text and error messages]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Supports DOCX, PPTX, XLSX, HTML, RTF, image, TXT, JSON, Markdown, and CSV outputs from PDF o"},{"path":"License.txt","content":"pdf to word - License Notice\n============================\n\nCopyright (c) 2014-2026 PDF Technologies, Inc., a KDAN Company.\nAll Rights Reserved.\n\nThis tool is built on the ComPDFKit Conversion SDK, a commercial product\nof PDF Technologies, Inc.\n\n------------------------------------------------------------------------\n1. LICENSE TYPE\n------------------------------------------------------------------------\n\nThe ComPDFKit Conversion SDK is provided under a COMMERCIAL / PROPRIETARY\nlicense. The license is non-exclusive, non-transferable, non-sublicensable,\nand revocable.\n\nYou must obtain a valid license key from PDF Technologies, Inc. before\nusing this software. A trial license is included for evaluation purposes\nonly, limited to 200 conversions.\n\n------------------------------------------------------------------------\n2. USAGE RESTRICTIONS\n------------------------------------------------------------------------\n\nBy using this software, you agree to the ComPDFKit Terms of Service:\nhttps://www.compdf.com/terms-of-service/?utm_source=clawhub&utm_medium=skillhub&utm_campaign=compdf_pdf_skill_to_word&ref_platform_id=clawhub_skills\n\nIn particular:\n\n  - You may NOT distribute, share, or disclose any documents, sample code,\n    or source code of the ComPDFKit Conversion SDK to any third party.\n\n  - You may NOT reverse-engineer, decompile, or disassemble the SDK\n    binaries.\n\n  - You may NOT use the SDK for any unlawful purpose or in violation of\n    applicable laws.\n\n  - The trial license is for evaluation only and must not be used in\n    production environments.\n\n------------------------------------------------------------------------\n3. DISCLAIMER OF WARRANTIES\n------------------------------------------------------------------------\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS\nOR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,\nFITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.\n\nIN NO EVENT SHALL PDF TECHNOLOGIES, INC. OR ITS AFFILIATES BE LIABLE FOR\nANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT,\nTORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE\nOR THE USE OR OTHER DEALINGS IN THE SOFTWARE.\n\n------------------------------------------------------------------------\n4. CONTACT & PURCHASE\n------------------------------------------------------------------------\n\n  Website: https://www.compdf.com/pdf-sdk?utm_source=clawhub&utm_medium=skillhub&utm_campaign=compdf_pdf_skill&ref_platform_id=clawhub_skills\n\n  Purchase License: https://www.compdf.com/contact-sales?utm_source=clawhub&utm_medium=skillhub&utm_campaign=pdf_skill_to_word&ref_platform_id=clawhub_skills\n\n  Email: support@compdf.com\n\n  Terms of Service: https://www.compdf.com/terms-of-service/?utm_source=clawhub&utm_medium=skillhub&utm_campaign=compdf_pdf_skill_to_word&ref_platform_id=clawhub_skills\n\n  Privacy Policy: https://www.compdf.com/privacy-policy/?utm_source=clawhub&utm_medium=skillhub&utm_camp"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":1721,"uniquenessScore":40,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T23:47:57.848Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T23:47:57.848Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T07:42:11.847Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}