{"id":"81395725-211a-4440-8b2f-16ed87fdaecb","entityType":"agent","slug":"clawhub-michealxie001-office-doc-extractor","name":"Office Document Extractor","canonicalUrl":"https://www.xpersona.co/agent/clawhub-michealxie001-office-doc-extractor","canonicalPath":"/agent/clawhub-michealxie001-office-doc-extractor","generatedAt":"2026-10-11T20:59:34.892Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-11T17:57:31.233Z","emptyReason":null},"description":"Convert Microsoft Office documents (DOCX, XLSX, PPTX) to Markdown without any external dependencies. Use when the user needs to extract text from Word docume... Skill: Office Document Extractor Owner: michealxie001 Summary: Convert Microsoft Office documents (DOCX, XLSX, PPTX) to Markdown without any external dependencies. Use when the user needs to extract text from Word docume... Tags: converter:1.0.0, docx:1.0.0, latest:1.0.1, markdown:1.0.0, office:1.0.0, pptx:1.0.0, python:1.0.0, xlsx:1.0.0 Version history: v1.0.1 | 2026-05-04T10:08:40.701Z | user Fix: Removed pycache,","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s170w0cbg2mpaewxv2748qq5cx83hwge:office-doc-extractor","sourceUrl":"https://clawhub.ai/michealxie001/office-doc-extractor","homepage":"https://clawhub.ai/michealxie001/skills/office-doc-extractor","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/michealxie001/office-doc-extractor","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/michealxie001/skills/office-doc-extractor","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":60,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Convert Microsoft Office documents (DOCX, XLSX, PPTX) to Markdown without any external dependencies. Use when the user needs to extract text from Word docume..."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T17:57:31.233Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T17:57:31.233Z","emptyReason":null},"stars":null,"forks":null,"downloads":1014,"packageName":null,"latestVersion":"1.0.1","tractionLabel":"1K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T17:57:31.170Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T17:57:31.233Z","lastCrawledAt":"2026-10-11T17:57:31.170Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T17:57:31.170Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.1","createdAt":"2026-05-04T10:08:40.701Z","changelog":"Fix: Removed pycache, repackaged clean build","fileCount":200,"zipByteSize":266572},{"version":"1.0.0","createdAt":"2026-05-04T09:51:40.027Z","changelog":"- Initial release of office-doc-extractor: convert DOCX, XLSX, and PPTX files to Markdown using a pure Python, zero-dependency approach. - Supports extraction of text and structure: Word headings/paragraphs, Excel tables, and PowerPoint slides. - Works offline—no pip installs, subprocess calls, or network access required. - Includes unified CLI for both single-file and batch directory conversion. - Bundles pure Python openpyxl and et_xmlfile for Excel support.","fileCount":199,"zipByteSize":265376}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s170w0cbg2mpaewxv2748qq5cx83hwge:office-doc-extractor","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-michealxie001-office-doc-extractor/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-michealxie001-office-doc-extractor/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-michealxie001-office-doc-extractor/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-michealxie001-office-doc-extractor/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-michealxie001-office-doc-extractor/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-michealxie001-office-doc-extractor/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T20:59:34.891Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-michealxie001-office-doc-extractor/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-michealxie001-office-doc-extractor/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-michealxie001-office-doc-extractor/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-michealxie001-office-doc-extractor/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-11T17:57:31.233Z","emptyReason":null},"readme":"Skill: Office Document Extractor\n\nOwner: michealxie001\n\nSummary: Convert Microsoft Office documents (DOCX, XLSX, PPTX) to Markdown without any external dependencies. Use when the user needs to extract text from Word docume...\n\nTags: converter:1.0.0, docx:1.0.0, latest:1.0.1, markdown:1.0.0, office:1.0.0, pptx:1.0.0, python:1.0.0, xlsx:1.0.0\n\nVersion history:\n\nv1.0.1 | 2026-05-04T10:08:40.701Z | user\n\nFix: Removed pycache, repackaged clean build\n\nv1.0.0 | 2026-05-04T09:51:40.027Z | auto\n\n- Initial release of office-doc-extractor: convert DOCX, XLSX, and PPTX files to Markdown using a pure Python, zero-dependency approach.\n- Supports extraction of text and structure: Word headings/paragraphs, Excel tables, and PowerPoint slides.\n- Works offline—no pip installs, subprocess calls, or network access required.\n- Includes unified CLI for both single-file and batch directory conversion.\n- Bundles pure Python openpyxl and et_xmlfile for Excel support.\n\nArchive index:\n\nArchive v1.0.1: 200 files, 266572 bytes\n\nFiles: scripts/docx_extractor.py (2908b), scripts/et_xmlfile/__init__.py (228b), scripts/et_xmlfile/incremental_tree.py (34534b), scripts/et_xmlfile/xmlfile.py (4886b), scripts/main.py (2974b), scripts/openpyxl/__init__.py (603b), scripts/openpyxl/_constants.py (306b), scripts/openpyxl/cell/__init__.py (122b), scripts/openpyxl/cell/_writer.py (4015b), scripts/openpyxl/cell/cell.py (8922b), scripts/openpyxl/cell/read_only.py (3097b), scripts/openpyxl/cell/rich_text.py (5628b), scripts/openpyxl/cell/text.py (4367b), scripts/openpyxl/chart/__init__.py (564b), scripts/openpyxl/chart/_3d.py (3104b), scripts/openpyxl/chart/_chart.py (5746b), scripts/openpyxl/chart/area_chart.py (2890b), scripts/openpyxl/chart/axis.py (12580b), scripts/openpyxl/chart/bar_chart.py (4142b), scripts/openpyxl/chart/bubble_chart.py (2004b), scripts/openpyxl/chart/chartspace.py (6069b), scripts/openpyxl/chart/data_source.py (5782b), scripts/openpyxl/chart/descriptors.py (736b), scripts/openpyxl/chart/error_bar.py (1832b), scripts/openpyxl/chart/label.py (4133b), scripts/openpyxl/chart/layout.py (2040b), scripts/openpyxl/chart/legend.py (2040b), scripts/openpyxl/chart/line_chart.py (3951b), scripts/openpyxl/chart/marker.py (2600b), scripts/openpyxl/chart/picture.py (1156b), scripts/openpyxl/chart/pie_chart.py (4793b), scripts/openpyxl/chart/pivot.py (1741b), scripts/openpyxl/chart/plotarea.py (5805b), scripts/openpyxl/chart/print_settings.py (1454b), scripts/openpyxl/chart/radar_chart.py (1521b), scripts/openpyxl/chart/reader.py (802b), scripts/openpyxl/chart/reference.py (3098b), scripts/openpyxl/chart/scatter_chart.py (1563b), scripts/openpyxl/chart/series_factory.py (1368b), scripts/openpyxl/chart/series.py (5896b), scripts/openpyxl/chart/shapes.py (2815b), scripts/openpyxl/chart/stock_chart.py (1604b), scripts/openpyxl/chart/surface_chart.py (2914b), scripts/openpyxl/chart/text.py (1847b), scripts/openpyxl/chart/title.py (1952b), scripts/openpyxl/chart/trendline.py (3045b), scripts/openpyxl/chart/updown_bars.py (897b), scripts/openpyxl/chartsheet/__init__.py (71b), scripts/openpyxl/chartsheet/chartsheet.py (3911b), scripts/openpyxl/chartsheet/custom.py (1691b), scripts/openpyxl/chartsheet/properties.py (679b), scripts/openpyxl/chartsheet/protection.py (1265b), scripts/openpyxl/chartsheet/publish.py (1587b), scripts/openpyxl/chartsheet/relation.py (2731b), scripts/openpyxl/chartsheet/views.py (1341b), scripts/openpyxl/comments/__init__.py (67b), scripts/openpyxl/comments/author.py (388b), scripts/openpyxl/comments/comment_sheet.py (5753b), scripts/openpyxl/comments/comments.py (1466b), scripts/openpyxl/comments/shape_writer.py (3809b), scripts/openpyxl/compat/__init__.py (1592b), scripts/openpyxl/compat/abc.py (155b), scripts/openpyxl/compat/numbers.py (1617b), scripts/openpyxl/compat/product.py (264b), scripts/openpyxl/compat/singleton.py (1023b), scripts/openpyxl/compat/strings.py (604b), scripts/openpyxl/descriptors/__init__.py (1952b), scripts/openpyxl/descriptors/base.py (7135b), scripts/openpyxl/descriptors/container.py (889b), scripts/openpyxl/descriptors/excel.py (2412b), scripts/openpyxl/descriptors/namespace.py (313b), scripts/openpyxl/descriptors/nested.py (2603b), scripts/openpyxl/descriptors/sequence.py (3490b), scripts/openpyxl/descriptors/serialisable.py (7361b), scripts/openpyxl/descriptors/slots.py (824b), scripts/openpyxl/drawing/__init__.py (66b), scripts/openpyxl/drawing/colors.py (15251b), scripts/openpyxl/drawing/connector.py (3863b), scripts/openpyxl/drawing/drawing.py (2339b), scripts/openpyxl/drawing/effect.py (9435b)\n\nFile v1.0.1:SKILL.md\n\n---\nname: office-doc-extractor\ndescription: Convert Microsoft Office documents (DOCX, XLSX, PPTX) to Markdown without any external dependencies. Use when the user needs to extract text from Word documents, Excel spreadsheets, or PowerPoint presentations for analysis, indexing, or LLM processing. Pure Python implementation — no pip install, no subprocess calls, no network downloads required. Works offline.\n---\n\n# Office Document Extractor\n\nZero-dependency converter for Microsoft Office documents. Extracts text and structure from DOCX, XLSX, and PPTX files into clean Markdown.\n\n## Quick Start\n\n```bash\n# Single file\npython3 scripts/main.py report.docx -o report.md\n\n# Batch convert a directory\npython3 scripts/main.py ./documents --batch -o ./markdown\n```\n\n## Supported Formats\n\n| Format | Extension | Output |\n|---|---|---|\n| Word | .docx | Headings, paragraphs |\n| Excel | .xlsx | Tables (one per sheet) |\n| PowerPoint | .pptx | Slides as sections |\n\n## How It Works\n\n- **DOCX**: Parses the ZIP archive's XML directly using Python's `zipfile` and `xml.etree`\n- **XLSX**: Uses bundled `openpyxl` (pure Python, no C extensions)\n- **PPTX**: Parses the ZIP archive's slide XML directly\n\nNo external commands, no network calls, no pip install required.\n\n## Usage\n\n### Single File\n\n```bash\npython3 scripts/main.py <input_file> [-o <output.md>]\n```\n\nAuto-detects format from file extension. If `-o` is omitted, outputs to `<input>.md`.\n\n### Batch Conversion\n\n```bash\npython3 scripts/main.py <input_directory> --batch [-o <output_directory>]\n```\n\nConverts all `.docx`, `.xlsx`, `.pptx` files in the directory. Results saved to `markdown_output/` by default.\n\n## Resources\n\n### scripts/\n\n- **main.py** — Unified CLI for single-file and batch conversion\n- **docx_extractor.py** — DOCX → Markdown (standard library only)\n- **xlsx_extractor.py** — XLSX → Markdown tables (bundled openpyxl)\n- **pptx_extractor.py** — PPTX → Markdown (standard library only)\n\n### Bundled Dependencies\n\n- **openpyxl/** — Pure Python Excel library (v3.1.5)\n- **et_xmlfile/** — openpyxl dependency (pure Python)\n\n## Limitations\n\n- Does not extract images or embedded objects (text only)\n- Does not preserve complex formatting (colors, fonts, layouts)\n- Does not handle encrypted/password-protected files\n- No OCR for scanned documents (use OpenClaw's native `pdf` tool for that)\n\n## Why This Skill?\n\nExisting markitdown-based skills require `pip install` or external CLI tools, which triggers ClawHub security warnings. This skill is **100% self-contained** — install it and use it immediately, even offline.\n\nFile v1.0.1:_meta.json\n\n{\n  \"ownerId\": \"kn79y85j3wvymevvn3nrz452a182atxe\",\n  \"slug\": \"office-doc-extractor\",\n  \"version\": \"1.0.1\",\n  \"publishedAt\": 1777889320701\n}\n\nFile v1.0.1:skill-card.md\n\n## Description:\n\nConvert Microsoft Office documents (DOCX, XLSX, PPTX) to Markdown without external dependencies for analysis, indexing, or LLM processing.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[michealxie001](https://clawhub.ai/user/michealxie001)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers, analysts, and agent operators use this skill to convert DOCX, XLSX, and PPTX files into Markdown for review, indexing, summarization, or downstream LLM workflows.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Large, malformed, or malicious Office files may cause excessive resource use during ZIP and XML parsing.\n\nMitigation: Process untrusted or bulk document uploads in a constrained environment and apply file size, decompression, timeout, and output limits before conversion.\n\nRisk: The converter extracts text and basic structure only, so images, embedded objects, complex formatting, encrypted files, and OCR needs may be omitted.\n\nMitigation: Confirm converted Markdown against the source document when complete fidelity is required, and use a dedicated OCR or secure document-processing workflow for unsupported content.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/michealxie001/skills/office-doc-extractor)\n- [openpyxl documentation](https://openpyxl.readthedocs.io)\n- [et_xmlfile project](https://foss.heptapod.net/openpyxl/et_xmlfile)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, Files, Shell commands, Guidance]\n\n**Output Format:** [Markdown files and concise command guidance]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Supports single-file and batch conversion; extracts text only and does not perform OCR.]\n\n## Skill Version(s):\n\n1.0.1 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v1.0.0: 199 files, 265376 bytes\n\nFiles: scripts/docx_extractor.py (2908b), scripts/et_xmlfile/__init__.py (228b), scripts/et_xmlfile/incremental_tree.py (34534b), scripts/et_xmlfile/xmlfile.py (4886b), scripts/main.py (2974b), scripts/openpyxl/__init__.py (603b), scripts/openpyxl/_constants.py (306b), scripts/openpyxl/cell/__init__.py (122b), scripts/openpyxl/cell/_writer.py (4015b), scripts/openpyxl/cell/cell.py (8922b), scripts/openpyxl/cell/read_only.py (3097b), scripts/openpyxl/cell/rich_text.py (5628b), scripts/openpyxl/cell/text.py (4367b), scripts/openpyxl/chart/__init__.py (564b), scripts/openpyxl/chart/_3d.py (3104b), scripts/openpyxl/chart/_chart.py (5746b), scripts/openpyxl/chart/area_chart.py (2890b), scripts/openpyxl/chart/axis.py (12580b), scripts/openpyxl/chart/bar_chart.py (4142b), scripts/openpyxl/chart/bubble_chart.py (2004b), scripts/openpyxl/chart/chartspace.py (6069b), scripts/openpyxl/chart/data_source.py (5782b), scripts/openpyxl/chart/descriptors.py (736b), scripts/openpyxl/chart/error_bar.py (1832b), scripts/openpyxl/chart/label.py (4133b), scripts/openpyxl/chart/layout.py (2040b), scripts/openpyxl/chart/legend.py (2040b), scripts/openpyxl/chart/line_chart.py (3951b), scripts/openpyxl/chart/marker.py (2600b), scripts/openpyxl/chart/picture.py (1156b), scripts/openpyxl/chart/pie_chart.py (4793b), scripts/openpyxl/chart/pivot.py (1741b), scripts/openpyxl/chart/plotarea.py (5805b), scripts/openpyxl/chart/print_settings.py (1454b), scripts/openpyxl/chart/radar_chart.py (1521b), scripts/openpyxl/chart/reader.py (802b), scripts/openpyxl/chart/reference.py (3098b), scripts/openpyxl/chart/scatter_chart.py (1563b), scripts/openpyxl/chart/series_factory.py (1368b), scripts/openpyxl/chart/series.py (5896b), scripts/openpyxl/chart/shapes.py (2815b), scripts/openpyxl/chart/stock_chart.py (1604b), scripts/openpyxl/chart/surface_chart.py (2914b), scripts/openpyxl/chart/text.py (1847b), scripts/openpyxl/chart/title.py (1952b), scripts/openpyxl/chart/trendline.py (3045b), scripts/openpyxl/chart/updown_bars.py (897b), scripts/openpyxl/chartsheet/__init__.py (71b), scripts/openpyxl/chartsheet/chartsheet.py (3911b), scripts/openpyxl/chartsheet/custom.py (1691b), scripts/openpyxl/chartsheet/properties.py (679b), scripts/openpyxl/chartsheet/protection.py (1265b), scripts/openpyxl/chartsheet/publish.py (1587b), scripts/openpyxl/chartsheet/relation.py (2731b), scripts/openpyxl/chartsheet/views.py (1341b), scripts/openpyxl/comments/__init__.py (67b), scripts/openpyxl/comments/author.py (388b), scripts/openpyxl/comments/comment_sheet.py (5753b), scripts/openpyxl/comments/comments.py (1466b), scripts/openpyxl/comments/shape_writer.py (3809b), scripts/openpyxl/compat/__init__.py (1592b), scripts/openpyxl/compat/abc.py (155b), scripts/openpyxl/compat/numbers.py (1617b), scripts/openpyxl/compat/product.py (264b), scripts/openpyxl/compat/singleton.py (1023b), scripts/openpyxl/compat/strings.py (604b), scripts/openpyxl/descriptors/__init__.py (1952b), scripts/openpyxl/descriptors/base.py (7135b), scripts/openpyxl/descriptors/container.py (889b), scripts/openpyxl/descriptors/excel.py (2412b), scripts/openpyxl/descriptors/namespace.py (313b), scripts/openpyxl/descriptors/nested.py (2603b), scripts/openpyxl/descriptors/sequence.py (3490b), scripts/openpyxl/descriptors/serialisable.py (7361b), scripts/openpyxl/descriptors/slots.py (824b), scripts/openpyxl/drawing/__init__.py (66b), scripts/openpyxl/drawing/colors.py (15251b), scripts/openpyxl/drawing/connector.py (3863b), scripts/openpyxl/drawing/drawing.py (2339b), scripts/openpyxl/drawing/effect.py (9435b)\n\nFile v1.0.0:SKILL.md\n\n---\nname: office-doc-extractor\ndescription: Convert Microsoft Office documents (DOCX, XLSX, PPTX) to Markdown without any external dependencies. Use when the user needs to extract text from Word documents, Excel spreadsheets, or PowerPoint presentations for analysis, indexing, or LLM processing. Pure Python implementation — no pip install, no subprocess calls, no network downloads required. Works offline.\n---\n\n# Office Document Extractor\n\nZero-dependency converter for Microsoft Office documents. Extracts text and structure from DOCX, XLSX, and PPTX files into clean Markdown.\n\n## Quick Start\n\n```bash\n# Single file\npython3 scripts/main.py report.docx -o report.md\n\n# Batch convert a directory\npython3 scripts/main.py ./documents --batch -o ./markdown\n```\n\n## Supported Formats\n\n| Format | Extension | Output |\n|---|---|---|\n| Word | .docx | Headings, paragraphs |\n| Excel | .xlsx | Tables (one per sheet) |\n| PowerPoint | .pptx | Slides as sections |\n\n## How It Works\n\n- **DOCX**: Parses the ZIP archive's XML directly using Python's `zipfile` and `xml.etree`\n- **XLSX**: Uses bundled `openpyxl` (pure Python, no C extensions)\n- **PPTX**: Parses the ZIP archive's slide XML directly\n\nNo external commands, no network calls, no pip install required.\n\n## Usage\n\n### Single File\n\n```bash\npython3 scripts/main.py <input_file> [-o <output.md>]\n```\n\nAuto-detects format from file extension. If `-o` is omitted, outputs to `<input>.md`.\n\n### Batch Conversion\n\n```bash\npython3 scripts/main.py <input_directory> --batch [-o <output_directory>]\n```\n\nConverts all `.docx`, `.xlsx`, `.pptx` files in the directory. Results saved to `markdown_output/` by default.\n\n## Resources\n\n### scripts/\n\n- **main.py** — Unified CLI for single-file and batch conversion\n- **docx_extractor.py** — DOCX → Markdown (standard library only)\n- **xlsx_extractor.py** — XLSX → Markdown tables (bundled openpyxl)\n- **pptx_extractor.py** — PPTX → Markdown (standard library only)\n\n### Bundled Dependencies\n\n- **openpyxl/** — Pure Python Excel library (v3.1.5)\n- **et_xmlfile/** — openpyxl dependency (pure Python)\n\n## Limitations\n\n- Does not extract images or embedded objects (text only)\n- Does not preserve complex formatting (colors, fonts, layouts)\n- Does not handle encrypted/password-protected files\n- No OCR for scanned documents (use OpenClaw's native `pdf` tool for that)\n\n## Why This Skill?\n\nExisting markitdown-based skills require `pip install` or external CLI tools, which triggers ClawHub security warnings. This skill is **100% self-contained** — install it and use it immediately, even offline.\n\nFile v1.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn79y85j3wvymevvn3nrz452a182atxe\",\n  \"slug\": \"office-doc-extractor\",\n  \"version\": \"1.0.0\",\n  \"publishedAt\": 1777888300027\n}","readmeExcerpt":"Skill: Office Document Extractor Owner: michealxie001 Summary: Convert Microsoft Office documents (DOCX, XLSX, PPTX) to Markdown without any external dependencies. Use when the user needs to extract text from Word docume... Tags: converter:1.0.0, docx:1.0.0, latest:1.0.1, markdown:1.0.0, office:1.0.0, pptx:1.0.0, python:1.0.0, xlsx:1.0.0 Version history: v1.0.1 | 2026-05-04T10:08:40.701Z | user Fix: Removed pycache, ","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"# Single file\npython3 scripts/main.py report.docx -o report.md\n\n# Batch convert a directory\npython3 scripts/main.py ./documents --batch -o ./markdown"},{"language":"bash","snippet":"python3 scripts/main.py <input_file> [-o <output.md>]"},{"language":"bash","snippet":"python3 scripts/main.py <input_directory> --batch [-o <output_directory>]"},{"language":"bash","snippet":"# Single file\npython3 scripts/main.py report.docx -o report.md\n\n# Batch convert a directory\npython3 scripts/main.py ./documents --batch -o ./markdown"},{"language":"bash","snippet":"python3 scripts/main.py <input_file> [-o <output.md>]"},{"language":"bash","snippet":"python3 scripts/main.py <input_directory> --batch [-o <output_directory>]"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: office-doc-extractor\ndescription: Convert Microsoft Office documents (DOCX, XLSX, PPTX) to Markdown without any external dependencies. Use when the user needs to extract text from Word documents, Excel spreadsheets, or PowerPoint presentations for analysis, indexing, or LLM processing. Pure Python implementation — no pip install, no subprocess calls, no network downloads required. Works offline.\n---\n\n# Office Document Extractor\n\nZero-dependency converter for Microsoft Office documents. Extracts text and structure from DOCX, XLSX, and PPTX files into clean Markdown.\n\n## Quick Start\n\n```bash\n# Single file\npython3 scripts/main.py report.docx -o report.md\n\n# Batch convert a directory\npython3 scripts/main.py ./documents --batch -o ./markdown\n```\n\n## Supported Formats\n\n| Format | Extension | Output |\n|---|---|---|\n| Word | .docx | Headings, paragraphs |\n| Excel | .xlsx | Tables (one per sheet) |\n| PowerPoint | .pptx | Slides as sections |\n\n## How It Works\n\n- **DOCX**: Parses the ZIP archive's XML directly using Python's `zipfile` and `xml.etree`\n- **XLSX**: Uses bundled `openpyxl` (pure Python, no C extensions)\n- **PPTX**: Parses the ZIP archive's slide XML directly\n\nNo external commands, no network calls, no pip install required.\n\n## Usage\n\n### Single File\n\n```bash\npython3 scripts/main.py <input_file> [-o <output.md>]\n```\n\nAuto-detects format from file extension. If `-o` is omitted, outputs to `<input>.md`.\n\n### Batch Conversion\n\n```bash\npython3 scripts/main.py <input_directory> --batch [-o <output_directory>]\n```\n\nConverts all `.docx`, `.xlsx`, `.pptx` files in the directory. Results saved to `markdown_output/` by default.\n\n## Resources\n\n### scripts/\n\n- **main.py** — Unified CLI for single-file and batch conversion\n- **docx_extractor.py** — DOCX → Markdown (standard library only)\n- **xlsx_extractor.py** — XLSX → Markdown tables (bundled openpyxl)\n- **pptx_extractor.py** — PPTX → Markdown (standard library only)\n\n### Bundled Dependencies\n\n- **openpyxl/** — Pure Python Excel library (v3.1.5)\n- **et_xmlfile/** — openpyxl dependency (pure Python)\n\n## Limitations\n\n- Does not extract images or embedded objects (text only)\n- Does not preserve complex formatting (colors, fonts, layouts)\n- Does not handle encrypted/password-protected files\n- No OCR for scanned documents (use OpenClaw's native `pdf` tool for that)\n\n## Why This Skill?\n\nExisting markitdown-based skills require `pip install` or external CLI tools, which triggers ClawHub security warnings. This skill is **100% self-contained** — install it and use it immediately, even offline."},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn79y85j3wvymevvn3nrz452a182atxe\",\n  \"slug\": \"office-doc-extractor\",\n  \"version\": \"1.0.1\",\n  \"publishedAt\": 1777889320701\n}"},{"path":"skill-card.md","content":"## Description:\n\nConvert Microsoft Office documents (DOCX, XLSX, PPTX) to Markdown without external dependencies for analysis, indexing, or LLM processing.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[michealxie001](https://clawhub.ai/user/michealxie001)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers, analysts, and agent operators use this skill to convert DOCX, XLSX, and PPTX files into Markdown for review, indexing, summarization, or downstream LLM workflows.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Large, malformed, or malicious Office files may cause excessive resource use during ZIP and XML parsing.\n\nMitigation: Process untrusted or bulk document uploads in a constrained environment and apply file size, decompression, timeout, and output limits before conversion.\n\nRisk: The converter extracts text and basic structure only, so images, embedded objects, complex formatting, encrypted files, and OCR needs may be omitted.\n\nMitigation: Confirm converted Markdown against the source document when complete fidelity is required, and use a dedicated OCR or secure document-processing workflow for unsupported content.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/michealxie001/skills/office-doc-extractor)\n- [openpyxl documentation](https://openpyxl.readthedocs.io)\n- [et_xmlfile project](https://foss.heptapod.net/openpyxl/et_xmlfile)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, Files, Shell commands, Guidance]\n\n**Output Format:** [Markdown files and concise command guidance]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Supports single-file and batch conversion; extracts text only and does not perform OCR.]\n\n## Skill Version(s):\n\n1.0.1 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Convert Microsoft Office documents (DOCX, XLSX, PPTX) to Markdown without any external dependencies. Use when the user needs to extract text from Word docume... Skill: Office Document Extractor Owner: michealxie001 Summary: Convert Microsoft Office documents (DOCX, XLSX, PPTX) to Markdown without any external dependencies. Use when the user needs to extract text from Word docume... Tags: converter:1.0.0, docx:1.0.0, latest:1.0.1, markdown:1.0.0, office:1.0.0, pptx:1.0.0, python:1.0.0, xlsx:1.0.0 Version history: v1.0.1 | 2026-05-04T10:08:40.701Z | user Fix: Removed pycache,","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1015,"uniquenessScore":51,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T17:57:31.233Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T17:57:31.233Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T20:59:34.892Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}