{"id":"a0bc03a6-281b-4906-9319-88a79b6e5ffa","entityType":"agent","slug":"clawhub-hunter-crk-ghost-eye","name":"Ghost Eye","canonicalUrl":"https://www.xpersona.co/agent/clawhub-hunter-crk-ghost-eye","canonicalPath":"/agent/clawhub-hunter-crk-ghost-eye","generatedAt":"2026-10-11T14:14:05.235Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-11T09:40:16.788Z","emptyReason":null},"description":"Ghost Eye 👁️ — Let any pure-text LLM see images through any vision model. OCR + visual summary in one shot. Skill: Ghost Eye Owner: hunter-crk Summary: Ghost Eye 👁️ — Let any pure-text LLM see images through any vision model. OCR + visual summary in one shot. Tags: image:1.0.2, latest:1.0.4, llm:1.0.2, ocr:1.0.2, vision:1.0.2 Version history: v1.0.4 | 2026-07-09T15:59:25.635Z | auto ghost-eye 1.0.4 - Added image analysis cache files for improved performance. - Removed the redundant skill-card.md documentation file. - Upda","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.1K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s179nt0fhpj5npbanyqj0jmhhs8a6kyq:ghost-eye","sourceUrl":"https://clawhub.ai/hunter-crk/ghost-eye","homepage":"https://clawhub.ai/hunter-crk/skills/ghost-eye","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/hunter-crk/ghost-eye","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/hunter-crk/skills/ghost-eye","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":61,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Ghost Eye 👁️ — Let any pure-text LLM see images through any vision model. OCR + visual summary in one shot. Skill: Ghost Eye Owner: hunter-crk Summary: Ghost E"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T09:40:16.788Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T09:40:16.788Z","emptyReason":null},"stars":null,"forks":null,"downloads":1097,"packageName":null,"latestVersion":"1.0.4","tractionLabel":"1.1K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T09:40:16.781Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T09:40:16.788Z","lastCrawledAt":"2026-10-11T09:40:16.781Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T09:40:16.782Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.4","createdAt":"2026-07-09T15:59:25.635Z","changelog":"ghost-eye 1.0.4 - Added image analysis cache files for improved performance. - Removed the redundant skill-card.md documentation file. - Updated scripts/analyze.py with changes to support caching and streamline image analysis. - No changes to core user workflow or environment variables.","fileCount":8,"zipByteSize":14162},{"version":"1.0.3","createdAt":"2026-07-09T15:40:20.108Z","changelog":"Moved category to agents","fileCount":6,"zipByteSize":11203},{"version":"1.0.2","createdAt":"2026-07-09T15:36:54.275Z","changelog":"Updated description: any vision model, not just Nex-N2-Pro. Category: image-understanding","fileCount":5,"zipByteSize":9688},{"version":"1.0.1","createdAt":"2026-07-09T15:35:51.228Z","changelog":"- Expanded model compatibility: Ghost Eye now supports any OpenAI-compatible vision model, not just Nex-N2-Pro. - Description, intro, and workflow updated to reflect support for all vision models (default remains Nex-N2-Pro). - Clarified tool name is unchanged for backward compatibility. - No code/API changes; documentation only.","fileCount":5,"zipByteSize":9702},{"version":"1.0.0","createdAt":"2026-07-09T15:33:17.319Z","changelog":"Initial release of Ghost Eye: enable pure-text LLMs to process images via the Nex-N2-Pro vision model. - Automatically analyzes images and injects OCR + visual summary as plain text for LLMs. - Offers two usage modes: zero-click auto-preprocess and explicit tool call. - Smart input handling (prefers file path, falls back to URL or base64). - Built-in robust caching, error handling, and environment-based configuration. - Compatible with SiliconFlow, OpenRouter, and OpenAI-compatible APIs. - Prioritizes security: never logs image content or keys, always cleans up temp files.","fileCount":5,"zipByteSize":9608}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s179nt0fhpj5npbanyqj0jmhhs8a6kyq:ghost-eye","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-hunter-crk-ghost-eye/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-hunter-crk-ghost-eye/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-hunter-crk-ghost-eye/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-hunter-crk-ghost-eye/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-hunter-crk-ghost-eye/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-hunter-crk-ghost-eye/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T14:14:05.234Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-hunter-crk-ghost-eye/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-hunter-crk-ghost-eye/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-hunter-crk-ghost-eye/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-hunter-crk-ghost-eye/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-11T09:40:16.788Z","emptyReason":null},"readme":"Skill: Ghost Eye\n\nOwner: hunter-crk\n\nSummary: Ghost Eye 👁️ — Let any pure-text LLM see images through any vision model. OCR + visual summary in one shot.\n\nTags: image:1.0.2, latest:1.0.4, llm:1.0.2, ocr:1.0.2, vision:1.0.2\n\nVersion history:\n\nv1.0.4 | 2026-07-09T15:59:25.635Z | auto\n\nghost-eye 1.0.4\n\n- Added image analysis cache files for improved performance.\n- Removed the redundant skill-card.md documentation file.\n- Updated scripts/analyze.py with changes to support caching and streamline image analysis.\n- No changes to core user workflow or environment variables.\n\nv1.0.3 | 2026-07-09T15:40:20.108Z | user\n\nMoved category to agents\n\nv1.0.2 | 2026-07-09T15:36:54.275Z | user\n\nUpdated description: any vision model, not just Nex-N2-Pro. Category: image-understanding\n\nv1.0.1 | 2026-07-09T15:35:51.228Z | auto\n\n- Expanded model compatibility: Ghost Eye now supports any OpenAI-compatible vision model, not just Nex-N2-Pro.\n- Description, intro, and workflow updated to reflect support for all vision models (default remains Nex-N2-Pro).\n- Clarified tool name is unchanged for backward compatibility.\n- No code/API changes; documentation only.\n\nv1.0.0 | 2026-07-09T15:33:17.319Z | auto\n\nInitial release of Ghost Eye: enable pure-text LLMs to process images via the Nex-N2-Pro vision model.\n\n- Automatically analyzes images and injects OCR + visual summary as plain text for LLMs.\n- Offers two usage modes: zero-click auto-preprocess and explicit tool call.\n- Smart input handling (prefers file path, falls back to URL or base64).\n- Built-in robust caching, error handling, and environment-based configuration.\n- Compatible with SiliconFlow, OpenRouter, and OpenAI-compatible APIs.\n- Prioritizes security: never logs image content or keys, always cleans up temp files.\n\nArchive index:\n\nArchive v1.0.4: 8 files, 14162 bytes\n\nFiles: cache/2a0206456252d9926270d09e5fea3640.json (2155b), cache/90a6d04c46344c78d09830c905ab65f7.json (1506b), cache/c5393b1697c9e7598ce6d35871b8ad41.json (2390b), references/multimodal-config.md (1933b), scripts/analyze.py (13173b), skill-card.md (2143b), SKILL.md (4853b), _meta.json (128b)\n\nFile v1.0.4:SKILL.md\n\n---\nname: \"ghost-eye\"\ndescription: \"Ghost Eye 👁️ — Let any pure-text LLM see images through any vision model. OCR + visual summary in one shot.\"\nmetadata:\n  {\n    \"openclaw\":\n      {\n        \"emoji\": \"👁️\",\n        \"requires\": { \"env\": [\"NEXN2_API_KEY\"] },\n        \"primaryEnv\": \"NEXN2_API_KEY\",\n      },\n  }\n---\n\n# Ghost Eye 👁️\n\nGive your text-only LLM the power to see. Ghost Eye is a lightweight image preprocessing bridge: when an image enters the conversation, it calls any OpenAI-compatible vision model (default: `nex-agi/Nex-N2-Pro`), produces a structured plain-text output (full OCR + visual summary), and feeds it back into the conversation — so your pure-text model can \"see\" without ever touching a multimodal API.\n\n## What it does\n\n```\nUser sends image → Ghost Eye detects → vision model analyzes → OCR text + scene summary → your LLM answers\n```\n\n## Two modes\n\n### Mode 1: Auto-preprocess (recommended)\nWhen `multimodalPreprocess` is configured, Ghost Eye fires automatically on any inbound image. The user never knows it's there — they just get answers about images.\n\n### Mode 2: Tool-call mode\nRegistered as `analyze_image_by_nexn2` tool (name stays for backward compatibility). Your LLM calls it explicitly when it sees an image. Add this to the system prompt:\n\n> When the user sends an image, screenshot, photo, or document scan, call the analyze_image_by_nexn2 tool to extract text and describe the image, then answer based on the returned content.\n\n## Workflow\n\n### Step 1: Receive image\n\n**Priority: `--image-path` > `--image-url` > `--image-base64`**\n\n⛔ Always prefer `--image-path` to avoid command-line `Argument list too long` errors with large base64 strings. Only fall back to `--image-url` or `--image-base64` when no local path is available.\n\n```bash\n# Preferred: local file path (no size limit)\npython3 {baseDir}/scripts/analyze.py --image-path \"<absolute path>\"\n\n# Fallback: public URL\npython3 {baseDir}/scripts/analyze.py --image-url \"<url>\"\n\n# Last resort: base64 (small images only, <50KB)\npython3 {baseDir}/scripts/analyze.py --image-base64 \"<base64>\"\n```\n\nThe script handles:\n- Format validation (JPG/PNG/WebP/GIF/BMP via magic bytes)\n- Cache check / read / write (MD5-based, 7-day TTL)\n- Image compression (Pillow, max 1920px longest edge, quality 85%)\n- API call with 1 automatic retry\n- Structured JSON output\n\n### Step 2: Parse JSON output\n\nSuccess:\n```json\n{\"success\": true, \"content\": \"【OCR文字提取】\\n...\\n\\n【画面内容总结】\\n...\", \"metadata\": {\"model\": \"...\", \"tokens_used\": 1200, \"cached\": false, \"process_time_ms\": 1500}}\n```\n\nError:\n```json\n{\"success\": false, \"content\": \"error message\", \"metadata\": {}}\n```\n\nPass `content` directly into the LLM conversation context.\n\n### Step 3: Caching\n\n- Cache directory: `{baseDir}/cache/` (auto-created)\n- Cache key: MD5 hash of raw image bytes\n- TTL: 7 days (`NEXN2_CACHE_TTL_DAYS`)\n- Clear cache: delete all `.json` files in `{baseDir}/cache/`\n- Toggle: `NEXN2_CACHE_ENABLE=true/false`\n\n## Environment variables\n\n| Variable | Required | Default |\n|----------|----------|---------|\n| NEXN2_API_KEY | ✅ Yes | — |\n| NEXN2_BASE_URL | No | https://api.siliconflow.cn/v1 |\n| NEXN2_MODEL_NAME | No | nex-agi/Nex-N2-Pro |\n| NEXN2_PROMPT_TEMPLATE | No | Built-in structured template |\n| NEXN2_IMAGE_COMPRESS | No | true |\n| NEXN2_CACHE_ENABLE | No | true |\n| NEXN2_CACHE_TTL_DAYS | No | 7 |\n| NEXN2_TIMEOUT_MS | No | 30000 |\n\nIf `NEXN2_API_KEY` is not set, returns a friendly Chinese error message.\n\n## Error handling\n\n| Scenario | Returns |\n|----------|---------|\n| Network/API failure (after retry) | Friendly \"service unavailable\" message |\n| Unsupported format | \"Please use JPG/PNG/WebP format\" |\n| Content safety block | \"Image flagged by safety filter\" |\n| Empty model output | \"No content returned, try a different image\" |\n\n⚠️ Errors never crash the conversation — structured JSON response is always returned.\n\n## Setup\n\nIn `openclaw.json` under `skills.entries`:\n\n```json5\n\"ghost-eye\": {\n  \"enabled\": true,\n  \"apiKey\": { \"source\": \"env\", \"provider\": \"default\", \"id\": \"NEXN2_API_KEY\" },\n  \"env\": {\n    \"NEXN2_BASE_URL\": \"https://api.siliconflow.cn/v1\",\n    \"NEXN2_MODEL_NAME\": \"nex-agi/Nex-N2-Pro\",\n    \"NEXN2_IMAGE_COMPRESS\": \"true\",\n    \"NEXN2_CACHE_ENABLE\": \"true\",\n    \"NEXN2_CACHE_TTL_DAYS\": \"7\",\n    \"NEXN2_TIMEOUT_MS\": \"30000\"\n  }\n}\n```\n\nFor `multimodalPreprocess` auto-mode, see `references/multimodal-config.md`.\n\n## Supported platforms\n\n- SiliconFlow (default, China-accessible)\n- OpenRouter\n- Any OpenAI-compatible chat completions endpoint\n\n## Safety\n\n- ⛔ Image base64 is never logged or written to conversation text\n- ⛔ API key is never hardcoded\n- ⛔ Temporary files are cleaned up immediately after processing\n- ⚠️ Output is always plain text / Markdown — never binary, never images\n\nFile v1.0.4:_meta.json\n\n{\n  \"ownerId\": \"kn78pm8jn9q0yydy5e2j0k169x8a7evs\",\n  \"slug\": \"ghost-eye\",\n  \"version\": \"1.0.4\",\n  \"publishedAt\": 1783612765635\n}\n\nFile v1.0.4:references/multimodal-config.md\n\n# OpenClaw 多模态预处理配置参考\n\n## multimodalPreprocess 配置\n\n在 `openclaw.json` 中添加：\n\n```json\n{\n  \"multimodalPreprocess\": {\n    \"enable\": true,\n    \"visionSkillId\": \"nex-n2-image-analyzer\",\n    \"promptTemplate\": \"以下是图片的完整分析结果，请严格基于该内容回答用户问题：\\n{{skillResult}}\\n\\n用户问题：{{userQuery}}\"\n  },\n  \"skills\": {\n    \"entries\": {\n      \"nex-n2-image-analyzer\": {\n        \"enabled\": true,\n        \"apiKey\": {\n          \"source\": \"env\",\n          \"provider\": \"default\",\n          \"id\": \"NEXN2_API_KEY\"\n        },\n        \"env\": {\n          \"NEXN2_BASE_URL\": \"https://api.siliconflow.cn/v1\",\n          \"NEXN2_MODEL_NAME\": \"nex-agi/Nex-N2-Pro\",\n          \"NEXN2_IMAGE_COMPRESS\": \"true\",\n          \"NEXN2_CACHE_ENABLE\": \"true\",\n          \"NEXN2_CACHE_TTL_DAYS\": \"7\",\n          \"NEXN2_TIMEOUT_MS\": \"30000\"\n        }\n      }\n    }\n  }\n}\n```\n\n## 配置项说明\n\n| 字段 | 说明 |\n|------|------|\n| `multimodalPreprocess.enable` | 开启全局图片预处理 |\n| `multimodalPreprocess.visionSkillId` | 指定处理图片的 Skill 名称 |\n| `multimodalPreprocess.promptTemplate` | 拼接结果的模板，`{{skillResult}}` 是 Skill 返回的 content，`{{userQuery}}` 是用户消息 |\n\n## 切换至 OpenRouter\n\n如果使用 OpenRouter 代替 SiliconFlow：\n\n```json\n{\n  \"NEXN2_BASE_URL\": \"https://openrouter.ai/api/v1\",\n  \"NEXN2_MODEL_NAME\": \"nex-agi/nex-n2-pro\"\n}\n```\n\n> 注意：OpenRouter 上的模型 ID 可能与 SiliconFlow 略有不同，以实际注册名称为准。\n\n## 不使用全局预处理\n\n如果仅需工具调用模式（非自动触发），保留 `skills.entries` 配置但**不添加** `multimodalPreprocess` 块，并在系统提示词中补充工具调用指令：\n\n> 当用户发送图片、截图、照片、文档截图时，请调用 analyze_image_by_nexn2 工具获取图片的文字与内容描述，再基于返回结果作答。\n\nFile v1.0.4:skill-card.md\n\n## Description:\n\nGhost Eye lets a pure-text LLM analyze images through an OpenAI-compatible vision model, returning OCR text and a visual summary.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[hunter-crk](https://clawhub.ai/user/hunter-crk)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and agents use this skill to convert incoming images, screenshots, photos, or scanned documents into text that a text-only LLM can reason over. It supports automatic image preprocessing or explicit tool-call use.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Inbound images can be sent automatically to a third-party vision API.\n\nMitigation: Use explicit tool-call mode for sensitive work and configure an approved OpenAI-compatible vision endpoint before deployment.\n\nRisk: OCR text and visual summaries are stored in plaintext cache files by default.\n\nMitigation: Disable caching or regularly clear the cache when images may contain personal, business, credential, or regulated data.\n\nRisk: Using image URLs can introduce uncontrolled network egress.\n\nMitigation: Prefer local image paths and use image URLs only when outbound network access is constrained and approved.\n\n## Reference(s):\n\n- [OpenClaw multimodal preprocessing configuration](references/multimodal-config.md)\n- [ClawHub Ghost Eye skill page](https://clawhub.ai/hunter-crk/skills/ghost-eye)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, JSON]\n\n**Output Format:** [JSON object with success, content, and metadata fields; content contains OCR text and a visual summary in plain text or Markdown.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Requires NEXN2_API_KEY; image analysis results may be cached as JSON files for up to 7 days by default.]\n\n## Skill Version(s):\n\n1.0.4 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.4:cache/2a0206456252d9926270d09e5fea3640.json\n\n{\"content\": \"【OCR文字提取】\\n\\n```text\\nopen\\n\\n弱酸性配方\\n添加酒精\\n\\n生产批号:20260530 13:57:03\\n限用日期:20280529 A060530A\\n\\n\\n象小家™EDI纯水湿巾\\n7道水净化工艺｜无刺激｜无酒精｜无荧光剂\\n\\n小象超市 自有品牌\\n\\n○产品名称:象小家™EDI纯水湿巾○净含量:90抽×1包○规格:150mm×200mm○主要成分:水刺无纺布,EDI纯水○生产批号及限期使用日期:见包装○保质期:两年○贮存条件:置于阴凉干燥处常温保存,避免阳光直射○使用方法:将盖子打开,揭开贴纸,抽出使用。建议使用完盖好贴纸和盖子,避免水分流失。○卫生标准:GB 15979○执行标准:GB/T 27728.1○委托商:北京象鲜科技有限公司○地址:北京市朝阳区小营北路15号院1号楼3层○联系方式:010-10107777○受委托商:浙江优全护理用品科技股份有限公司○地址:浙江省湖州市长兴县太湖街道陆汇路68号(代码:A)○卫生许可证号:浙卫消证字(2016)第002\\n```\\n\\n【画面内容总结】\\n\\n1. 核心主题：这是一包“象小家™EDI纯水湿巾”的包装实拍，用于日常清洁/擦拭，主打纯水、温和、无酒精等卖点。\\n\\n2. 元素与布局：图片主体为一包浅蓝色湿巾，放置在浅色木纹桌面上；包装上方是翻盖开口，可见“open”字样及“弱酸性配方”等宣传语；包装中部为产品名称和核心卖点；包装下方密集排列产品参数、成分、使用方法、执行标准、委托商和受委托商等信息。背景中有纸巾盒、杯子、绿色包装等物品，但处于虚化状态。\\n\\n3. 关键信息提炼：\\n   - 产品名称：象小家™EDI纯水湿巾。\\n   - 核心卖点：7道水净化工艺、无刺激、无酒精、无荧光剂。\\n   - 规格信息：90抽×1包，规格为150mm×200mm。\\n   - 主要成分：水刺无纺布、EDI纯水。\\n   - 日期信息：生产批号为20260530 13:57:03，限用日期为20280529。\", \"cached_at\": \"2026-07-09T15:51:51.620772+00:00\", \"expires_at\": \"2026-07-16T15:51:51.620772+00:00\", \"model\": \"nex-agi/Nex-N2-Pro\", \"tokens_used\": 6270}\n\nFile v1.0.4:cache/90a6d04c46344c78d09830c905ab65f7.json\n\n{\"content\": \"【OCR文字提取】\\n- DELL  \\n- healthy  \\n\\n其余区域未见清晰可辨的文字、标题、注释或表格内容。\\n\\n【画面内容总结】\\n1. 核心主题：这是一张现代开放式办公室场景图，主要用于展示办公空间布局、工位配置和日常办公环境。\\n\\n2. 元素与布局：  \\n   - 画面整体为一个较大的开放式办公区，空间宽敞，地面铺设浅灰色地毯。  \\n   - 左侧和右侧分布多组办公桌，桌面为浅木色，搭配白色桌腿、抽屉柜和灰白色网布办公椅。  \\n   - 工位上摆放有显示器、键盘、鼠标、笔记本电脑、纸巾、水杯、文件、绿植等办公及生活用品。  \\n   - 天花板为深色工业风设计，安装有多排长条形LED灯和中央空调出风口。  \\n   - 中间有白色立柱/隔断，后方可见走廊或更深处的办公区域。  \\n   - 右侧背景为大面积窗户，配有灰色卷帘。\\n\\n3. 关键信息提炼：  \\n   - 该空间是一个配置齐全的现代办公区，工位数量较多，适合团队办公。  \\n   - 环境明亮、整洁，灯光充足，整体风格偏简约、商务。  \\n   - 当前画面中没有人员，呈现出安静、无人办公的状态。  \\n   - 可见文字信息较少，仅有显示器品牌“DELL”以及桌面上的“healthy”字样。\", \"cached_at\": \"2026-07-09T15:53:17.422685+00:00\", \"expires_at\": \"2026-07-16T15:53:17.422685+00:00\", \"model\": \"nex-agi/Nex-N2-Pro\", \"tokens_used\": 3646}\n\nFile v1.0.4:cache/c5393b1697c9e7598ce6d35871b8ad41.json\n\n{\"content\": \"【OCR文字提取】\\n\\n← Back to Ghost Eye\\n\\n# Skill settings\\n\\nIntegrations  \\nAutomation  \\nResearch  \\nDevelopment  \\nProductivity  \\nCommunication  \\nCreative  \\nKnowledge  \\nAgents  \\nOperations  \\nSecurity  \\nFinance  \\nLifestyle  \\n✓ Other\\n\\nUpdate skill files\\n\\n## Publish a new version\\n\\nUpload a replacement release for this skill. New releases get a fresh scan.\\n\\nNew Version\\n\\n## Short summary\\n\\nUpdate the short summary used in cards, search, and previews.\\n\\nages through any vision model. OCR + visual\\n\\nSave\\n\\n## Catalog metadata\\n\\nChoose browse categories and author topics for this skill.\\n\\nOther\\n\\n## TOPICS\\n\\nAdd a topic\\n\\n【画面内容总结】\\n\\n1. 核心主题：  \\n   图片展示的是一个名为 “Skill settings” 的技能设置页面，用户正在编辑或配置某个技能的基本信息、分类、摘要、版本发布和主题等元数据。\\n\\n2. 元素与布局：  \\n   - 页面左上角有返回入口 “Back to Ghost Eye”。  \\n   - 主标题为 “Skill settings”。  \\n   - 页面主要分为左侧设置说明区和右侧操作/输入区。  \\n   - 中间偏上位置弹出了一个分类选择下拉菜单，包含 Integrations、Automation、Research、Development、Productivity、Communication、Creative、Knowledge、Agents、Operations、Security、Finance、Lifestyle、Other 等分类。  \\n   - 下拉菜单中 “Automation” 被红色边框高亮，表示当前鼠标悬停或正在选择该项。  \\n   - 下方当前选中的分类显示为 “Other”，并带有勾选标记。  \\n   - 页面右侧有 “Update skill files”、“New Version”、“Save” 等操作按钮。  \\n   - 页面底部有 “TOPICS” 主题输入框，占位文字为 “Add a topic”。\\n\\n3. 关键信息提炼：  \\n   - 当前页面用于配置技能设置。  \\n   - 用户正在从分类下拉菜单中选择 “Automation”，但当前已选分类仍显示为 “Other”。  \\n   - 页面支持发布新版本、更新技能文件、编辑短摘要、设置目录分类和添加主题。  \\n   - “Automation” 是下拉菜单中被重点标记的选项，说明用户可能准备将技能分类从 “Other” 改为 “Automation”。\", \"cached_at\": \"2026-07-09T15:38:45.194194+00:00\", \"expires_at\": \"2026-07-16T15:38:45.194194+00:00\", \"model\": \"nex-agi/Nex-N2-Pro\", \"tokens_used\": 4393}\n\nArchive v1.0.3: 6 files, 11203 bytes\n\nFiles: cache/c5393b1697c9e7598ce6d35871b8ad41.json (2390b), references/multimodal-config.md (1933b), scripts/analyze.py (12005b), skill-card.md (2483b), SKILL.md (4853b), _meta.json (128b)\n\nFile v1.0.3:SKILL.md\n\n---\nname: \"ghost-eye\"\ndescription: \"Ghost Eye 👁️ — Let any pure-text LLM see images through any vision model. OCR + visual summary in one shot.\"\nmetadata:\n  {\n    \"openclaw\":\n      {\n        \"emoji\": \"👁️\",\n        \"requires\": { \"env\": [\"NEXN2_API_KEY\"] },\n        \"primaryEnv\": \"NEXN2_API_KEY\",\n      },\n  }\n---\n\n# Ghost Eye 👁️\n\nGive your text-only LLM the power to see. Ghost Eye is a lightweight image preprocessing bridge: when an image enters the conversation, it calls any OpenAI-compatible vision model (default: `nex-agi/Nex-N2-Pro`), produces a structured plain-text output (full OCR + visual summary), and feeds it back into the conversation — so your pure-text model can \"see\" without ever touching a multimodal API.\n\n## What it does\n\n```\nUser sends image → Ghost Eye detects → vision model analyzes → OCR text + scene summary → your LLM answers\n```\n\n## Two modes\n\n### Mode 1: Auto-preprocess (recommended)\nWhen `multimodalPreprocess` is configured, Ghost Eye fires automatically on any inbound image. The user never knows it's there — they just get answers about images.\n\n### Mode 2: Tool-call mode\nRegistered as `analyze_image_by_nexn2` tool (name stays for backward compatibility). Your LLM calls it explicitly when it sees an image. Add this to the system prompt:\n\n> When the user sends an image, screenshot, photo, or document scan, call the analyze_image_by_nexn2 tool to extract text and describe the image, then answer based on the returned content.\n\n## Workflow\n\n### Step 1: Receive image\n\n**Priority: `--image-path` > `--image-url` > `--image-base64`**\n\n⛔ Always prefer `--image-path` to avoid command-line `Argument list too long` errors with large base64 strings. Only fall back to `--image-url` or `--image-base64` when no local path is available.\n\n```bash\n# Preferred: local file path (no size limit)\npython3 {baseDir}/scripts/analyze.py --image-path \"<absolute path>\"\n\n# Fallback: public URL\npython3 {baseDir}/scripts/analyze.py --image-url \"<url>\"\n\n# Last resort: base64 (small images only, <50KB)\npython3 {baseDir}/scripts/analyze.py --image-base64 \"<base64>\"\n```\n\nThe script handles:\n- Format validation (JPG/PNG/WebP/GIF/BMP via magic bytes)\n- Cache check / read / write (MD5-based, 7-day TTL)\n- Image compression (Pillow, max 1920px longest edge, quality 85%)\n- API call with 1 automatic retry\n- Structured JSON output\n\n### Step 2: Parse JSON output\n\nSuccess:\n```json\n{\"success\": true, \"content\": \"【OCR文字提取】\\n...\\n\\n【画面内容总结】\\n...\", \"metadata\": {\"model\": \"...\", \"tokens_used\": 1200, \"cached\": false, \"process_time_ms\": 1500}}\n```\n\nError:\n```json\n{\"success\": false, \"content\": \"error message\", \"metadata\": {}}\n```\n\nPass `content` directly into the LLM conversation context.\n\n### Step 3: Caching\n\n- Cache directory: `{baseDir}/cache/` (auto-created)\n- Cache key: MD5 hash of raw image bytes\n- TTL: 7 days (`NEXN2_CACHE_TTL_DAYS`)\n- Clear cache: delete all `.json` files in `{baseDir}/cache/`\n- Toggle: `NEXN2_CACHE_ENABLE=true/false`\n\n## Environment variables\n\n| Variable | Required | Default |\n|----------|----------|---------|\n| NEXN2_API_KEY | ✅ Yes | — |\n| NEXN2_BASE_URL | No | https://api.siliconflow.cn/v1 |\n| NEXN2_MODEL_NAME | No | nex-agi/Nex-N2-Pro |\n| NEXN2_PROMPT_TEMPLATE | No | Built-in structured template |\n| NEXN2_IMAGE_COMPRESS | No | true |\n| NEXN2_CACHE_ENABLE | No | true |\n| NEXN2_CACHE_TTL_DAYS | No | 7 |\n| NEXN2_TIMEOUT_MS | No | 30000 |\n\nIf `NEXN2_API_KEY` is not set, returns a friendly Chinese error message.\n\n## Error handling\n\n| Scenario | Returns |\n|----------|---------|\n| Network/API failure (after retry) | Friendly \"service unavailable\" message |\n| Unsupported format | \"Please use JPG/PNG/WebP format\" |\n| Content safety block | \"Image flagged by safety filter\" |\n| Empty model output | \"No content returned, try a different image\" |\n\n⚠️ Errors never crash the conversation — structured JSON response is always returned.\n\n## Setup\n\nIn `openclaw.json` under `skills.entries`:\n\n```json5\n\"ghost-eye\": {\n  \"enabled\": true,\n  \"apiKey\": { \"source\": \"env\", \"provider\": \"default\", \"id\": \"NEXN2_API_KEY\" },\n  \"env\": {\n    \"NEXN2_BASE_URL\": \"https://api.siliconflow.cn/v1\",\n    \"NEXN2_MODEL_NAME\": \"nex-agi/Nex-N2-Pro\",\n    \"NEXN2_IMAGE_COMPRESS\": \"true\",\n    \"NEXN2_CACHE_ENABLE\": \"true\",\n    \"NEXN2_CACHE_TTL_DAYS\": \"7\",\n    \"NEXN2_TIMEOUT_MS\": \"30000\"\n  }\n}\n```\n\nFor `multimodalPreprocess` auto-mode, see `references/multimodal-config.md`.\n\n## Supported platforms\n\n- SiliconFlow (default, China-accessible)\n- OpenRouter\n- Any OpenAI-compatible chat completions endpoint\n\n## Safety\n\n- ⛔ Image base64 is never logged or written to conversation text\n- ⛔ API key is never hardcoded\n- ⛔ Temporary files are cleaned up immediately after processing\n- ⚠️ Output is always plain text / Markdown — never binary, never images\n\nFile v1.0.3:_meta.json\n\n{\n  \"ownerId\": \"kn78pm8jn9q0yydy5e2j0k169x8a7evs\",\n  \"slug\": \"ghost-eye\",\n  \"version\": \"1.0.3\",\n  \"publishedAt\": 1783611620108\n}\n\nFile v1.0.3:references/multimodal-config.md\n\n# OpenClaw 多模态预处理配置参考\n\n## multimodalPreprocess 配置\n\n在 `openclaw.json` 中添加：\n\n```json\n{\n  \"multimodalPreprocess\": {\n    \"enable\": true,\n    \"visionSkillId\": \"nex-n2-image-analyzer\",\n    \"promptTemplate\": \"以下是图片的完整分析结果，请严格基于该内容回答用户问题：\\n{{skillResult}}\\n\\n用户问题：{{userQuery}}\"\n  },\n  \"skills\": {\n    \"entries\": {\n      \"nex-n2-image-analyzer\": {\n        \"enabled\": true,\n        \"apiKey\": {\n          \"source\": \"env\",\n          \"provider\": \"default\",\n          \"id\": \"NEXN2_API_KEY\"\n        },\n        \"env\": {\n          \"NEXN2_BASE_URL\": \"https://api.siliconflow.cn/v1\",\n          \"NEXN2_MODEL_NAME\": \"nex-agi/Nex-N2-Pro\",\n          \"NEXN2_IMAGE_COMPRESS\": \"true\",\n          \"NEXN2_CACHE_ENABLE\": \"true\",\n          \"NEXN2_CACHE_TTL_DAYS\": \"7\",\n          \"NEXN2_TIMEOUT_MS\": \"30000\"\n        }\n      }\n    }\n  }\n}\n```\n\n## 配置项说明\n\n| 字段 | 说明 |\n|------|------|\n| `multimodalPreprocess.enable` | 开启全局图片预处理 |\n| `multimodalPreprocess.visionSkillId` | 指定处理图片的 Skill 名称 |\n| `multimodalPreprocess.promptTemplate` | 拼接结果的模板，`{{skillResult}}` 是 Skill 返回的 content，`{{userQuery}}` 是用户消息 |\n\n## 切换至 OpenRouter\n\n如果使用 OpenRouter 代替 SiliconFlow：\n\n```json\n{\n  \"NEXN2_BASE_URL\": \"https://openrouter.ai/api/v1\",\n  \"NEXN2_MODEL_NAME\": \"nex-agi/nex-n2-pro\"\n}\n```\n\n> 注意：OpenRouter 上的模型 ID 可能与 SiliconFlow 略有不同，以实际注册名称为准。\n\n## 不使用全局预处理\n\n如果仅需工具调用模式（非自动触发），保留 `skills.entries` 配置但**不添加** `multimodalPreprocess` 块，并在系统提示词中补充工具调用指令：\n\n> 当用户发送图片、截图、照片、文档截图时，请调用 analyze_image_by_nexn2 工具获取图片的文字与内容描述，再基于返回结果作答。\n\nFile v1.0.3:skill-card.md\n\n## Description: <br>\nGhost Eye lets a text-only LLM analyze images through an OpenAI-compatible vision model, returning OCR text and a visual summary. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[hunter-crk](https://clawhub.ai/user/hunter-crk) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and agent builders use this skill to let text-only assistants process images, screenshots, photos, and document scans by extracting visible text and summarizing scene content before answering the user. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Images may be sent to the configured third-party vision provider. <br>\nMitigation: Review or change NEXN2_BASE_URL before installation, and use the skill only with an approved provider and appropriate consent. <br>\nRisk: Automatic preprocessing can analyze images without an explicit tool call in the conversation. <br>\nMitigation: Prefer explicit tool-call mode when users need clear control over which images are analyzed. <br>\nRisk: OCR text and visual summaries may be cached locally by default. <br>\nMitigation: Disable caching for sensitive workflows or clear the cache before processing images that may contain secrets, personal data, financial data, or credentials. <br>\n\n\n## Reference(s): <br>\n- [OpenClaw multimodal preprocessing configuration](artifact/references/multimodal-config.md) <br>\n- [ClawHub skill page](https://clawhub.ai/hunter-crk/skills/ghost-eye) <br>\n- [SiliconFlow OpenAI-compatible API endpoint](https://api.siliconflow.cn/v1) <br>\n- [OpenRouter OpenAI-compatible API endpoint](https://openrouter.ai/api/v1) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, shell commands, configuration, guidance] <br>\n**Output Format:** [JSON response containing plain-text or Markdown OCR and visual-summary content] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Requires NEXN2_API_KEY; supports image path, URL, or small base64 input; may cache OCR and summaries locally for the configured TTL.] <br>\n\n## Skill Version(s): <br>\n1.0.3 (source: server release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.0.3:cache/c5393b1697c9e7598ce6d35871b8ad41.json\n\n{\"content\": \"【OCR文字提取】\\n\\n← Back to Ghost Eye\\n\\n# Skill settings\\n\\nIntegrations  \\nAutomation  \\nResearch  \\nDevelopment  \\nProductivity  \\nCommunication  \\nCreative  \\nKnowledge  \\nAgents  \\nOperations  \\nSecurity  \\nFinance  \\nLifestyle  \\n✓ Other\\n\\nUpdate skill files\\n\\n## Publish a new version\\n\\nUpload a replacement release for this skill. New releases get a fresh scan.\\n\\nNew Version\\n\\n## Short summary\\n\\nUpdate the short summary used in cards, search, and previews.\\n\\nages through any vision model. OCR + visual\\n\\nSave\\n\\n## Catalog metadata\\n\\nChoose browse categories and author topics for this skill.\\n\\nOther\\n\\n## TOPICS\\n\\nAdd a topic\\n\\n【画面内容总结】\\n\\n1. 核心主题：  \\n   图片展示的是一个名为 “Skill settings” 的技能设置页面，用户正在编辑或配置某个技能的基本信息、分类、摘要、版本发布和主题等元数据。\\n\\n2. 元素与布局：  \\n   - 页面左上角有返回入口 “Back to Ghost Eye”。  \\n   - 主标题为 “Skill settings”。  \\n   - 页面主要分为左侧设置说明区和右侧操作/输入区。  \\n   - 中间偏上位置弹出了一个分类选择下拉菜单，包含 Integrations、Automation、Research、Development、Productivity、Communication、Creative、Knowledge、Agents、Operations、Security、Finance、Lifestyle、Other 等分类。  \\n   - 下拉菜单中 “Automation” 被红色边框高亮，表示当前鼠标悬停或正在选择该项。  \\n   - 下方当前选中的分类显示为 “Other”，并带有勾选标记。  \\n   - 页面右侧有 “Update skill files”、“New Version”、“Save” 等操作按钮。  \\n   - 页面底部有 “TOPICS” 主题输入框，占位文字为 “Add a topic”。\\n\\n3. 关键信息提炼：  \\n   - 当前页面用于配置技能设置。  \\n   - 用户正在从分类下拉菜单中选择 “Automation”，但当前已选分类仍显示为 “Other”。  \\n   - 页面支持发布新版本、更新技能文件、编辑短摘要、设置目录分类和添加主题。  \\n   - “Automation” 是下拉菜单中被重点标记的选项，说明用户可能准备将技能分类从 “Other” 改为 “Automation”。\", \"cached_at\": \"2026-07-09T15:38:45.194194+00:00\", \"expires_at\": \"2026-07-16T15:38:45.194194+00:00\", \"model\": \"nex-agi/Nex-N2-Pro\", \"tokens_used\": 4393}\n\nArchive v1.0.2: 5 files, 9688 bytes\n\nFiles: references/multimodal-config.md (1933b), scripts/analyze.py (12005b), skill-card.md (2440b), SKILL.md (4853b), _meta.json (128b)\n\nFile v1.0.2:SKILL.md\n\n---\nname: \"ghost-eye\"\ndescription: \"Ghost Eye 👁️ — Let any pure-text LLM see images through any vision model. OCR + visual summary in one shot.\"\nmetadata:\n  {\n    \"openclaw\":\n      {\n        \"emoji\": \"👁️\",\n        \"requires\": { \"env\": [\"NEXN2_API_KEY\"] },\n        \"primaryEnv\": \"NEXN2_API_KEY\",\n      },\n  }\n---\n\n# Ghost Eye 👁️\n\nGive your text-only LLM the power to see. Ghost Eye is a lightweight image preprocessing bridge: when an image enters the conversation, it calls any OpenAI-compatible vision model (default: `nex-agi/Nex-N2-Pro`), produces a structured plain-text output (full OCR + visual summary), and feeds it back into the conversation — so your pure-text model can \"see\" without ever touching a multimodal API.\n\n## What it does\n\n```\nUser sends image → Ghost Eye detects → vision model analyzes → OCR text + scene summary → your LLM answers\n```\n\n## Two modes\n\n### Mode 1: Auto-preprocess (recommended)\nWhen `multimodalPreprocess` is configured, Ghost Eye fires automatically on any inbound image. The user never knows it's there — they just get answers about images.\n\n### Mode 2: Tool-call mode\nRegistered as `analyze_image_by_nexn2` tool (name stays for backward compatibility). Your LLM calls it explicitly when it sees an image. Add this to the system prompt:\n\n> When the user sends an image, screenshot, photo, or document scan, call the analyze_image_by_nexn2 tool to extract text and describe the image, then answer based on the returned content.\n\n## Workflow\n\n### Step 1: Receive image\n\n**Priority: `--image-path` > `--image-url` > `--image-base64`**\n\n⛔ Always prefer `--image-path` to avoid command-line `Argument list too long` errors with large base64 strings. Only fall back to `--image-url` or `--image-base64` when no local path is available.\n\n```bash\n# Preferred: local file path (no size limit)\npython3 {baseDir}/scripts/analyze.py --image-path \"<absolute path>\"\n\n# Fallback: public URL\npython3 {baseDir}/scripts/analyze.py --image-url \"<url>\"\n\n# Last resort: base64 (small images only, <50KB)\npython3 {baseDir}/scripts/analyze.py --image-base64 \"<base64>\"\n```\n\nThe script handles:\n- Format validation (JPG/PNG/WebP/GIF/BMP via magic bytes)\n- Cache check / read / write (MD5-based, 7-day TTL)\n- Image compression (Pillow, max 1920px longest edge, quality 85%)\n- API call with 1 automatic retry\n- Structured JSON output\n\n### Step 2: Parse JSON output\n\nSuccess:\n```json\n{\"success\": true, \"content\": \"【OCR文字提取】\\n...\\n\\n【画面内容总结】\\n...\", \"metadata\": {\"model\": \"...\", \"tokens_used\": 1200, \"cached\": false, \"process_time_ms\": 1500}}\n```\n\nError:\n```json\n{\"success\": false, \"content\": \"error message\", \"metadata\": {}}\n```\n\nPass `content` directly into the LLM conversation context.\n\n### Step 3: Caching\n\n- Cache directory: `{baseDir}/cache/` (auto-created)\n- Cache key: MD5 hash of raw image bytes\n- TTL: 7 days (`NEXN2_CACHE_TTL_DAYS`)\n- Clear cache: delete all `.json` files in `{baseDir}/cache/`\n- Toggle: `NEXN2_CACHE_ENABLE=true/false`\n\n## Environment variables\n\n| Variable | Required | Default |\n|----------|----------|---------|\n| NEXN2_API_KEY | ✅ Yes | — |\n| NEXN2_BASE_URL | No | https://api.siliconflow.cn/v1 |\n| NEXN2_MODEL_NAME | No | nex-agi/Nex-N2-Pro |\n| NEXN2_PROMPT_TEMPLATE | No | Built-in structured template |\n| NEXN2_IMAGE_COMPRESS | No | true |\n| NEXN2_CACHE_ENABLE | No | true |\n| NEXN2_CACHE_TTL_DAYS | No | 7 |\n| NEXN2_TIMEOUT_MS | No | 30000 |\n\nIf `NEXN2_API_KEY` is not set, returns a friendly Chinese error message.\n\n## Error handling\n\n| Scenario | Returns |\n|----------|---------|\n| Network/API failure (after retry) | Friendly \"service unavailable\" message |\n| Unsupported format | \"Please use JPG/PNG/WebP format\" |\n| Content safety block | \"Image flagged by safety filter\" |\n| Empty model output | \"No content returned, try a different image\" |\n\n⚠️ Errors never crash the conversation — structured JSON response is always returned.\n\n## Setup\n\nIn `openclaw.json` under `skills.entries`:\n\n```json5\n\"ghost-eye\": {\n  \"enabled\": true,\n  \"apiKey\": { \"source\": \"env\", \"provider\": \"default\", \"id\": \"NEXN2_API_KEY\" },\n  \"env\": {\n    \"NEXN2_BASE_URL\": \"https://api.siliconflow.cn/v1\",\n    \"NEXN2_MODEL_NAME\": \"nex-agi/Nex-N2-Pro\",\n    \"NEXN2_IMAGE_COMPRESS\": \"true\",\n    \"NEXN2_CACHE_ENABLE\": \"true\",\n    \"NEXN2_CACHE_TTL_DAYS\": \"7\",\n    \"NEXN2_TIMEOUT_MS\": \"30000\"\n  }\n}\n```\n\nFor `multimodalPreprocess` auto-mode, see `references/multimodal-config.md`.\n\n## Supported platforms\n\n- SiliconFlow (default, China-accessible)\n- OpenRouter\n- Any OpenAI-compatible chat completions endpoint\n\n## Safety\n\n- ⛔ Image base64 is never logged or written to conversation text\n- ⛔ API key is never hardcoded\n- ⛔ Temporary files are cleaned up immediately after processing\n- ⚠️ Output is always plain text / Markdown — never binary, never images\n\nFile v1.0.2:_meta.json\n\n{\n  \"ownerId\": \"kn78pm8jn9q0yydy5e2j0k169x8a7evs\",\n  \"slug\": \"ghost-eye\",\n  \"version\": \"1.0.2\",\n  \"publishedAt\": 1783611414275\n}\n\nFile v1.0.2:references/multimodal-config.md\n\n# OpenClaw 多模态预处理配置参考\n\n## multimodalPreprocess 配置\n\n在 `openclaw.json` 中添加：\n\n```json\n{\n  \"multimodalPreprocess\": {\n    \"enable\": true,\n    \"visionSkillId\": \"nex-n2-image-analyzer\",\n    \"promptTemplate\": \"以下是图片的完整分析结果，请严格基于该内容回答用户问题：\\n{{skillResult}}\\n\\n用户问题：{{userQuery}}\"\n  },\n  \"skills\": {\n    \"entries\": {\n      \"nex-n2-image-analyzer\": {\n        \"enabled\": true,\n        \"apiKey\": {\n          \"source\": \"env\",\n          \"provider\": \"default\",\n          \"id\": \"NEXN2_API_KEY\"\n        },\n        \"env\": {\n          \"NEXN2_BASE_URL\": \"https://api.siliconflow.cn/v1\",\n          \"NEXN2_MODEL_NAME\": \"nex-agi/Nex-N2-Pro\",\n          \"NEXN2_IMAGE_COMPRESS\": \"true\",\n          \"NEXN2_CACHE_ENABLE\": \"true\",\n          \"NEXN2_CACHE_TTL_DAYS\": \"7\",\n          \"NEXN2_TIMEOUT_MS\": \"30000\"\n        }\n      }\n    }\n  }\n}\n```\n\n## 配置项说明\n\n| 字段 | 说明 |\n|------|------|\n| `multimodalPreprocess.enable` | 开启全局图片预处理 |\n| `multimodalPreprocess.visionSkillId` | 指定处理图片的 Skill 名称 |\n| `multimodalPreprocess.promptTemplate` | 拼接结果的模板，`{{skillResult}}` 是 Skill 返回的 content，`{{userQuery}}` 是用户消息 |\n\n## 切换至 OpenRouter\n\n如果使用 OpenRouter 代替 SiliconFlow：\n\n```json\n{\n  \"NEXN2_BASE_URL\": \"https://openrouter.ai/api/v1\",\n  \"NEXN2_MODEL_NAME\": \"nex-agi/nex-n2-pro\"\n}\n```\n\n> 注意：OpenRouter 上的模型 ID 可能与 SiliconFlow 略有不同，以实际注册名称为准。\n\n## 不使用全局预处理\n\n如果仅需工具调用模式（非自动触发），保留 `skills.entries` 配置但**不添加** `multimodalPreprocess` 块，并在系统提示词中补充工具调用指令：\n\n> 当用户发送图片、截图、照片、文档截图时，请调用 analyze_image_by_nexn2 工具获取图片的文字与内容描述，再基于返回结果作答。\n\nFile v1.0.2:skill-card.md\n\n## Description: <br>\nGhost Eye lets pure-text LLMs analyze images by calling an OpenAI-compatible vision model and returning OCR text plus a visual summary. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[hunter-crk](https://clawhub.ai/user/hunter-crk) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and agent operators use Ghost Eye to add image OCR and scene summaries to text-only LLM workflows through automatic multimodal preprocessing or explicit tool calls. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Images may be sent automatically to the configured third-party vision provider without per-image user notice. <br>\nMitigation: Use tool-call/manual mode for sensitive workflows and review the provider endpoint, API key handling, and retention policies before enabling global image preprocessing. <br>\nRisk: OCR text and visual summaries may be cached and can contain private information from submitted images. <br>\nMitigation: Disable caching or shorten NEXN2_CACHE_TTL_DAYS for sensitive use cases, and clear cached JSON files when needed. <br>\n\n\n## Reference(s): <br>\n- [Ghost Eye ClawHub Release](https://clawhub.ai/hunter-crk/skills/ghost-eye) <br>\n- [hunter-crk Publisher Profile](https://clawhub.ai/user/hunter-crk) <br>\n- [OpenClaw Multimodal Preprocess Configuration](references/multimodal-config.md) <br>\n- [SiliconFlow OpenAI-Compatible API Endpoint](https://api.siliconflow.cn/v1) <br>\n- [OpenRouter OpenAI-Compatible API Endpoint](https://openrouter.ai/api/v1) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, JSON, shell commands, configuration, guidance] <br>\n**Output Format:** [Structured JSON with success, content, and metadata fields; content is plain text or Markdown OCR and image summary text.] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Accepts local image paths, image URLs, or small base64 images; requires NEXN2_API_KEY; caching is enabled by default with a 7-day TTL.] <br>\n\n## Skill Version(s): <br>\n1.0.2 (source: evidence.release.version) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.0.1: 5 files, 9702 bytes\n\nFiles: references/multimodal-config.md (1933b), scripts/analyze.py (12005b), skill-card.md (2524b), SKILL.md (4853b), _meta.json (128b)\n\nFile v1.0.1:SKILL.md\n\n---\nname: \"ghost-eye\"\ndescription: \"Ghost Eye 👁️ — Let any pure-text LLM see images through any vision model. OCR + visual summary in one shot.\"\nmetadata:\n  {\n    \"openclaw\":\n      {\n        \"emoji\": \"👁️\",\n        \"requires\": { \"env\": [\"NEXN2_API_KEY\"] },\n        \"primaryEnv\": \"NEXN2_API_KEY\",\n      },\n  }\n---\n\n# Ghost Eye 👁️\n\nGive your text-only LLM the power to see. Ghost Eye is a lightweight image preprocessing bridge: when an image enters the conversation, it calls any OpenAI-compatible vision model (default: `nex-agi/Nex-N2-Pro`), produces a structured plain-text output (full OCR + visual summary), and feeds it back into the conversation — so your pure-text model can \"see\" without ever touching a multimodal API.\n\n## What it does\n\n```\nUser sends image → Ghost Eye detects → vision model analyzes → OCR text + scene summary → your LLM answers\n```\n\n## Two modes\n\n### Mode 1: Auto-preprocess (recommended)\nWhen `multimodalPreprocess` is configured, Ghost Eye fires automatically on any inbound image. The user never knows it's there — they just get answers about images.\n\n### Mode 2: Tool-call mode\nRegistered as `analyze_image_by_nexn2` tool (name stays for backward compatibility). Your LLM calls it explicitly when it sees an image. Add this to the system prompt:\n\n> When the user sends an image, screenshot, photo, or document scan, call the analyze_image_by_nexn2 tool to extract text and describe the image, then answer based on the returned content.\n\n## Workflow\n\n### Step 1: Receive image\n\n**Priority: `--image-path` > `--image-url` > `--image-base64`**\n\n⛔ Always prefer `--image-path` to avoid command-line `Argument list too long` errors with large base64 strings. Only fall back to `--image-url` or `--image-base64` when no local path is available.\n\n```bash\n# Preferred: local file path (no size limit)\npython3 {baseDir}/scripts/analyze.py --image-path \"<absolute path>\"\n\n# Fallback: public URL\npython3 {baseDir}/scripts/analyze.py --image-url \"<url>\"\n\n# Last resort: base64 (small images only, <50KB)\npython3 {baseDir}/scripts/analyze.py --image-base64 \"<base64>\"\n```\n\nThe script handles:\n- Format validation (JPG/PNG/WebP/GIF/BMP via magic bytes)\n- Cache check / read / write (MD5-based, 7-day TTL)\n- Image compression (Pillow, max 1920px longest edge, quality 85%)\n- API call with 1 automatic retry\n- Structured JSON output\n\n### Step 2: Parse JSON output\n\nSuccess:\n```json\n{\"success\": true, \"content\": \"【OCR文字提取】\\n...\\n\\n【画面内容总结】\\n...\", \"metadata\": {\"model\": \"...\", \"tokens_used\": 1200, \"cached\": false, \"process_time_ms\": 1500}}\n```\n\nError:\n```json\n{\"success\": false, \"content\": \"error message\", \"metadata\": {}}\n```\n\nPass `content` directly into the LLM conversation context.\n\n### Step 3: Caching\n\n- Cache directory: `{baseDir}/cache/` (auto-created)\n- Cache key: MD5 hash of raw image bytes\n- TTL: 7 days (`NEXN2_CACHE_TTL_DAYS`)\n- Clear cache: delete all `.json` files in `{baseDir}/cache/`\n- Toggle: `NEXN2_CACHE_ENABLE=true/false`\n\n## Environment variables\n\n| Variable | Required | Default |\n|----------|----------|---------|\n| NEXN2_API_KEY | ✅ Yes | — |\n| NEXN2_BASE_URL | No | https://api.siliconflow.cn/v1 |\n| NEXN2_MODEL_NAME | No | nex-agi/Nex-N2-Pro |\n| NEXN2_PROMPT_TEMPLATE | No | Built-in structured template |\n| NEXN2_IMAGE_COMPRESS | No | true |\n| NEXN2_CACHE_ENABLE | No | true |\n| NEXN2_CACHE_TTL_DAYS | No | 7 |\n| NEXN2_TIMEOUT_MS | No | 30000 |\n\nIf `NEXN2_API_KEY` is not set, returns a friendly Chinese error message.\n\n## Error handling\n\n| Scenario | Returns |\n|----------|---------|\n| Network/API failure (after retry) | Friendly \"service unavailable\" message |\n| Unsupported format | \"Please use JPG/PNG/WebP format\" |\n| Content safety block | \"Image flagged by safety filter\" |\n| Empty model output | \"No content returned, try a different image\" |\n\n⚠️ Errors never crash the conversation — structured JSON response is always returned.\n\n## Setup\n\nIn `openclaw.json` under `skills.entries`:\n\n```json5\n\"ghost-eye\": {\n  \"enabled\": true,\n  \"apiKey\": { \"source\": \"env\", \"provider\": \"default\", \"id\": \"NEXN2_API_KEY\" },\n  \"env\": {\n    \"NEXN2_BASE_URL\": \"https://api.siliconflow.cn/v1\",\n    \"NEXN2_MODEL_NAME\": \"nex-agi/Nex-N2-Pro\",\n    \"NEXN2_IMAGE_COMPRESS\": \"true\",\n    \"NEXN2_CACHE_ENABLE\": \"true\",\n    \"NEXN2_CACHE_TTL_DAYS\": \"7\",\n    \"NEXN2_TIMEOUT_MS\": \"30000\"\n  }\n}\n```\n\nFor `multimodalPreprocess` auto-mode, see `references/multimodal-config.md`.\n\n## Supported platforms\n\n- SiliconFlow (default, China-accessible)\n- OpenRouter\n- Any OpenAI-compatible chat completions endpoint\n\n## Safety\n\n- ⛔ Image base64 is never logged or written to conversation text\n- ⛔ API key is never hardcoded\n- ⛔ Temporary files are cleaned up immediately after processing\n- ⚠️ Output is always plain text / Markdown — never binary, never images\n\nFile v1.0.1:_meta.json\n\n{\n  \"ownerId\": \"kn78pm8jn9q0yydy5e2j0k169x8a7evs\",\n  \"slug\": \"ghost-eye\",\n  \"version\": \"1.0.1\",\n  \"publishedAt\": 1783611351228\n}\n\nFile v1.0.1:references/multimodal-config.md\n\n# OpenClaw 多模态预处理配置参考\n\n## multimodalPreprocess 配置\n\n在 `openclaw.json` 中添加：\n\n```json\n{\n  \"multimodalPreprocess\": {\n    \"enable\": true,\n    \"visionSkillId\": \"nex-n2-image-analyzer\",\n    \"promptTemplate\": \"以下是图片的完整分析结果，请严格基于该内容回答用户问题：\\n{{skillResult}}\\n\\n用户问题：{{userQuery}}\"\n  },\n  \"skills\": {\n    \"entries\": {\n      \"nex-n2-image-analyzer\": {\n        \"enabled\": true,\n        \"apiKey\": {\n          \"source\": \"env\",\n          \"provider\": \"default\",\n          \"id\": \"NEXN2_API_KEY\"\n        },\n        \"env\": {\n          \"NEXN2_BASE_URL\": \"https://api.siliconflow.cn/v1\",\n          \"NEXN2_MODEL_NAME\": \"nex-agi/Nex-N2-Pro\",\n          \"NEXN2_IMAGE_COMPRESS\": \"true\",\n          \"NEXN2_CACHE_ENABLE\": \"true\",\n          \"NEXN2_CACHE_TTL_DAYS\": \"7\",\n          \"NEXN2_TIMEOUT_MS\": \"30000\"\n        }\n      }\n    }\n  }\n}\n```\n\n## 配置项说明\n\n| 字段 | 说明 |\n|------|------|\n| `multimodalPreprocess.enable` | 开启全局图片预处理 |\n| `multimodalPreprocess.visionSkillId` | 指定处理图片的 Skill 名称 |\n| `multimodalPreprocess.promptTemplate` | 拼接结果的模板，`{{skillResult}}` 是 Skill 返回的 content，`{{userQuery}}` 是用户消息 |\n\n## 切换至 OpenRouter\n\n如果使用 OpenRouter 代替 SiliconFlow：\n\n```json\n{\n  \"NEXN2_BASE_URL\": \"https://openrouter.ai/api/v1\",\n  \"NEXN2_MODEL_NAME\": \"nex-agi/nex-n2-pro\"\n}\n```\n\n> 注意：OpenRouter 上的模型 ID 可能与 SiliconFlow 略有不同，以实际注册名称为准。\n\n## 不使用全局预处理\n\n如果仅需工具调用模式（非自动触发），保留 `skills.entries` 配置但**不添加** `multimodalPreprocess` 块，并在系统提示词中补充工具调用指令：\n\n> 当用户发送图片、截图、照片、文档截图时，请调用 analyze_image_by_nexn2 工具获取图片的文字与内容描述，再基于返回结果作答。\n\nFile v1.0.1:skill-card.md\n\n## Description: <br>\nGhost Eye lets a text-only LLM process images by sending them to an OpenAI-compatible vision model and returning OCR text plus a visual summary. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[hunter-crk](https://clawhub.ai/user/hunter-crk) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and agent builders use Ghost Eye to add image OCR and scene-summary preprocessing to text-only LLM workflows. It can run as automatic multimodal preprocessing or as an explicit image-analysis tool call. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Silent global preprocessing can send inbound images to an external vision provider without clear user awareness. <br>\nMitigation: Use explicit tool-call mode or inform users before enabling automatic preprocessing, especially in shared or sensitive environments. <br>\nRisk: Analyzed image content may be cached locally for the configured TTL. <br>\nMitigation: Disable caching or shorten NEXN2_CACHE_TTL_DAYS for confidential images, and clear the cache after sensitive sessions. <br>\nRisk: The configured default endpoint determines where image data is processed. <br>\nMitigation: Pin NEXN2_BASE_URL and NEXN2_MODEL_NAME to an approved provider and model before deployment. <br>\n\n\n## Reference(s): <br>\n- [OpenClaw multimodal preprocessing configuration](artifact/references/multimodal-config.md) <br>\n- [SiliconFlow OpenAI-compatible API endpoint](https://api.siliconflow.cn/v1) <br>\n- [OpenRouter OpenAI-compatible API endpoint](https://openrouter.ai/api/v1) <br>\n- [Ghost Eye ClawHub listing](https://clawhub.ai/hunter-crk/skills/ghost-eye) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, json, shell commands, configuration, guidance] <br>\n**Output Format:** [JSON response with success status, OCR and visual-summary text, and processing metadata] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Requires NEXN2_API_KEY; supports local path, URL, or small base64 image input; may cache analyzed image content for the configured TTL.] <br>\n\n## Skill Version(s): <br>\n1.0.1 (source: server release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.0.0: 5 files, 9608 bytes\n\nFiles: references/multimodal-config.md (1933b), scripts/analyze.py (12005b), skill-card.md (2296b), SKILL.md (4788b), _meta.json (128b)\n\nFile v1.0.0:SKILL.md\n\n---\nname: \"ghost-eye\"\ndescription: \"Ghost Eye 👁️ — Let any pure-text LLM see images through Nex-N2-Pro vision model. OCR + visual summary in one shot.\"\nmetadata:\n  {\n    \"openclaw\":\n      {\n        \"emoji\": \"👁️\",\n        \"requires\": { \"env\": [\"NEXN2_API_KEY\"] },\n        \"primaryEnv\": \"NEXN2_API_KEY\",\n      },\n  }\n---\n\n# Ghost Eye 👁️\n\nGive your text-only LLM the power to see. Ghost Eye is a lightweight image preprocessing bridge: when an image enters the conversation, it calls the `nex-agi/Nex-N2-Pro` multimodal model, produces a structured plain-text output (full OCR + visual summary), and feeds it back into the conversation — so your pure-text model can \"see\" without ever touching a multimodal API.\n\n## What it does\n\n```\nUser sends image → Ghost Eye detects → Nex-N2-Pro analyzes → OCR text + scene summary → your LLM answers\n```\n\n## Two modes\n\n### Mode 1: Auto-preprocess (recommended)\nWhen `multimodalPreprocess` is configured, Ghost Eye fires automatically on any inbound image. The user never knows it's there — they just get answers about images.\n\n### Mode 2: Tool-call mode\nRegistered as `analyze_image_by_nexn2`. Your LLM calls it explicitly when it sees an image. Add this to the system prompt:\n\n> When the user sends an image, screenshot, photo, or document scan, call the analyze_image_by_nexn2 tool to extract text and describe the image, then answer based on the returned content.\n\n## Workflow\n\n### Step 1: Receive image\n\n**Priority: `--image-path` > `--image-url` > `--image-base64`**\n\n⛔ Always prefer `--image-path` to avoid command-line `Argument list too long` errors with large base64 strings. Only fall back to `--image-url` or `--image-base64` when no local path is available.\n\n```bash\n# Preferred: local file path (no size limit)\npython3 {baseDir}/scripts/analyze.py --image-path \"<absolute path>\"\n\n# Fallback: public URL\npython3 {baseDir}/scripts/analyze.py --image-url \"<url>\"\n\n# Last resort: base64 (small images only, <50KB)\npython3 {baseDir}/scripts/analyze.py --image-base64 \"<base64>\"\n```\n\nThe script handles:\n- Format validation (JPG/PNG/WebP/GIF/BMP via magic bytes)\n- Cache check / read / write (MD5-based, 7-day TTL)\n- Image compression (Pillow, max 1920px longest edge, quality 85%)\n- API call with 1 automatic retry\n- Structured JSON output\n\n### Step 2: Parse JSON output\n\nSuccess:\n```json\n{\"success\": true, \"content\": \"【OCR文字提取】\\n...\\n\\n【画面内容总结】\\n...\", \"metadata\": {\"model\": \"...\", \"tokens_used\": 1200, \"cached\": false, \"process_time_ms\": 1500}}\n```\n\nError:\n```json\n{\"success\": false, \"content\": \"error message\", \"metadata\": {}}\n```\n\nPass `content` directly into the LLM conversation context.\n\n### Step 3: Caching\n\n- Cache directory: `{baseDir}/cache/` (auto-created)\n- Cache key: MD5 hash of raw image bytes\n- TTL: 7 days (`NEXN2_CACHE_TTL_DAYS`)\n- Clear cache: delete all `.json` files in `{baseDir}/cache/`\n- Toggle: `NEXN2_CACHE_ENABLE=true/false`\n\n## Environment variables\n\n| Variable | Required | Default |\n|----------|----------|---------|\n| NEXN2_API_KEY | ✅ Yes | — |\n| NEXN2_BASE_URL | No | https://api.siliconflow.cn/v1 |\n| NEXN2_MODEL_NAME | No | nex-agi/Nex-N2-Pro |\n| NEXN2_PROMPT_TEMPLATE | No | Built-in structured template |\n| NEXN2_IMAGE_COMPRESS | No | true |\n| NEXN2_CACHE_ENABLE | No | true |\n| NEXN2_CACHE_TTL_DAYS | No | 7 |\n| NEXN2_TIMEOUT_MS | No | 30000 |\n\nIf `NEXN2_API_KEY` is not set, returns a friendly Chinese error message.\n\n## Error handling\n\n| Scenario | Returns |\n|----------|---------|\n| Network/API failure (after retry) | Friendly \"service unavailable\" message |\n| Unsupported format | \"Please use JPG/PNG/WebP format\" |\n| Content safety block | \"Image flagged by safety filter\" |\n| Empty model output | \"No content returned, try a different image\" |\n\n⚠️ Errors never crash the conversation — structured JSON response is always returned.\n\n## Setup\n\nIn `openclaw.json` under `skills.entries`:\n\n```json5\n\"ghost-eye\": {\n  \"enabled\": true,\n  \"apiKey\": { \"source\": \"env\", \"provider\": \"default\", \"id\": \"NEXN2_API_KEY\" },\n  \"env\": {\n    \"NEXN2_BASE_URL\": \"https://api.siliconflow.cn/v1\",\n    \"NEXN2_MODEL_NAME\": \"nex-agi/Nex-N2-Pro\",\n    \"NEXN2_IMAGE_COMPRESS\": \"true\",\n    \"NEXN2_CACHE_ENABLE\": \"true\",\n    \"NEXN2_CACHE_TTL_DAYS\": \"7\",\n    \"NEXN2_TIMEOUT_MS\": \"30000\"\n  }\n}\n```\n\nFor `multimodalPreprocess` auto-mode, see `references/multimodal-config.md`.\n\n## Supported platforms\n\n- SiliconFlow (default, China-accessible)\n- OpenRouter\n- Any OpenAI-compatible chat completions endpoint\n\n## Safety\n\n- ⛔ Image base64 is never logged or written to conversation text\n- ⛔ API key is never hardcoded\n- ⛔ Temporary files are cleaned up immediately after processing\n- ⚠️ Output is always plain text / Markdown — never binary, never images\n\nFile v1.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn78pm8jn9q0yydy5e2j0k169x8a7evs\",\n  \"slug\": \"ghost-eye\",\n  \"version\": \"1.0.0\",\n  \"publishedAt\": 1783611197319\n}\n\nFile v1.0.0:references/multimodal-config.md\n\n# OpenClaw 多模态预处理配置参考\n\n## multimodalPreprocess 配置\n\n在 `openclaw.json` 中添加：\n\n```json\n{\n  \"multimodalPreprocess\": {\n    \"enable\": true,\n    \"visionSkillId\": \"nex-n2-image-analyzer\",\n    \"promptTemplate\": \"以下是图片的完整分析结果，请严格基于该内容回答用户问题：\\n{{skillResult}}\\n\\n用户问题：{{userQuery}}\"\n  },\n  \"skills\": {\n    \"entries\": {\n      \"nex-n2-image-analyzer\": {\n        \"enabled\": true,\n        \"apiKey\": {\n          \"source\": \"env\",\n          \"provider\": \"default\",\n          \"id\": \"NEXN2_API_KEY\"\n        },\n        \"env\": {\n          \"NEXN2_BASE_URL\": \"https://api.siliconflow.cn/v1\",\n          \"NEXN2_MODEL_NAME\": \"nex-agi/Nex-N2-Pro\",\n          \"NEXN2_IMAGE_COMPRESS\": \"true\",\n          \"NEXN2_CACHE_ENABLE\": \"true\",\n          \"NEXN2_CACHE_TTL_DAYS\": \"7\",\n          \"NEXN2_TIMEOUT_MS\": \"30000\"\n        }\n      }\n    }\n  }\n}\n```\n\n## 配置项说明\n\n| 字段 | 说明 |\n|------|------|\n| `multimodalPreprocess.enable` | 开启全局图片预处理 |\n| `multimodalPreprocess.visionSkillId` | 指定处理图片的 Skill 名称 |\n| `multimodalPreprocess.promptTemplate` | 拼接结果的模板，`{{skillResult}}` 是 Skill 返回的 content，`{{userQuery}}` 是用户消息 |\n\n## 切换至 OpenRouter\n\n如果使用 OpenRouter 代替 SiliconFlow：\n\n```json\n{\n  \"NEXN2_BASE_URL\": \"https://openrouter.ai/api/v1\",\n  \"NEXN2_MODEL_NAME\": \"nex-agi/nex-n2-pro\"\n}\n```\n\n> 注意：OpenRouter 上的模型 ID 可能与 SiliconFlow 略有不同，以实际注册名称为准。\n\n## 不使用全局预处理\n\n如果仅需工具调用模式（非自动触发），保留 `skills.entries` 配置但**不添加** `multimodalPreprocess` 块，并在系统提示词中补充工具调用指令：\n\n> 当用户发送图片、截图、照片、文档截图时，请调用 analyze_image_by_nexn2 工具获取图片的文字与内容描述，再基于返回结果作答。\n\nFile v1.0.0:skill-card.md\n\n## Description: <br>\nGhost Eye lets pure-text LLMs process images by using the Nex-N2-Pro vision model to produce OCR text and a visual summary. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[hunter-crk](https://clawhub.ai/user/hunter-crk) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and external users can add image understanding to text-only agent workflows. The skill analyzes image paths, URLs, or base64 input and returns OCR plus a concise scene summary for use in the conversation context. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Images, screenshots, or document scans may be sent to a configured third-party vision API. <br>\nMitigation: Use only approved providers, document provider retention expectations, and prefer explicit tool-call mode for sensitive workflows. <br>\nRisk: Extracted image text is cached by default. <br>\nMitigation: Disable caching or reduce cache retention for sensitive environments, and clear cached JSON files when retention is not appropriate. <br>\nRisk: URL-based image input can fetch remote content. <br>\nMitigation: Prefer local file paths and restrict or avoid URL-based image fetching in controlled deployments. <br>\n\n\n## Reference(s): <br>\n- [OpenClaw multimodal preprocessing configuration](references/multimodal-config.md) <br>\n- [ClawHub Ghost Eye release page](https://clawhub.ai/hunter-crk/skills/ghost-eye) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, json, configuration, shell commands] <br>\n**Output Format:** [Structured JSON containing plain-text or Markdown OCR and visual-summary content.] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Includes success status and metadata such as model name, token count, cache status, and processing time; API responses are capped at 4096 tokens by the script.] <br>\n\n## Skill Version(s): <br>\n1.0.0 (source: server release metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>","readmeExcerpt":"Skill: Ghost Eye Owner: hunter-crk Summary: Ghost Eye 👁️ — Let any pure-text LLM see images through any vision model. OCR + visual summary in one shot. Tags: image:1.0.2, latest:1.0.4, llm:1.0.2, ocr:1.0.2, vision:1.0.2 Version history: v1.0.4 | 2026-07-09T15:59:25.635Z | auto ghost-eye 1.0.4 - Added image analysis cache files for improved performance. - Removed the redundant skill-card.md documentation file. - Upda","codeSnippets":[],"executableExamples":[{"language":"text","snippet":"User sends image → Ghost Eye detects → vision model analyzes → OCR text + scene summary → your LLM answers"},{"language":"bash","snippet":"# Preferred: local file path (no size limit)\npython3 {baseDir}/scripts/analyze.py --image-path \"<absolute path>\"\n\n# Fallback: public URL\npython3 {baseDir}/scripts/analyze.py --image-url \"<url>\"\n\n# Last resort: base64 (small images only, <50KB)\npython3 {baseDir}/scripts/analyze.py --image-base64 \"<base64>\""},{"language":"json","snippet":"{\"success\": true, \"content\": \"【OCR文字提取】\\n...\\n\\n【画面内容总结】\\n...\", \"metadata\": {\"model\": \"...\", \"tokens_used\": 1200, \"cached\": false, \"process_time_ms\": 1500}}"},{"language":"json","snippet":"{\"success\": false, \"content\": \"error message\", \"metadata\": {}}"},{"language":"json5","snippet":"\"ghost-eye\": {\n  \"enabled\": true,\n  \"apiKey\": { \"source\": \"env\", \"provider\": \"default\", \"id\": \"NEXN2_API_KEY\" },\n  \"env\": {\n    \"NEXN2_BASE_URL\": \"https://api.siliconflow.cn/v1\",\n    \"NEXN2_MODEL_NAME\": \"nex-agi/Nex-N2-Pro\",\n    \"NEXN2_IMAGE_COMPRESS\": \"true\",\n    \"NEXN2_CACHE_ENABLE\": \"true\",\n    \"NEXN2_CACHE_TTL_DAYS\": \"7\",\n    \"NEXN2_TIMEOUT_MS\": \"30000\"\n  }\n}"},{"language":"json","snippet":"{\n  \"multimodalPreprocess\": {\n    \"enable\": true,\n    \"visionSkillId\": \"nex-n2-image-analyzer\",\n    \"promptTemplate\": \"以下是图片的完整分析结果，请严格基于该内容回答用户问题：\\n{{skillResult}}\\n\\n用户问题：{{userQuery}}\"\n  },\n  \"skills\": {\n    \"entries\": {\n      \"nex-n2-image-analyzer\": {\n        \"enabled\": true,\n        \"apiKey\": {\n          \"source\": \"env\",\n          \"provider\": \"default\",\n          \"id\": \"NEXN2_API_KEY\"\n        },\n        \"env\": {\n          \"NEXN2_BASE_URL\": \"https://api.siliconflow.cn/v1\",\n          \"NEXN2_MODEL_NAME\": \"nex-agi/Nex-N2-Pro\",\n          \"NEXN2_IMAGE_COMPRESS\": \"true\",\n          \"NEXN2_CACHE_ENABLE\": \"true\",\n          \"NEXN2_CACHE_TTL_DAYS\": \"7\",\n          \"NEXN2_TIMEOUT_MS\": \"30000\"\n        }\n      }\n    }\n  }\n}"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: \"ghost-eye\"\ndescription: \"Ghost Eye 👁️ — Let any pure-text LLM see images through any vision model. OCR + visual summary in one shot.\"\nmetadata:\n  {\n    \"openclaw\":\n      {\n        \"emoji\": \"👁️\",\n        \"requires\": { \"env\": [\"NEXN2_API_KEY\"] },\n        \"primaryEnv\": \"NEXN2_API_KEY\",\n      },\n  }\n---\n\n# Ghost Eye 👁️\n\nGive your text-only LLM the power to see. Ghost Eye is a lightweight image preprocessing bridge: when an image enters the conversation, it calls any OpenAI-compatible vision model (default: `nex-agi/Nex-N2-Pro`), produces a structured plain-text output (full OCR + visual summary), and feeds it back into the conversation — so your pure-text model can \"see\" without ever touching a multimodal API.\n\n## What it does\n\n```\nUser sends image → Ghost Eye detects → vision model analyzes → OCR text + scene summary → your LLM answers\n```\n\n## Two modes\n\n### Mode 1: Auto-preprocess (recommended)\nWhen `multimodalPreprocess` is configured, Ghost Eye fires automatically on any inbound image. The user never knows it's there — they just get answers about images.\n\n### Mode 2: Tool-call mode\nRegistered as `analyze_image_by_nexn2` tool (name stays for backward compatibility). Your LLM calls it explicitly when it sees an image. Add this to the system prompt:\n\n> When the user sends an image, screenshot, photo, or document scan, call the analyze_image_by_nexn2 tool to extract text and describe the image, then answer based on the returned content.\n\n## Workflow\n\n### Step 1: Receive image\n\n**Priority: `--image-path` > `--image-url` > `--image-base64`**\n\n⛔ Always prefer `--image-path` to avoid command-line `Argument list too long` errors with large base64 strings. Only fall back to `--image-url` or `--image-base64` when no local path is available.\n\n```bash\n# Preferred: local file path (no size limit)\npython3 {baseDir}/scripts/analyze.py --image-path \"<absolute path>\"\n\n# Fallback: public URL\npython3 {baseDir}/scripts/analyze.py --image-url \"<url>\"\n\n# Last resort: base64 (small images only, <50KB)\npython3 {baseDir}/scripts/analyze.py --image-base64 \"<base64>\"\n```\n\nThe script handles:\n- Format validation (JPG/PNG/WebP/GIF/BMP via magic bytes)\n- Cache check / read / write (MD5-based, 7-day TTL)\n- Image compression (Pillow, max 1920px longest edge, quality 85%)\n- API call with 1 automatic retry\n- Structured JSON output\n\n### Step 2: Parse JSON output\n\nSuccess:\n```json\n{\"success\": true, \"content\": \"【OCR文字提取】\\n...\\n\\n【画面内容总结】\\n...\", \"metadata\": {\"model\": \"...\", \"tokens_used\": 1200, \"cached\": false, \"process_time_ms\": 1500}}\n```\n\nError:\n```json\n{\"success\": false, \"content\": \"error message\", \"metadata\": {}}\n```\n\nPass `content` directly into the LLM conversation context.\n\n### Step 3: Caching\n\n- Cache directory: `{baseDir}/cache/` (auto-created)\n- Cache key: MD5 hash of raw image bytes\n- TTL: 7 days (`NEXN2_CACHE_TTL_DAYS`)\n- Clear cache: delete all `.json` files in `{baseDir}/cache/`\n- Toggle: `NEXN2_CACHE_ENABLE=true/false`\n\n## Environment variables\n\n| Variabl"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn78pm8jn9q0yydy5e2j0k169x8a7evs\",\n  \"slug\": \"ghost-eye\",\n  \"version\": \"1.0.4\",\n  \"publishedAt\": 1783612765635\n}"},{"path":"references/multimodal-config.md","content":"# OpenClaw 多模态预处理配置参考\n\n## multimodalPreprocess 配置\n\n在 `openclaw.json` 中添加：\n\n```json\n{\n  \"multimodalPreprocess\": {\n    \"enable\": true,\n    \"visionSkillId\": \"nex-n2-image-analyzer\",\n    \"promptTemplate\": \"以下是图片的完整分析结果，请严格基于该内容回答用户问题：\\n{{skillResult}}\\n\\n用户问题：{{userQuery}}\"\n  },\n  \"skills\": {\n    \"entries\": {\n      \"nex-n2-image-analyzer\": {\n        \"enabled\": true,\n        \"apiKey\": {\n          \"source\": \"env\",\n          \"provider\": \"default\",\n          \"id\": \"NEXN2_API_KEY\"\n        },\n        \"env\": {\n          \"NEXN2_BASE_URL\": \"https://api.siliconflow.cn/v1\",\n          \"NEXN2_MODEL_NAME\": \"nex-agi/Nex-N2-Pro\",\n          \"NEXN2_IMAGE_COMPRESS\": \"true\",\n          \"NEXN2_CACHE_ENABLE\": \"true\",\n          \"NEXN2_CACHE_TTL_DAYS\": \"7\",\n          \"NEXN2_TIMEOUT_MS\": \"30000\"\n        }\n      }\n    }\n  }\n}\n```\n\n## 配置项说明\n\n| 字段 | 说明 |\n|------|------|\n| `multimodalPreprocess.enable` | 开启全局图片预处理 |\n| `multimodalPreprocess.visionSkillId` | 指定处理图片的 Skill 名称 |\n| `multimodalPreprocess.promptTemplate` | 拼接结果的模板，`{{skillResult}}` 是 Skill 返回的 content，`{{userQuery}}` 是用户消息 |\n\n## 切换至 OpenRouter\n\n如果使用 OpenRouter 代替 SiliconFlow：\n\n```json\n{\n  \"NEXN2_BASE_URL\": \"https://openrouter.ai/api/v1\",\n  \"NEXN2_MODEL_NAME\": \"nex-agi/nex-n2-pro\"\n}\n```\n\n> 注意：OpenRouter 上的模型 ID 可能与 SiliconFlow 略有不同，以实际注册名称为准。\n\n## 不使用全局预处理\n\n如果仅需工具调用模式（非自动触发），保留 `skills.entries` 配置但**不添加** `multimodalPreprocess` 块，并在系统提示词中补充工具调用指令：\n\n> 当用户发送图片、截图、照片、文档截图时，请调用 analyze_image_by_nexn2 工具获取图片的文字与内容描述，再基于返回结果作答。"},{"path":"skill-card.md","content":"## Description:\n\nGhost Eye lets a pure-text LLM analyze images through an OpenAI-compatible vision model, returning OCR text and a visual summary.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[hunter-crk](https://clawhub.ai/user/hunter-crk)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and agents use this skill to convert incoming images, screenshots, photos, or scanned documents into text that a text-only LLM can reason over. It supports automatic image preprocessing or explicit tool-call use.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Inbound images can be sent automatically to a third-party vision API.\n\nMitigation: Use explicit tool-call mode for sensitive work and configure an approved OpenAI-compatible vision endpoint before deployment.\n\nRisk: OCR text and visual summaries are stored in plaintext cache files by default.\n\nMitigation: Disable caching or regularly clear the cache when images may contain personal, business, credential, or regulated data.\n\nRisk: Using image URLs can introduce uncontrolled network egress.\n\nMitigation: Prefer local image paths and use image URLs only when outbound network access is constrained and approved.\n\n## Reference(s):\n\n- [OpenClaw multimodal preprocessing configuration](references/multimodal-config.md)\n- [ClawHub Ghost Eye skill page](https://clawhub.ai/hunter-crk/skills/ghost-eye)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, JSON]\n\n**Output Format:** [JSON object with success, content, and metadata fields; content contains OCR text and a visual summary in plain text or Markdown.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Requires NEXN2_API_KEY; image analysis results may be cached as JSON files for up to 7 days by default.]\n\n## Skill Version(s):\n\n1.0.4 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."},{"path":"cache/2a0206456252d9926270d09e5fea3640.json","content":"{\"content\": \"【OCR文字提取】\\n\\n```text\\nopen\\n\\n弱酸性配方\\n添加酒精\\n\\n生产批号:20260530 13:57:03\\n限用日期:20280529 A060530A\\n\\n\\n象小家™EDI纯水湿巾\\n7道水净化工艺｜无刺激｜无酒精｜无荧光剂\\n\\n小象超市 自有品牌\\n\\n○产品名称:象小家™EDI纯水湿巾○净含量:90抽×1包○规格:150mm×200mm○主要成分:水刺无纺布,EDI纯水○生产批号及限期使用日期:见包装○保质期:两年○贮存条件:置于阴凉干燥处常温保存,避免阳光直射○使用方法:将盖子打开,揭开贴纸,抽出使用。建议使用完盖好贴纸和盖子,避免水分流失。○卫生标准:GB 15979○执行标准:GB/T 27728.1○委托商:北京象鲜科技有限公司○地址:北京市朝阳区小营北路15号院1号楼3层○联系方式:010-10107777○受委托商:浙江优全护理用品科技股份有限公司○地址:浙江省湖州市长兴县太湖街道陆汇路68号(代码:A)○卫生许可证号:浙卫消证字(2016)第002\\n```\\n\\n【画面内容总结】\\n\\n1. 核心主题：这是一包“象小家™EDI纯水湿巾”的包装实拍，用于日常清洁/擦拭，主打纯水、温和、无酒精等卖点。\\n\\n2. 元素与布局：图片主体为一包浅蓝色湿巾，放置在浅色木纹桌面上；包装上方是翻盖开口，可见“open”字样及“弱酸性配方”等宣传语；包装中部为产品名称和核心卖点；包装下方密集排列产品参数、成分、使用方法、执行标准、委托商和受委托商等信息。背景中有纸巾盒、杯子、绿色包装等物品，但处于虚化状态。\\n\\n3. 关键信息提炼：\\n   - 产品名称：象小家™EDI纯水湿巾。\\n   - 核心卖点：7道水净化工艺、无刺激、无酒精、无荧光剂。\\n   - 规格信息：90抽×1包，规格为150mm×200mm。\\n   - 主要成分：水刺无纺布、EDI纯水。\\n   - 日期信息：生产批号为20260530 13:57:03，限用日期为20280529。\", \"cached_at\": \"2026-07-09T15:51:51.620772+00:00\", \"expires_at\": \"2026-07-16T15:51:51.620772+00:00\", \"model\": \"nex-agi/Nex-N2-Pro\", \"tokens_used\": 6270}"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Ghost Eye 👁️ — Let any pure-text LLM see images through any vision model. OCR + visual summary in one shot. Skill: Ghost Eye Owner: hunter-crk Summary: Ghost Eye 👁️ — Let any pure-text LLM see images through any vision model. OCR + visual summary in one shot. Tags: image:1.0.2, latest:1.0.4, llm:1.0.2, ocr:1.0.2, vision:1.0.2 Version history: v1.0.4 | 2026-07-09T15:59:25.635Z | auto ghost-eye 1.0.4 - Added image analysis cache files for improved performance. - Removed the redundant skill-card.md documentation file. - Upda","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1269,"uniquenessScore":50,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T09:40:16.788Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T09:40:16.788Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T14:14:05.235Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}