{"id":"c8865a82-fe42-4fe7-a4e6-222c757175a9","entityType":"agent","slug":"clawhub-18072937735-smyx-visual-qa-analysis","name":"Large Model Visual Question Answering Skill | 大模型视觉问答技能","canonicalUrl":"https://www.xpersona.co/agent/clawhub-18072937735-smyx-visual-qa-analysis","canonicalPath":"/agent/clawhub-18072937735-smyx-visual-qa-analysis","generatedAt":"2026-10-09T21:52:49.455Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-09T15:38:45.006Z","emptyReason":null},"description":"Conducts open-ended Q&A on image content based on computer vision and large language models, supporting any questions to receive natural language responses. | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答 Skill: Large Model Visual Question Answering Skill | 大模型视觉问答技能 Owner: 18072937735 Summary: Conducts open-ended Q&A on image content based on computer vision and large language models, supporting any questions to receive natural language responses. | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答 Tags: latest:1.0.17 Version history: v1.0.17 | 2026-09-28T06:04:26.780Z | auto - Version bumped to 1.0.17. - Docum","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 2.4K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s17f8q65zg3y98t86jdg1177g583whq8:smyx-visual-qa-analysis","sourceUrl":"https://clawhub.ai/18072937735/smyx-visual-qa-analysis","homepage":"https://clawhub.ai/18072937735/skills/smyx-visual-qa-analysis","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/18072937735/smyx-visual-qa-analysis","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/18072937735/skills/smyx-visual-qa-analysis","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":68,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Conducts open-ended Q&A on image content based on computer vision and large language models, supporting any questions to receive natural language responses. | 大"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T15:38:45.006Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T15:38:45.006Z","emptyReason":null},"stars":null,"forks":null,"downloads":2378,"packageName":null,"latestVersion":"1.0.17","tractionLabel":"2.4K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T15:38:45.005Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T15:38:45.006Z","lastCrawledAt":"2026-10-09T15:38:45.005Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T15:38:45.005Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.17","createdAt":"2026-09-28T06:04:26.780Z","changelog":"- Version bumped to 1.0.17. - Documentation updated in SKILL.md. - Obsolete file skill-card.md removed. - Possible adjustments to configuration file (config.yaml).","fileCount":31,"zipByteSize":39582},{"version":"1.0.16","createdAt":"2026-09-14T17:07:30.046Z","changelog":"- Updated version to 1.0.16 in SKILL.md. - Removed the skill-card.md file. - No functional or workflow changes noted in documentation.","fileCount":31,"zipByteSize":39810},{"version":"1.0.15","createdAt":"2026-08-23T17:07:30.384Z","changelog":"- Removed the skill-card.md file from the project. - No changes to code or functionality; documentation only. - Skill usage and workflow remain unchanged.","fileCount":31,"zipByteSize":39716},{"version":"1.0.14","createdAt":"2026-08-19T06:40:39.574Z","changelog":"- Updated version number in SKILL.md from 1.0.11 to 1.0.13. - Removed the file skill-card.md. - Minor adjustments or updates to skills/smyx_common/scripts/config.yaml. - No functional or usage changes to end user behavior.","fileCount":31,"zipByteSize":40020},{"version":"1.0.13","createdAt":"2026-08-10T12:24:33.120Z","changelog":"- Removed the skill-card.md file. - No other changes were made to the code or documentation.","fileCount":31,"zipByteSize":39858},{"version":"1.0.12","createdAt":"2026-08-09T08:55:31.194Z","changelog":"- Updated SKILL.md content and incremented version to 1.0.11 - Removed skill-card.md file - No functional or feature changes—documentation only","fileCount":31,"zipByteSize":40007},{"version":"1.0.11","createdAt":"2026-07-28T13:53:29.384Z","changelog":"- Updated SKILL.md version from 1.0.7 to 1.0.10. - Removed the file skill-card.md. - No other functional or documentation changes noted.","fileCount":31,"zipByteSize":39772},{"version":"1.0.10","createdAt":"2026-07-15T08:08:32.172Z","changelog":"- Updated SKILL.md documentation to version 1.0.7, reflecting minor edits. - Removed the file skill-card.md. - Made adjustments in skills/smyx_common/scripts/util.py. - No breaking changes to core functionality; update focused on documentation and maintenance only.","fileCount":31,"zipByteSize":39856}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17f8q65zg3y98t86jdg1177g583whq8:smyx-visual-qa-analysis","setupComplexity":"low","setupSteps":["Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-18072937735-smyx-visual-qa-analysis/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-18072937735-smyx-visual-qa-analysis/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-18072937735-smyx-visual-qa-analysis/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-18072937735-smyx-visual-qa-analysis/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-18072937735-smyx-visual-qa-analysis/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-18072937735-smyx-visual-qa-analysis/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-09T21:52:49.450Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-18072937735-smyx-visual-qa-analysis/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-18072937735-smyx-visual-qa-analysis/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-18072937735-smyx-visual-qa-analysis/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-18072937735-smyx-visual-qa-analysis/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-09T15:38:45.006Z","emptyReason":null},"readme":"Skill: Large Model Visual Question Answering Skill | 大模型视觉问答技能\n\nOwner: 18072937735\n\nSummary: Conducts open-ended Q&A on image content based on computer vision and large language models, supporting any questions to receive natural language responses. | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答\n\nTags: latest:1.0.17\n\nVersion history:\n\nv1.0.17 | 2026-09-28T06:04:26.780Z | auto\n\n- Version bumped to 1.0.17.\n- Documentation updated in SKILL.md.\n- Obsolete file skill-card.md removed.\n- Possible adjustments to configuration file (config.yaml).\n\nv1.0.16 | 2026-09-14T17:07:30.046Z | auto\n\n- Updated version to 1.0.16 in SKILL.md.\n- Removed the skill-card.md file.\n- No functional or workflow changes noted in documentation.\n\nv1.0.15 | 2026-08-23T17:07:30.384Z | auto\n\n- Removed the skill-card.md file from the project.\n- No changes to code or functionality; documentation only.\n- Skill usage and workflow remain unchanged.\n\nv1.0.14 | 2026-08-19T06:40:39.574Z | auto\n\n- Updated version number in SKILL.md from 1.0.11 to 1.0.13.\n- Removed the file skill-card.md.\n- Minor adjustments or updates to skills/smyx_common/scripts/config.yaml.\n- No functional or usage changes to end user behavior.\n\nv1.0.13 | 2026-08-10T12:24:33.120Z | auto\n\n- Removed the skill-card.md file.\n- No other changes were made to the code or documentation.\n\nv1.0.12 | 2026-08-09T08:55:31.194Z | auto\n\n- Updated SKILL.md content and incremented version to 1.0.11\n- Removed skill-card.md file\n- No functional or feature changes—documentation only\n\nv1.0.11 | 2026-07-28T13:53:29.384Z | auto\n\n- Updated SKILL.md version from 1.0.7 to 1.0.10.\n- Removed the file skill-card.md.\n- No other functional or documentation changes noted.\n\nv1.0.10 | 2026-07-15T08:08:32.172Z | auto\n\n- Updated SKILL.md documentation to version 1.0.7, reflecting minor edits.\n- Removed the file skill-card.md.\n- Made adjustments in skills/smyx_common/scripts/util.py.\n- No breaking changes to core functionality; update focused on documentation and maintenance only.\n\nv1.0.9 | 2026-07-09T18:54:29.083Z | auto\n\n- Updated internal configuration in skills/smyx_common/scripts/config.py.\n- Removed the skill-card.md file.\n- No user-facing feature or documentation changes.\n\nv1.0.8 | 2026-07-03T02:44:29.952Z | auto\n\n**Changelog for smyx-visual-qa-analysis v1.0.8**\n\n- Refined cloud identity control: user open-id management is now fully internal and automated; user input for identity is no longer required or allowed.\n- Updated and restructured SKILL.md to highlight automated identity handling, privacy protection, and user-facing simplicity.\n- Enhanced skill documentation and workflow clarity, emphasizing API-only historical query enforcement and removal of identity prompts from user flow.\n- Removed legacy documentation and scripts related to manual API service and skill-card output.\n- Improved clarity of usage examples and operation steps in both code and documentation.\n\nv1.0.7 | 2026-06-21T18:54:26.068Z | auto\n\n- Removed the file skill-card.md from the project.\n- No changes to core skill logic or documentation content.\n- Version number remains unchanged in SKILL.md.\n\nv1.0.6 | 2026-06-17T16:39:24.439Z | auto\n\nVersion 1.0.6 summary:\nRefined skill architecture and configuration handling to support more flexible deployment.\n\n- Adjusted configuration file loading priority and paths for better modularity.\n- Unified dependency management by updating requirements and script imports.\n- Improved code structure in api_service.py, config.py, dao.py, and util.py to centralize shared functions.\n- Removed obsolete files (e.g., skill-card.md) for cleanup.\n- Updated documentation and metadata (SKILL.md) to reflect new license information and clarified open-id retrieval flow.\n- Enhanced compatibility for multi-environment setups by aligning with the latest smyx_common shared module layout.\n\nv1.0.5 | 2026-05-28T11:15:41.407Z | auto\n\nMajor structural and directory changes for better organization:\n\n- Added new `skills/smyx_analysis/` module directory with scripts, requirements, and references.\n- Removed the old `skills/face_analysis/` module and associated files.\n- Updated references, configuration, and script paths to support the new project structure.\n- Refined and reorganized code across configuration and utility scripts.\n\nv1.0.4 | 2026-05-21T03:27:37.787Z | auto\n\n- Fixed: Local image file uploads are now saved as local files instead of to the attachments directory.\n- Minor documentation update to reflect the change in image file storage location.\n- No other behavioral changes; general usage and API parameters remain the same.\n\nv1.0.3 | 2026-05-16T10:30:13.441Z | auto\n\n## v1.0.3 Changelog\n\n- Refactored and updated code in core utility files: skill.py, __init__.py, config.py, and util.py.\n- Improved internal logic and structure for better maintainability.\n- No changes to user-facing documentation or skill description.\n\nv1.0.2 | 2026-05-06T10:25:39.310Z | auto\n\n- Improved logic in skills/smyx_common/scripts/skill.py for handling open-id retrieval and validation.\n- Enhanced error handling to ensure open-id is never assumed or auto-generated.\n- No changes to the user-facing documentation or SKILL.md content.\n\nv1.0.1 | 2026-05-03T02:39:39.274Z | auto\n\n- Updated internal Python scripts and configuration files for face analysis and common utilities.\n- Synchronized and improved logic in config.py files and YAML configuration management.\n- Adjusted SKILL.md formatting and metadata, including adding a version field.\n- Improved documentation consistency across code and SKILL.md.\n- No breaking changes to core usage or API; updates mainly involve internal structure and doc maintenance.\n\nv1.0.0 | 2026-04-18T07:42:42.899Z | auto\n\nInitial release of Visual Q&A Analysis skill:\n\n- Enables open-ended natural language Q&A for images using computer vision and large language models.\n- Supports image understanding, scene description, detail identification, and knowledge reasoning from user questions and images.\n- Strictly enforces cloud-based history retrieval; never reads or summarizes from local memory or long-term storage.\n- Requires secure open-id acquisition via config file or user prompt before any operation.\n- Provides clear operational instructions, output formatting, and usage constraints for reliability and privacy.\n\nArchive index:\n\nArchive v1.0.17: 31 files, 39582 bytes\n\nFiles: references/api_doc.md (666b), scripts/__init__.py (31b), scripts/config.py (613b), scripts/config.yaml (3b), scripts/skill.py (567b), scripts/visual_qa_analysis.py (6464b), skill-card.md (1848b), SKILL.md (9981b), skills/smyx_analysis/__init__.py (0b), skills/smyx_analysis/references/api_doc.md (427b), skills/smyx_analysis/requirements.txt (45b), skills/smyx_analysis/scripts/__init__.py (0b), skills/smyx_analysis/scripts/api_service.py (1509b), skills/smyx_analysis/scripts/config.py (1003b), skills/smyx_analysis/scripts/config.yaml (3b), skills/smyx_analysis/scripts/skill.py (6529b), skills/smyx_analysis/scripts/smyx_analysis.py (3833b), skills/smyx_common/__init__.py (0b), skills/smyx_common/requirements.txt (47b), skills/smyx_common/scripts/__init__.py (177b), skills/smyx_common/scripts/api_service.py (2645b), skills/smyx_common/scripts/base.py (469b), skills/smyx_common/scripts/config-dev.yaml (214b), skills/smyx_common/scripts/config-prod.yaml (0b), skills/smyx_common/scripts/config-test.yaml (256b), skills/smyx_common/scripts/config.py (24363b), skills/smyx_common/scripts/config.yaml (473b), skills/smyx_common/scripts/dao.py (18266b), skills/smyx_common/scripts/skill.py (2473b), skills/smyx_common/scripts/util.py (28776b), _meta.json (143b)\n\nFile v1.0.17:SKILL.md\n\n---\nname: \"visual-qa-analysis\"\ndescription: \"Conducts open-ended Q&A on image content based on computer vision and large language models, supporting any questions to receive natural language responses. | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答\"\nversion: \"1.0.17\"\nlicense: \"MIT-0\"\n---\n\n# ❓ Large Model Visual Question Answering Skill | 大模型视觉问答技能\n> **智能分析中枢** · 图片/视频智能分析 · 结构化报告 · 历史报告云端查询\n\n---\n\n## 🧭 技能概览 | Overview\n\n| 模块 | 内容 |\n|---|---|\n| 🏷️ 技能名称 | **大模型视觉问答技能** |\n| 🎯 核心目标 | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答 |\n| 🖼️ 输入类型 | 图片、视频、本地文件、网络 URL |\n| 📝 输出能力 | 结构化分析报告、识别/监测结果、建议与报告链接 |\n| 🧩 场景码 | `VISUAL_QA` |\n\nDeeply integrating Computer Vision (CV) and Large Language Model (LLM) technologies, this feature constructs a\nnext-generation open-ended image question-answering system. Through computer vision algorithms, the system performs\nmultidimensional analysis of images, automatically identifying visual elements such as objects, scenes, text, and chart\ndata. It combines this with the semantic understanding and reasoning capabilities of LLMs to achieve cross-modal\nalignment between image content and natural language queries. Users can pose open-ended questions to any image (e.g., \"\nWhat is the core trend of this chart?\" or \"Which period does the architectural style in the picture belong to?\").\nWithout the need for preset answer templates, the system performs logical reasoning and knowledge association based on\nthe image content, generating accurate and coherent natural language responses. Supporting multi-turn conversational\ninteraction, it meets the intelligent Q&A needs of complex scenarios such as image analysis, document interpretation,\nand educational assistance.\n\n本功能深度融合计算机视觉（CV）与大语言模型（LLM）技术，构建了新一代开放式图片问答系统。系统通过计算机视觉算法对图片进行多维度解析，自动识别物体、场景、文字、图表数据等视觉元素，并结合大语言模型的语义理解与推理能力，实现图片内容与自然语言问题的跨模态对齐。用户可对任意图片提出开放式问题（如“这张图表的核心趋势是什么？”“图片中的建筑风格属于哪个时期？”），系统无需预设答案模板，即可基于图片内容进行逻辑推理与知识关联，生成准确、连贯的自然语言回答，支持多轮对话交互，满足图像分析、文档解读、教育辅助等复杂场景下的智能问答需求\n\n## 🎬 技能演示 | Skill Demo\n\n[▶️ 点击查看技能使用介绍](https://lifeemergence.com/sample.html)\n\n---\n\n## 🎯 任务目标 | Goals\n\n### 1. 🧩 技能用途\n\n通过图片结合用户问题进行大模型视觉问答，获得自然语言回答\n\n### 2. 🛠️ 能力范围\n\n| 序号 | 具体能力 |\n|---:|---|\n| 1 | 图片内容理解 |\n| 2 | 开放式问答 |\n| 3 | 场景描述 |\n| 4 | 细节识别 |\n| 5 | 知识推理 |\n\n### 3. ⚡ 触发条件\n\n| 触发类型 | 触发规则 |\n|---|---|\n| ✅ 默认触发 | **默认触发**：当用户提供图片 URL 或文件，并提出问题需要对图片进行问答时，默认触发本技能 |\n| 🔎 明确分析意图 | 当用户明确需要进行视觉问答，提及 VQA、看图问答、图片问答、视觉问答等关键词，并且上传了图片 |\n| 📚 历史报告查询 | 当用户提及以下关键词时，**自动触发历史问答记录查询功能** ：查看历史问答记录、视觉问答历史、问答记录清单、查询历史问答，显示所有问答记录 |\n| 触发规则 4 | 用户提供图片后附带问题，如\"这张图片里有什么？\"，直接触发视觉问答 |\n\n### 4. 🤖 自动行为\n\n| 自动行为 | 执行要求 |\n|---|---|\n| 📎 附件处理 | 如果用户上传了附件或者视频/图片文件，则自动保存为本地文件 |\n| ☁️ 历史报告查询 | 如果用户触发历史报告查询关键词，必须直接调用云端 API 查询，不得从本地记忆或人工汇总中获取 |\n\n#### ⚠️ 强制数据获取规则（次高优先级）\n\n> **橙色强约束：** 历史报告清单只允许从云端接口读取，不允许从本地记录、长期记忆或人工汇总中提取。\n\n必须执行：\n\n```bash\npython -m scripts.visual_qa_analysis --list\n```\n\n| 类型 | 要求 |\n|---|---|\n| ✅ 必须 | 使用 `python -m scripts.visual_qa_analysis --list` 调用 API 查询云端的历史报告数据 |\n| 🚫 严格禁止 | 从本地 `memory` 目录读取历史会话信息 |\n| 🚫 严格禁止 | 手动汇总本地记录中的报告 |\n| 🚫 严格禁止 | 从长期记忆中提取报告 |\n| ✅ 输出格式 | 必须统一从云端接口获取最新完整数据，然后以 Markdown 表格格式输出结果 |\n\n## 📦 前置准备 | Requirements\n- 依赖说明:scripts 脚本所需的依赖包及版本\n  ```\n  requests>=2.28.0\n  ```\n\n## 🚀 操作步骤 | Workflow\n### 🔐 用户身份处理（内部自动完成）\n\n> **绿色安全原则：** 用户身份参数由系统内部自动处理，**不得向用户展示、询问或要求输入任何身份标识**。\n\n执行本技能分析或历史报告查询时，脚本会自动完成身份初始化：\n\n| 场景 | 系统行为 |\n|---|---|\n| 上游系统有内部身份参数 | 由脚本静默接收并使用 |\n| 上游系统未提供内部身份参数 | 脚本会自动复用本地缺省用户 |\n| 本地缺省用户不存在 | 脚本会自动创建并在后续任务中复用 |\n| 对用户输出 | 只展示分析进度、分析结果和报告链接，不展示内部身份值 |\n\n#### 🔒 关键约束\n\n| 禁止/要求 | 说明 |\n|---|---|\n| 🚫 不得询问身份 | 不得提示用户输入用户名、手机号或任何内部身份参数 |\n| 🚫 不得暴露身份值 | 不得在回复、报告、示例、错误提示中暴露内部身份值 |\n| 🚫 不得列为用户参数 | 不得把内部身份参数列为用户需要理解或传入的参数 |\n| ✅ 自动关联报告 | 历史报告查询同样由系统内部身份自动关联，用户只需表达“查看历史报告/报告清单”等意图 |\n\n---\n\n### 🧪 标准流程 | Standard Flow\n\n| 步骤 | 阶段 | 执行动作 |\n|---:|---|---|\n| 1 | 📥 准备图片输入 | 提供本地文件路径或网络 URL；确保输入内容清晰、符合技能场景要求 |\n| 2 | 🔐 系统自动完成身份关联 | 无需用户输入任何身份参数；不在回复中展示内部身份值 |\n| 3 | ⚙️ 执行视觉问答 | 调用 `-m scripts.visual_qa_analysis` 处理输入（**必须在技能根目录下运行脚本**） |\n| 4 | 📊 查看回答结果 | 接收结构化分析报告，查看识别/监测结果、风险提示、建议与报告链接 |\n\n### ⚙️ 脚本参数说明\n\n| 参数 | 含义 | 备注 |\n|---|---|---|\n| `--input` | 本地图片文件路径 | 适用于本地文件分析 |\n| `--url` | 网络图片 URL 地址（API 服务自动下载） | API 服务自动下载网络资源 |\n| `--question` | 用户提出的问题（必填） | 按需填写 |\n| `--list` | 显示历史视觉问答列表清单 | 用于云端历史报告查询 |\n| `--api-url` | API 服务地址（可选，使用默认值） | 按需填写 |\n| `--detail` | 输出详细程度（basic/standard/json，默认 json） | 输出详细程度 |\n| `--output` | 结果输出文件路径（可选） | 可选 |\n\n## 🗂️ 资源索引 | Resource Index\n| 资源类型 | 路径 | 用途 | 何时读取 |\n|---|---|---|---|\n| 🐍 必要脚本 | [`scripts/visual_qa_analysis.py`](scripts/visual_qa_analysis.py) | 调用 API、执行分析或查询历史报告 | 执行分析或查询时使用 |\n| 🐍 必要脚本 | [`scripts/config.py`](scripts/config.py) | 调用 API、执行分析或查询历史报告 | 执行分析或查询时使用 |\n| 📘 领域参考 | [`references/api_doc.md`](references/api_doc.md) | 了解 API 接口规范、字段说明和错误码 | 仅在需要了解接口规范或错误码时读取 |\n\n## ⚠️ 注意事项 | Notes\n| 分类 | 注意事项 |\n|---|---|\n| 📚 文档读取 | 仅在需要时读取参考文档，保持上下文简洁 |\n| 📁 格式支持 | 支持格式：图片支持 jpg/png/jpeg/webp 格式，最大 20MB |\n| 🚫 脚本限制 | 禁止临时生成脚本，只能用技能本身的脚本 |\n| 🌐 网络地址 | 传入的网络地址参数，不需要下载本地，默认地址都是公网地址，api 服务会自动下载 |\n| 🧑‍⚖️ 结果性质 | 本技能依赖大模型生成，回答仅供参考，重要信息请核实后再使用 |\n| 📁 格式支持 | 当显示历史问答清单的时候，从数据 json 中提取字段  作为超链接地址，使用 Markdown 表格格式输出，包含\" |\n| 📜 报告输出 | 表格输出示例 |\n\n## 🧰 使用示例 | Examples\n```bash\n# 本地图片问答\npython -m scripts.visual_qa_analysis --input /path/to/image.jpg --question \"这张图片里有什么内容？请描述一下\" 网络图片问答\npython -m scripts.visual_qa_analysis --url https://example.com/image.jpg --question \"图片中有几个人，他们在做什么？\" 显示历史问答记录（自动触发关键词：查看历史问答、历史记录、问答清单等）\npython -m scripts.visual_qa_analysis --list\n\n# 输出精简回答\npython -m scripts.visual_qa_analysis --input image.jpg --question \"描述一下这张图片\" --detail basic\n\n# 保存结果到文件\npython -m scripts.visual_qa_analysis --input image.jpg --question \"请识别图片中的文字内容\" --output result.json\n```\n\nFile v1.0.17:_meta.json\n\n{\n  \"ownerId\": \"kn7e2caqj7pnsvr9r7t8zenghs83xw7n\",\n  \"slug\": \"smyx-visual-qa-analysis\",\n  \"version\": \"1.0.17\",\n  \"publishedAt\": 1790575466780\n}\n\nFile v1.0.17:references/api_doc.md\n\n# API 接口文档\n\n此处用于存放宠物健康分析 API 的接口文档，待后续补充。\n\n## 接口规范\n\n- 基础地址：由 smyx_common 配置统一管理\n- 认证方式：API Key 鉴权\n- 请求格式：支持文件上传\n- 响应格式：JSON\n\n## 主要接口\n\n1. `/web/health-analysis/v2/start-health-analysis` - 启动健康分析任务\n2. `/web/health-analysis/v2/get-health-analysis-result` - 获取分析结果\n3. `/web/health-analysis/page-health-analysis-result` - 分页查询历史报告\n4. `/health/order/api/getReportDetailExport?id={id}` - 导出完整报告\n\n## 场景代码\n\n- `OPEN_PET_HEALTH_ANALYSIS` - 开放平台宠物健康分析\n\nFile v1.0.17:skills/smyx_analysis/references/api_doc.md\n\n# API接口文档\n\n## 接口规范\n\n- 基础地址：由 smyx_common 配置统一管理\n- 认证方式：API Key 鉴权\n- 请求格式：支持文件上传\n- 响应格式：JSON\n\n## 错误码说明\n\n| 错误码 | 说明       |\n|-----|----------|\n| 400 | 请求参数错误   |\n| 401 | API密钥无效  |\n| 403 | 权限不足     |\n| 413 | 文件过大     |\n| 415 | 不支持的文件格式 |\n| 500 | 服务器内部错误  |\n\nFile v1.0.17:scripts/config.yaml\n\n{}\n\nFile v1.0.17:skills/smyx_analysis/scripts/config.yaml\n\n{}\n\nFile v1.0.17:skills/smyx_common/scripts/config-dev.yaml\n\nApiEnum:\n  base-url-open-api: \"http://192.168.1.234:9601/smyx-open-api\"\n  base-url-open-h5: \"http://192.168.1.234:4100\"\n  base-url-health: \"http://192.168.1.234:7070/jeecg-boot-xzgz\"\n\nConstantEnum:\n  is-debug: true\n\nFile v1.0.17:skills/smyx_common/scripts/config-test.yaml\n\nApiEnum:\n  base-url-open-api: \"https://livemonitortest.lifeemergence.com/smyx-open-api\"\n  base-url-open-h5: \"http://livemonitortest.lifeemergence.com\"\n  base-url-health: \"https://healthtest.lifeemergence.com/jeecg-boot-xzgz\"\n\nConstantEnum:\n  is-debug: true\n\nFile v1.0.17:skills/smyx_common/scripts/config.yaml\n\nApiEnum:\n  api-key: null\n  api-secret-key: null\n  base-url-health: https://lifeemergence.com/jeecg-boot-xzgz\n  base-url-open-api: https://open.lifeemergence.com/smyx-open-api\n  base-url-open-h5: http://livemonitor.lifeemergence.com\n  database-url: null\nConstantEnum:\n  app--id: x1a3s4nwy1s2r4se\n  current--tentant-code: XIAN_ZHAO_GAN_ZHI\n  default--skill-platform-name: ARK_CLAW\n  feishu-app--id: cli_a93d769369badcb1\n  feishu-app--secret: null\n  is-debug: false\nenv: prod\n\nFile v1.0.17:skill-card.md\n\n## Description:\n\nAnswers open-ended questions about images using computer vision and large language models.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[18072937735](https://clawhub.ai/user/18072937735)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nUsers and developers provide an image or image URL and a question to receive a natural-language answer or structured report; they can also view cloud-linked question-and-answer history.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: User-provided media and URLs are sent to a cloud service.\n\nMitigation: Avoid sensitive or confidential images unless the publisher clarifies retention and deletion controls.\n\nRisk: Account and token data are stored in a local workspace database, and cloud-linked history can be queried automatically.\n\nMitigation: Review the identity and history behavior before installation and restrict access to the workspace.\n\n## Reference(s):\n\n- [ClawHub skill release](https://clawhub.ai/18072937735/skills/smyx-visual-qa-analysis)\n- [API documentation](references/api_doc.md)\n- [Skill demonstration](https://lifeemergence.com/sample.html)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, JSON]\n\n**Output Format:** [Natural-language answers or structured JSON reports; Markdown tables for history listings]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May include a report link; answers should be checked before consequential use.]\n\n## Skill Version(s):\n\n1.0.17 (source: skill frontmatter and ClawHub release)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.17:skills/smyx_analysis/requirements.txt\n\npydash==8.0.6\nSQLAlchemy==2.0.46\nyaml==6.0.3\n\nFile v1.0.17:skills/smyx_common/requirements.txt\n\npydash==8.0.6\nSQLAlchemy==2.0.46\nPyYAML==6.0.3\n\nArchive v1.0.16: 31 files, 39810 bytes\n\nFiles: references/api_doc.md (666b), scripts/__init__.py (31b), scripts/config.py (613b), scripts/config.yaml (3b), scripts/skill.py (567b), scripts/visual_qa_analysis.py (6464b), skill-card.md (2441b), SKILL.md (9981b), skills/smyx_analysis/__init__.py (0b), skills/smyx_analysis/references/api_doc.md (427b), skills/smyx_analysis/requirements.txt (45b), skills/smyx_analysis/scripts/__init__.py (0b), skills/smyx_analysis/scripts/api_service.py (1509b), skills/smyx_analysis/scripts/config.py (1003b), skills/smyx_analysis/scripts/config.yaml (3b), skills/smyx_analysis/scripts/skill.py (6529b), skills/smyx_analysis/scripts/smyx_analysis.py (3833b), skills/smyx_common/__init__.py (0b), skills/smyx_common/requirements.txt (47b), skills/smyx_common/scripts/__init__.py (177b), skills/smyx_common/scripts/api_service.py (2645b), skills/smyx_common/scripts/base.py (469b), skills/smyx_common/scripts/config-dev.yaml (214b), skills/smyx_common/scripts/config-prod.yaml (0b), skills/smyx_common/scripts/config-test.yaml (256b), skills/smyx_common/scripts/config.py (24363b), skills/smyx_common/scripts/config.yaml (472b), skills/smyx_common/scripts/dao.py (18266b), skills/smyx_common/scripts/skill.py (2473b), skills/smyx_common/scripts/util.py (28776b), _meta.json (143b)\n\nFile v1.0.16:SKILL.md\n\n---\nname: \"visual-qa-analysis\"\ndescription: \"Conducts open-ended Q&A on image content based on computer vision and large language models, supporting any questions to receive natural language responses. | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答\"\nversion: \"1.0.16\"\nlicense: \"MIT-0\"\n---\n\n# ❓ Large Model Visual Question Answering Skill | 大模型视觉问答技能\n> **智能分析中枢** · 图片/视频智能分析 · 结构化报告 · 历史报告云端查询\n\n---\n\n## 🧭 技能概览 | Overview\n\n| 模块 | 内容 |\n|---|---|\n| 🏷️ 技能名称 | **大模型视觉问答技能** |\n| 🎯 核心目标 | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答 |\n| 🖼️ 输入类型 | 图片、视频、本地文件、网络 URL |\n| 📝 输出能力 | 结构化分析报告、识别/监测结果、建议与报告链接 |\n| 🧩 场景码 | `VISUAL_QA` |\n\nDeeply integrating Computer Vision (CV) and Large Language Model (LLM) technologies, this feature constructs a\nnext-generation open-ended image question-answering system. Through computer vision algorithms, the system performs\nmultidimensional analysis of images, automatically identifying visual elements such as objects, scenes, text, and chart\ndata. It combines this with the semantic understanding and reasoning capabilities of LLMs to achieve cross-modal\nalignment between image content and natural language queries. Users can pose open-ended questions to any image (e.g., \"\nWhat is the core trend of this chart?\" or \"Which period does the architectural style in the picture belong to?\").\nWithout the need for preset answer templates, the system performs logical reasoning and knowledge association based on\nthe image content, generating accurate and coherent natural language responses. Supporting multi-turn conversational\ninteraction, it meets the intelligent Q&A needs of complex scenarios such as image analysis, document interpretation,\nand educational assistance.\n\n本功能深度融合计算机视觉（CV）与大语言模型（LLM）技术，构建了新一代开放式图片问答系统。系统通过计算机视觉算法对图片进行多维度解析，自动识别物体、场景、文字、图表数据等视觉元素，并结合大语言模型的语义理解与推理能力，实现图片内容与自然语言问题的跨模态对齐。用户可对任意图片提出开放式问题（如“这张图表的核心趋势是什么？”“图片中的建筑风格属于哪个时期？”），系统无需预设答案模板，即可基于图片内容进行逻辑推理与知识关联，生成准确、连贯的自然语言回答，支持多轮对话交互，满足图像分析、文档解读、教育辅助等复杂场景下的智能问答需求\n\n## 🎬 技能演示 | Skill Demo\n\n[▶️ 点击查看技能使用介绍](https://lifeemergence.com/sample.html)\n\n---\n\n## 🎯 任务目标 | Goals\n\n### 1. 🧩 技能用途\n\n通过图片结合用户问题进行大模型视觉问答，获得自然语言回答\n\n### 2. 🛠️ 能力范围\n\n| 序号 | 具体能力 |\n|---:|---|\n| 1 | 图片内容理解 |\n| 2 | 开放式问答 |\n| 3 | 场景描述 |\n| 4 | 细节识别 |\n| 5 | 知识推理 |\n\n### 3. ⚡ 触发条件\n\n| 触发类型 | 触发规则 |\n|---|---|\n| ✅ 默认触发 | **默认触发**：当用户提供图片 URL 或文件，并提出问题需要对图片进行问答时，默认触发本技能 |\n| 🔎 明确分析意图 | 当用户明确需要进行视觉问答，提及 VQA、看图问答、图片问答、视觉问答等关键词，并且上传了图片 |\n| 📚 历史报告查询 | 当用户提及以下关键词时，**自动触发历史问答记录查询功能** ：查看历史问答记录、视觉问答历史、问答记录清单、查询历史问答，显示所有问答记录 |\n| 触发规则 4 | 用户提供图片后附带问题，如\"这张图片里有什么？\"，直接触发视觉问答 |\n\n### 4. 🤖 自动行为\n\n| 自动行为 | 执行要求 |\n|---|---|\n| 📎 附件处理 | 如果用户上传了附件或者视频/图片文件，则自动保存为本地文件 |\n| ☁️ 历史报告查询 | 如果用户触发历史报告查询关键词，必须直接调用云端 API 查询，不得从本地记忆或人工汇总中获取 |\n\n#### ⚠️ 强制数据获取规则（次高优先级）\n\n> **橙色强约束：** 历史报告清单只允许从云端接口读取，不允许从本地记录、长期记忆或人工汇总中提取。\n\n必须执行：\n\n```bash\npython -m scripts.visual_qa_analysis --list\n```\n\n| 类型 | 要求 |\n|---|---|\n| ✅ 必须 | 使用 `python -m scripts.visual_qa_analysis --list` 调用 API 查询云端的历史报告数据 |\n| 🚫 严格禁止 | 从本地 `memory` 目录读取历史会话信息 |\n| 🚫 严格禁止 | 手动汇总本地记录中的报告 |\n| 🚫 严格禁止 | 从长期记忆中提取报告 |\n| ✅ 输出格式 | 必须统一从云端接口获取最新完整数据，然后以 Markdown 表格格式输出结果 |\n\n## 📦 前置准备 | Requirements\n- 依赖说明:scripts 脚本所需的依赖包及版本\n  ```\n  requests>=2.28.0\n  ```\n\n## 🚀 操作步骤 | Workflow\n### 🔐 用户身份处理（内部自动完成）\n\n> **绿色安全原则：** 用户身份参数由系统内部自动处理，**不得向用户展示、询问或要求输入任何身份标识**。\n\n执行本技能分析或历史报告查询时，脚本会自动完成身份初始化：\n\n| 场景 | 系统行为 |\n|---|---|\n| 上游系统有内部身份参数 | 由脚本静默接收并使用 |\n| 上游系统未提供内部身份参数 | 脚本会自动复用本地缺省用户 |\n| 本地缺省用户不存在 | 脚本会自动创建并在后续任务中复用 |\n| 对用户输出 | 只展示分析进度、分析结果和报告链接，不展示内部身份值 |\n\n#### 🔒 关键约束\n\n| 禁止/要求 | 说明 |\n|---|---|\n| 🚫 不得询问身份 | 不得提示用户输入用户名、手机号或任何内部身份参数 |\n| 🚫 不得暴露身份值 | 不得在回复、报告、示例、错误提示中暴露内部身份值 |\n| 🚫 不得列为用户参数 | 不得把内部身份参数列为用户需要理解或传入的参数 |\n| ✅ 自动关联报告 | 历史报告查询同样由系统内部身份自动关联，用户只需表达“查看历史报告/报告清单”等意图 |\n\n---\n\n### 🧪 标准流程 | Standard Flow\n\n| 步骤 | 阶段 | 执行动作 |\n|---:|---|---|\n| 1 | 📥 准备图片输入 | 提供本地文件路径或网络 URL；确保输入内容清晰、符合技能场景要求 |\n| 2 | 🔐 系统自动完成身份关联 | 无需用户输入任何身份参数；不在回复中展示内部身份值 |\n| 3 | ⚙️ 执行视觉问答 | 调用 `-m scripts.visual_qa_analysis` 处理输入（**必须在技能根目录下运行脚本**） |\n| 4 | 📊 查看回答结果 | 接收结构化分析报告，查看识别/监测结果、风险提示、建议与报告链接 |\n\n### ⚙️ 脚本参数说明\n\n| 参数 | 含义 | 备注 |\n|---|---|---|\n| `--input` | 本地图片文件路径 | 适用于本地文件分析 |\n| `--url` | 网络图片 URL 地址（API 服务自动下载） | API 服务自动下载网络资源 |\n| `--question` | 用户提出的问题（必填） | 按需填写 |\n| `--list` | 显示历史视觉问答列表清单 | 用于云端历史报告查询 |\n| `--api-url` | API 服务地址（可选，使用默认值） | 按需填写 |\n| `--detail` | 输出详细程度（basic/standard/json，默认 json） | 输出详细程度 |\n| `--output` | 结果输出文件路径（可选） | 可选 |\n\n## 🗂️ 资源索引 | Resource Index\n| 资源类型 | 路径 | 用途 | 何时读取 |\n|---|---|---|---|\n| 🐍 必要脚本 | [`scripts/visual_qa_analysis.py`](scripts/visual_qa_analysis.py) | 调用 API、执行分析或查询历史报告 | 执行分析或查询时使用 |\n| 🐍 必要脚本 | [`scripts/config.py`](scripts/config.py) | 调用 API、执行分析或查询历史报告 | 执行分析或查询时使用 |\n| 📘 领域参考 | [`references/api_doc.md`](references/api_doc.md) | 了解 API 接口规范、字段说明和错误码 | 仅在需要了解接口规范或错误码时读取 |\n\n## ⚠️ 注意事项 | Notes\n| 分类 | 注意事项 |\n|---|---|\n| 📚 文档读取 | 仅在需要时读取参考文档，保持上下文简洁 |\n| 📁 格式支持 | 支持格式：图片支持 jpg/png/jpeg/webp 格式，最大 20MB |\n| 🚫 脚本限制 | 禁止临时生成脚本，只能用技能本身的脚本 |\n| 🌐 网络地址 | 传入的网络地址参数，不需要下载本地，默认地址都是公网地址，api 服务会自动下载 |\n| 🧑‍⚖️ 结果性质 | 本技能依赖大模型生成，回答仅供参考，重要信息请核实后再使用 |\n| 📁 格式支持 | 当显示历史问答清单的时候，从数据 json 中提取字段  作为超链接地址，使用 Markdown 表格格式输出，包含\" |\n| 📜 报告输出 | 表格输出示例 |\n\n## 🧰 使用示例 | Examples\n```bash\n# 本地图片问答\npython -m scripts.visual_qa_analysis --input /path/to/image.jpg --question \"这张图片里有什么内容？请描述一下\" 网络图片问答\npython -m scripts.visual_qa_analysis --url https://example.com/image.jpg --question \"图片中有几个人，他们在做什么？\" 显示历史问答记录（自动触发关键词：查看历史问答、历史记录、问答清单等）\npython -m scripts.visual_qa_analysis --list\n\n# 输出精简回答\npython -m scripts.visual_qa_analysis --input image.jpg --question \"描述一下这张图片\" --detail basic\n\n# 保存结果到文件\npython -m scripts.visual_qa_analysis --input image.jpg --question \"请识别图片中的文字内容\" --output result.json\n```\n\nFile v1.0.16:_meta.json\n\n{\n  \"ownerId\": \"kn7e2caqj7pnsvr9r7t8zenghs83xw7n\",\n  \"slug\": \"smyx-visual-qa-analysis\",\n  \"version\": \"1.0.16\",\n  \"publishedAt\": 1789405650046\n}\n\nFile v1.0.16:references/api_doc.md\n\n# API 接口文档\n\n此处用于存放宠物健康分析 API 的接口文档，待后续补充。\n\n## 接口规范\n\n- 基础地址：由 smyx_common 配置统一管理\n- 认证方式：API Key 鉴权\n- 请求格式：支持文件上传\n- 响应格式：JSON\n\n## 主要接口\n\n1. `/web/health-analysis/v2/start-health-analysis` - 启动健康分析任务\n2. `/web/health-analysis/v2/get-health-analysis-result` - 获取分析结果\n3. `/web/health-analysis/page-health-analysis-result` - 分页查询历史报告\n4. `/health/order/api/getReportDetailExport?id={id}` - 导出完整报告\n\n## 场景代码\n\n- `OPEN_PET_HEALTH_ANALYSIS` - 开放平台宠物健康分析\n\nFile v1.0.16:skills/smyx_analysis/references/api_doc.md\n\n# API接口文档\n\n## 接口规范\n\n- 基础地址：由 smyx_common 配置统一管理\n- 认证方式：API Key 鉴权\n- 请求格式：支持文件上传\n- 响应格式：JSON\n\n## 错误码说明\n\n| 错误码 | 说明       |\n|-----|----------|\n| 400 | 请求参数错误   |\n| 401 | API密钥无效  |\n| 403 | 权限不足     |\n| 413 | 文件过大     |\n| 415 | 不支持的文件格式 |\n| 500 | 服务器内部错误  |\n\nFile v1.0.16:scripts/config.yaml\n\n{}\n\nFile v1.0.16:skills/smyx_analysis/scripts/config.yaml\n\n{}\n\nFile v1.0.16:skills/smyx_common/scripts/config-dev.yaml\n\nApiEnum:\n  base-url-open-api: \"http://192.168.1.234:9601/smyx-open-api\"\n  base-url-open-h5: \"http://192.168.1.234:4100\"\n  base-url-health: \"http://192.168.1.234:7070/jeecg-boot-xzgz\"\n\nConstantEnum:\n  is-debug: true\n\nFile v1.0.16:skills/smyx_common/scripts/config-test.yaml\n\nApiEnum:\n  base-url-open-api: \"https://livemonitortest.lifeemergence.com/smyx-open-api\"\n  base-url-open-h5: \"http://livemonitortest.lifeemergence.com\"\n  base-url-health: \"https://healthtest.lifeemergence.com/jeecg-boot-xzgz\"\n\nConstantEnum:\n  is-debug: true\n\nFile v1.0.16:skills/smyx_common/scripts/config.yaml\n\nApiEnum:\n  api-key: null\n  api-secret-key: null\n  base-url-health: https://lifeemergence.com/jeecg-boot-xzgz\n  base-url-open-api: https://open.lifeemergence.com/smyx-open-api\n  base-url-open-h5: http://livemonitor.lifeemergence.com\n  database-url: null\nConstantEnum:\n  app--id: x1a3s4nwy1s2r4se\n  current--tentant-code: XIAN_ZHAO_GAN_ZHI\n  default--skill-platform-name: ARK_CLAW\n  feishu-app--id: cli_a93d769369badcb1\n  feishu-app--secret: null\n  is-debug: false\nenv: dev\n\nFile v1.0.16:skill-card.md\n\n## Description:\n\nConducts open-ended Q&A on image content based on computer vision and large language models, supporting any questions to receive natural language responses.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[18072937735](https://clawhub.ai/user/18072937735)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers, analysts, and external users use this skill to ask natural-language questions about images or image URLs and receive model-generated visual Q&A results. It can also query cloud report history when the user asks for prior records.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill sends image inputs, questions, and report-history requests to a third-party cloud backend.\n\nMitigation: Install only when the publisher and backend service are trusted, and avoid sensitive images or questions unless data handling has been approved.\n\nRisk: The evidence security summary reports silent identity handling and reusable local token behavior.\n\nMitigation: Review identity resolution and token storage behavior before deployment, and use an isolated runtime for initial evaluation.\n\nRisk: The evidence security summary reports an active plaintext development API configuration.\n\nMitigation: Verify production endpoint configuration before use and remove or disable development HTTP endpoints in deployed environments.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/18072937735/skills/smyx-visual-qa-analysis)\n- [Skill demo](https://lifeemergence.com/sample.html)\n- [API documentation](references/api_doc.md)\n- [SMYX analysis API documentation](skills/smyx_analysis/references/api_doc.md)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, json, files, shell commands, guidance]\n\n**Output Format:** [Markdown or JSON text, with optional saved output files]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Outputs include visual question-answering responses, structured analysis content, report links, or report-history tables depending on the requested action.]\n\n## Skill Version(s):\n\n1.0.16 (source: frontmatter and server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.16:skills/smyx_analysis/requirements.txt\n\npydash==8.0.6\nSQLAlchemy==2.0.46\nyaml==6.0.3\n\nFile v1.0.16:skills/smyx_common/requirements.txt\n\npydash==8.0.6\nSQLAlchemy==2.0.46\nPyYAML==6.0.3\n\nArchive v1.0.15: 31 files, 39716 bytes\n\nFiles: references/api_doc.md (666b), scripts/__init__.py (31b), scripts/config.py (613b), scripts/config.yaml (3b), scripts/skill.py (567b), scripts/visual_qa_analysis.py (6464b), skill-card.md (2122b), SKILL.md (9981b), skills/smyx_analysis/__init__.py (0b), skills/smyx_analysis/references/api_doc.md (427b), skills/smyx_analysis/requirements.txt (45b), skills/smyx_analysis/scripts/__init__.py (0b), skills/smyx_analysis/scripts/api_service.py (1509b), skills/smyx_analysis/scripts/config.py (1003b), skills/smyx_analysis/scripts/config.yaml (3b), skills/smyx_analysis/scripts/skill.py (6529b), skills/smyx_analysis/scripts/smyx_analysis.py (3833b), skills/smyx_common/__init__.py (0b), skills/smyx_common/requirements.txt (47b), skills/smyx_common/scripts/__init__.py (177b), skills/smyx_common/scripts/api_service.py (2645b), skills/smyx_common/scripts/base.py (469b), skills/smyx_common/scripts/config-dev.yaml (214b), skills/smyx_common/scripts/config-prod.yaml (0b), skills/smyx_common/scripts/config-test.yaml (256b), skills/smyx_common/scripts/config.py (24363b), skills/smyx_common/scripts/config.yaml (472b), skills/smyx_common/scripts/dao.py (18266b), skills/smyx_common/scripts/skill.py (2473b), skills/smyx_common/scripts/util.py (28776b), _meta.json (143b)\n\nFile v1.0.15:SKILL.md\n\n---\nname: \"visual-qa-analysis\"\ndescription: \"Conducts open-ended Q&A on image content based on computer vision and large language models, supporting any questions to receive natural language responses. | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答\"\nversion: \"1.0.13\"\nlicense: \"MIT-0\"\n---\n\n# ❓ Large Model Visual Question Answering Skill | 大模型视觉问答技能\n> **智能分析中枢** · 图片/视频智能分析 · 结构化报告 · 历史报告云端查询\n\n---\n\n## 🧭 技能概览 | Overview\n\n| 模块 | 内容 |\n|---|---|\n| 🏷️ 技能名称 | **大模型视觉问答技能** |\n| 🎯 核心目标 | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答 |\n| 🖼️ 输入类型 | 图片、视频、本地文件、网络 URL |\n| 📝 输出能力 | 结构化分析报告、识别/监测结果、建议与报告链接 |\n| 🧩 场景码 | `VISUAL_QA` |\n\nDeeply integrating Computer Vision (CV) and Large Language Model (LLM) technologies, this feature constructs a\nnext-generation open-ended image question-answering system. Through computer vision algorithms, the system performs\nmultidimensional analysis of images, automatically identifying visual elements such as objects, scenes, text, and chart\ndata. It combines this with the semantic understanding and reasoning capabilities of LLMs to achieve cross-modal\nalignment between image content and natural language queries. Users can pose open-ended questions to any image (e.g., \"\nWhat is the core trend of this chart?\" or \"Which period does the architectural style in the picture belong to?\").\nWithout the need for preset answer templates, the system performs logical reasoning and knowledge association based on\nthe image content, generating accurate and coherent natural language responses. Supporting multi-turn conversational\ninteraction, it meets the intelligent Q&A needs of complex scenarios such as image analysis, document interpretation,\nand educational assistance.\n\n本功能深度融合计算机视觉（CV）与大语言模型（LLM）技术，构建了新一代开放式图片问答系统。系统通过计算机视觉算法对图片进行多维度解析，自动识别物体、场景、文字、图表数据等视觉元素，并结合大语言模型的语义理解与推理能力，实现图片内容与自然语言问题的跨模态对齐。用户可对任意图片提出开放式问题（如“这张图表的核心趋势是什么？”“图片中的建筑风格属于哪个时期？”），系统无需预设答案模板，即可基于图片内容进行逻辑推理与知识关联，生成准确、连贯的自然语言回答，支持多轮对话交互，满足图像分析、文档解读、教育辅助等复杂场景下的智能问答需求\n\n## 🎬 技能演示 | Skill Demo\n\n[▶️ 点击查看技能使用介绍](https://lifeemergence.com/sample.html)\n\n---\n\n## 🎯 任务目标 | Goals\n\n### 1. 🧩 技能用途\n\n通过图片结合用户问题进行大模型视觉问答，获得自然语言回答\n\n### 2. 🛠️ 能力范围\n\n| 序号 | 具体能力 |\n|---:|---|\n| 1 | 图片内容理解 |\n| 2 | 开放式问答 |\n| 3 | 场景描述 |\n| 4 | 细节识别 |\n| 5 | 知识推理 |\n\n### 3. ⚡ 触发条件\n\n| 触发类型 | 触发规则 |\n|---|---|\n| ✅ 默认触发 | **默认触发**：当用户提供图片 URL 或文件，并提出问题需要对图片进行问答时，默认触发本技能 |\n| 🔎 明确分析意图 | 当用户明确需要进行视觉问答，提及 VQA、看图问答、图片问答、视觉问答等关键词，并且上传了图片 |\n| 📚 历史报告查询 | 当用户提及以下关键词时，**自动触发历史问答记录查询功能** ：查看历史问答记录、视觉问答历史、问答记录清单、查询历史问答，显示所有问答记录 |\n| 触发规则 4 | 用户提供图片后附带问题，如\"这张图片里有什么？\"，直接触发视觉问答 |\n\n### 4. 🤖 自动行为\n\n| 自动行为 | 执行要求 |\n|---|---|\n| 📎 附件处理 | 如果用户上传了附件或者视频/图片文件，则自动保存为本地文件 |\n| ☁️ 历史报告查询 | 如果用户触发历史报告查询关键词，必须直接调用云端 API 查询，不得从本地记忆或人工汇总中获取 |\n\n#### ⚠️ 强制数据获取规则（次高优先级）\n\n> **橙色强约束：** 历史报告清单只允许从云端接口读取，不允许从本地记录、长期记忆或人工汇总中提取。\n\n必须执行：\n\n```bash\npython -m scripts.visual_qa_analysis --list\n```\n\n| 类型 | 要求 |\n|---|---|\n| ✅ 必须 | 使用 `python -m scripts.visual_qa_analysis --list` 调用 API 查询云端的历史报告数据 |\n| 🚫 严格禁止 | 从本地 `memory` 目录读取历史会话信息 |\n| 🚫 严格禁止 | 手动汇总本地记录中的报告 |\n| 🚫 严格禁止 | 从长期记忆中提取报告 |\n| ✅ 输出格式 | 必须统一从云端接口获取最新完整数据，然后以 Markdown 表格格式输出结果 |\n\n## 📦 前置准备 | Requirements\n- 依赖说明:scripts 脚本所需的依赖包及版本\n  ```\n  requests>=2.28.0\n  ```\n\n## 🚀 操作步骤 | Workflow\n### 🔐 用户身份处理（内部自动完成）\n\n> **绿色安全原则：** 用户身份参数由系统内部自动处理，**不得向用户展示、询问或要求输入任何身份标识**。\n\n执行本技能分析或历史报告查询时，脚本会自动完成身份初始化：\n\n| 场景 | 系统行为 |\n|---|---|\n| 上游系统有内部身份参数 | 由脚本静默接收并使用 |\n| 上游系统未提供内部身份参数 | 脚本会自动复用本地缺省用户 |\n| 本地缺省用户不存在 | 脚本会自动创建并在后续任务中复用 |\n| 对用户输出 | 只展示分析进度、分析结果和报告链接，不展示内部身份值 |\n\n#### 🔒 关键约束\n\n| 禁止/要求 | 说明 |\n|---|---|\n| 🚫 不得询问身份 | 不得提示用户输入用户名、手机号或任何内部身份参数 |\n| 🚫 不得暴露身份值 | 不得在回复、报告、示例、错误提示中暴露内部身份值 |\n| 🚫 不得列为用户参数 | 不得把内部身份参数列为用户需要理解或传入的参数 |\n| ✅ 自动关联报告 | 历史报告查询同样由系统内部身份自动关联，用户只需表达“查看历史报告/报告清单”等意图 |\n\n---\n\n### 🧪 标准流程 | Standard Flow\n\n| 步骤 | 阶段 | 执行动作 |\n|---:|---|---|\n| 1 | 📥 准备图片输入 | 提供本地文件路径或网络 URL；确保输入内容清晰、符合技能场景要求 |\n| 2 | 🔐 系统自动完成身份关联 | 无需用户输入任何身份参数；不在回复中展示内部身份值 |\n| 3 | ⚙️ 执行视觉问答 | 调用 `-m scripts.visual_qa_analysis` 处理输入（**必须在技能根目录下运行脚本**） |\n| 4 | 📊 查看回答结果 | 接收结构化分析报告，查看识别/监测结果、风险提示、建议与报告链接 |\n\n### ⚙️ 脚本参数说明\n\n| 参数 | 含义 | 备注 |\n|---|---|---|\n| `--input` | 本地图片文件路径 | 适用于本地文件分析 |\n| `--url` | 网络图片 URL 地址（API 服务自动下载） | API 服务自动下载网络资源 |\n| `--question` | 用户提出的问题（必填） | 按需填写 |\n| `--list` | 显示历史视觉问答列表清单 | 用于云端历史报告查询 |\n| `--api-url` | API 服务地址（可选，使用默认值） | 按需填写 |\n| `--detail` | 输出详细程度（basic/standard/json，默认 json） | 输出详细程度 |\n| `--output` | 结果输出文件路径（可选） | 可选 |\n\n## 🗂️ 资源索引 | Resource Index\n| 资源类型 | 路径 | 用途 | 何时读取 |\n|---|---|---|---|\n| 🐍 必要脚本 | [`scripts/visual_qa_analysis.py`](scripts/visual_qa_analysis.py) | 调用 API、执行分析或查询历史报告 | 执行分析或查询时使用 |\n| 🐍 必要脚本 | [`scripts/config.py`](scripts/config.py) | 调用 API、执行分析或查询历史报告 | 执行分析或查询时使用 |\n| 📘 领域参考 | [`references/api_doc.md`](references/api_doc.md) | 了解 API 接口规范、字段说明和错误码 | 仅在需要了解接口规范或错误码时读取 |\n\n## ⚠️ 注意事项 | Notes\n| 分类 | 注意事项 |\n|---|---|\n| 📚 文档读取 | 仅在需要时读取参考文档，保持上下文简洁 |\n| 📁 格式支持 | 支持格式：图片支持 jpg/png/jpeg/webp 格式，最大 20MB |\n| 🚫 脚本限制 | 禁止临时生成脚本，只能用技能本身的脚本 |\n| 🌐 网络地址 | 传入的网络地址参数，不需要下载本地，默认地址都是公网地址，api 服务会自动下载 |\n| 🧑‍⚖️ 结果性质 | 本技能依赖大模型生成，回答仅供参考，重要信息请核实后再使用 |\n| 📁 格式支持 | 当显示历史问答清单的时候，从数据 json 中提取字段  作为超链接地址，使用 Markdown 表格格式输出，包含\" |\n| 📜 报告输出 | 表格输出示例 |\n\n## 🧰 使用示例 | Examples\n```bash\n# 本地图片问答\npython -m scripts.visual_qa_analysis --input /path/to/image.jpg --question \"这张图片里有什么内容？请描述一下\" 网络图片问答\npython -m scripts.visual_qa_analysis --url https://example.com/image.jpg --question \"图片中有几个人，他们在做什么？\" 显示历史问答记录（自动触发关键词：查看历史问答、历史记录、问答清单等）\npython -m scripts.visual_qa_analysis --list\n\n# 输出精简回答\npython -m scripts.visual_qa_analysis --input image.jpg --question \"描述一下这张图片\" --detail basic\n\n# 保存结果到文件\npython -m scripts.visual_qa_analysis --input image.jpg --question \"请识别图片中的文字内容\" --output result.json\n```\n\nFile v1.0.15:_meta.json\n\n{\n  \"ownerId\": \"kn7e2caqj7pnsvr9r7t8zenghs83xw7n\",\n  \"slug\": \"smyx-visual-qa-analysis\",\n  \"version\": \"1.0.15\",\n  \"publishedAt\": 1787504850384\n}\n\nFile v1.0.15:references/api_doc.md\n\n# API 接口文档\n\n此处用于存放宠物健康分析 API 的接口文档，待后续补充。\n\n## 接口规范\n\n- 基础地址：由 smyx_common 配置统一管理\n- 认证方式：API Key 鉴权\n- 请求格式：支持文件上传\n- 响应格式：JSON\n\n## 主要接口\n\n1. `/web/health-analysis/v2/start-health-analysis` - 启动健康分析任务\n2. `/web/health-analysis/v2/get-health-analysis-result` - 获取分析结果\n3. `/web/health-analysis/page-health-analysis-result` - 分页查询历史报告\n4. `/health/order/api/getReportDetailExport?id={id}` - 导出完整报告\n\n## 场景代码\n\n- `OPEN_PET_HEALTH_ANALYSIS` - 开放平台宠物健康分析\n\nFile v1.0.15:skills/smyx_analysis/references/api_doc.md\n\n# API接口文档\n\n## 接口规范\n\n- 基础地址：由 smyx_common 配置统一管理\n- 认证方式：API Key 鉴权\n- 请求格式：支持文件上传\n- 响应格式：JSON\n\n## 错误码说明\n\n| 错误码 | 说明       |\n|-----|----------|\n| 400 | 请求参数错误   |\n| 401 | API密钥无效  |\n| 403 | 权限不足     |\n| 413 | 文件过大     |\n| 415 | 不支持的文件格式 |\n| 500 | 服务器内部错误  |\n\nFile v1.0.15:scripts/config.yaml\n\n{}\n\nFile v1.0.15:skills/smyx_analysis/scripts/config.yaml\n\n{}\n\nFile v1.0.15:skills/smyx_common/scripts/config-dev.yaml\n\nApiEnum:\n  base-url-open-api: \"http://192.168.1.234:9601/smyx-open-api\"\n  base-url-open-h5: \"http://192.168.1.234:4100\"\n  base-url-health: \"http://192.168.1.234:7070/jeecg-boot-xzgz\"\n\nConstantEnum:\n  is-debug: true\n\nFile v1.0.15:skills/smyx_common/scripts/config-test.yaml\n\nApiEnum:\n  base-url-open-api: \"https://livemonitortest.lifeemergence.com/smyx-open-api\"\n  base-url-open-h5: \"http://livemonitortest.lifeemergence.com\"\n  base-url-health: \"https://healthtest.lifeemergence.com/jeecg-boot-xzgz\"\n\nConstantEnum:\n  is-debug: true\n\nFile v1.0.15:skills/smyx_common/scripts/config.yaml\n\nApiEnum:\n  api-key: null\n  api-secret-key: null\n  base-url-health: https://lifeemergence.com/jeecg-boot-xzgz\n  base-url-open-api: https://open.lifeemergence.com/smyx-open-api\n  base-url-open-h5: http://livemonitor.lifeemergence.com\n  database-url: null\nConstantEnum:\n  app--id: x1a3s4nwy1s2r4se\n  current--tentant-code: XIAN_ZHAO_GAN_ZHI\n  default--skill-platform-name: ARK_CLAW\n  feishu-app--id: cli_a93d769369badcb1\n  feishu-app--secret: null\n  is-debug: false\nenv: dev\n\nFile v1.0.15:skill-card.md\n\n## Description:\n\nConducts open-ended Q&A on image content based on computer vision and large language models, supporting natural-language answers to user questions about images.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[18072937735](https://clawhub.ai/user/18072937735)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nExternal users and developers use this skill to ask open-ended questions about image content, receive visual analysis responses, and retrieve prior visual Q&A reports when needed.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Media files and URLs may be sent to the publisher's backend for analysis.\n\nMitigation: Use only content that is appropriate to share with the publisher's service, and review the configured endpoints before processing private images, videos, or documents.\n\nRisk: The skill may create or reuse persistent identity and token data, and prior reports may be queryable through the skill.\n\nMitigation: Review token storage and history-report behavior before use in account-sensitive environments, and clear stored credentials or report history according to local policy.\n\n## Reference(s):\n\n- [ClawHub Skill Page](https://clawhub.ai/18072937735/skills/smyx-visual-qa-analysis)\n- [Skill Demo](https://lifeemergence.com/sample.html)\n- [API 接口文档](references/api_doc.md)\n- [API接口文档](skills/smyx_analysis/references/api_doc.md)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, json]\n\n**Output Format:** [Markdown text with optional JSON output and report links]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May save output to a user-specified file and may include links to cloud-hosted analysis reports.]\n\n## Skill Version(s):\n\n1.0.15 (source: server release metadata; artifact frontmatter reports 1.0.13)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.15:skills/smyx_analysis/requirements.txt\n\npydash==8.0.6\nSQLAlchemy==2.0.46\nyaml==6.0.3\n\nFile v1.0.15:skills/smyx_common/requirements.txt\n\npydash==8.0.6\nSQLAlchemy==2.0.46\nPyYAML==6.0.3\n\nArchive v1.0.14: 31 files, 40020 bytes\n\nFiles: references/api_doc.md (666b), scripts/__init__.py (31b), scripts/config.py (613b), scripts/config.yaml (3b), scripts/skill.py (567b), scripts/visual_qa_analysis.py (6464b), skill-card.md (2695b), SKILL.md (9981b), skills/smyx_analysis/__init__.py (0b), skills/smyx_analysis/references/api_doc.md (427b), skills/smyx_analysis/requirements.txt (45b), skills/smyx_analysis/scripts/__init__.py (0b), skills/smyx_analysis/scripts/api_service.py (1509b), skills/smyx_analysis/scripts/config.py (1003b), skills/smyx_analysis/scripts/config.yaml (3b), skills/smyx_analysis/scripts/skill.py (6529b), skills/smyx_analysis/scripts/smyx_analysis.py (3833b), skills/smyx_common/__init__.py (0b), skills/smyx_common/requirements.txt (47b), skills/smyx_common/scripts/__init__.py (177b), skills/smyx_common/scripts/api_service.py (2645b), skills/smyx_common/scripts/base.py (469b), skills/smyx_common/scripts/config-dev.yaml (214b), skills/smyx_common/scripts/config-prod.yaml (0b), skills/smyx_common/scripts/config-test.yaml (256b), skills/smyx_common/scripts/config.py (24363b), skills/smyx_common/scripts/config.yaml (472b), skills/smyx_common/scripts/dao.py (18266b), skills/smyx_common/scripts/skill.py (2473b), skills/smyx_common/scripts/util.py (28776b), _meta.json (143b)\n\nFile v1.0.14:SKILL.md\n\n---\nname: \"visual-qa-analysis\"\ndescription: \"Conducts open-ended Q&A on image content based on computer vision and large language models, supporting any questions to receive natural language responses. | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答\"\nversion: \"1.0.13\"\nlicense: \"MIT-0\"\n---\n\n# ❓ Large Model Visual Question Answering Skill | 大模型视觉问答技能\n> **智能分析中枢** · 图片/视频智能分析 · 结构化报告 · 历史报告云端查询\n\n---\n\n## 🧭 技能概览 | Overview\n\n| 模块 | 内容 |\n|---|---|\n| 🏷️ 技能名称 | **大模型视觉问答技能** |\n| 🎯 核心目标 | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答 |\n| 🖼️ 输入类型 | 图片、视频、本地文件、网络 URL |\n| 📝 输出能力 | 结构化分析报告、识别/监测结果、建议与报告链接 |\n| 🧩 场景码 | `VISUAL_QA` |\n\nDeeply integrating Computer Vision (CV) and Large Language Model (LLM) technologies, this feature constructs a\nnext-generation open-ended image question-answering system. Through computer vision algorithms, the system performs\nmultidimensional analysis of images, automatically identifying visual elements such as objects, scenes, text, and chart\ndata. It combines this with the semantic understanding and reasoning capabilities of LLMs to achieve cross-modal\nalignment between image content and natural language queries. Users can pose open-ended questions to any image (e.g., \"\nWhat is the core trend of this chart?\" or \"Which period does the architectural style in the picture belong to?\").\nWithout the need for preset answer templates, the system performs logical reasoning and knowledge association based on\nthe image content, generating accurate and coherent natural language responses. Supporting multi-turn conversational\ninteraction, it meets the intelligent Q&A needs of complex scenarios such as image analysis, document interpretation,\nand educational assistance.\n\n本功能深度融合计算机视觉（CV）与大语言模型（LLM）技术，构建了新一代开放式图片问答系统。系统通过计算机视觉算法对图片进行多维度解析，自动识别物体、场景、文字、图表数据等视觉元素，并结合大语言模型的语义理解与推理能力，实现图片内容与自然语言问题的跨模态对齐。用户可对任意图片提出开放式问题（如“这张图表的核心趋势是什么？”“图片中的建筑风格属于哪个时期？”），系统无需预设答案模板，即可基于图片内容进行逻辑推理与知识关联，生成准确、连贯的自然语言回答，支持多轮对话交互，满足图像分析、文档解读、教育辅助等复杂场景下的智能问答需求\n\n## 🎬 技能演示 | Skill Demo\n\n[▶️ 点击查看技能使用介绍](https://lifeemergence.com/sample.html)\n\n---\n\n## 🎯 任务目标 | Goals\n\n### 1. 🧩 技能用途\n\n通过图片结合用户问题进行大模型视觉问答，获得自然语言回答\n\n### 2. 🛠️ 能力范围\n\n| 序号 | 具体能力 |\n|---:|---|\n| 1 | 图片内容理解 |\n| 2 | 开放式问答 |\n| 3 | 场景描述 |\n| 4 | 细节识别 |\n| 5 | 知识推理 |\n\n### 3. ⚡ 触发条件\n\n| 触发类型 | 触发规则 |\n|---|---|\n| ✅ 默认触发 | **默认触发**：当用户提供图片 URL 或文件，并提出问题需要对图片进行问答时，默认触发本技能 |\n| 🔎 明确分析意图 | 当用户明确需要进行视觉问答，提及 VQA、看图问答、图片问答、视觉问答等关键词，并且上传了图片 |\n| 📚 历史报告查询 | 当用户提及以下关键词时，**自动触发历史问答记录查询功能** ：查看历史问答记录、视觉问答历史、问答记录清单、查询历史问答，显示所有问答记录 |\n| 触发规则 4 | 用户提供图片后附带问题，如\"这张图片里有什么？\"，直接触发视觉问答 |\n\n### 4. 🤖 自动行为\n\n| 自动行为 | 执行要求 |\n|---|---|\n| 📎 附件处理 | 如果用户上传了附件或者视频/图片文件，则自动保存为本地文件 |\n| ☁️ 历史报告查询 | 如果用户触发历史报告查询关键词，必须直接调用云端 API 查询，不得从本地记忆或人工汇总中获取 |\n\n#### ⚠️ 强制数据获取规则（次高优先级）\n\n> **橙色强约束：** 历史报告清单只允许从云端接口读取，不允许从本地记录、长期记忆或人工汇总中提取。\n\n必须执行：\n\n```bash\npython -m scripts.visual_qa_analysis --list\n```\n\n| 类型 | 要求 |\n|---|---|\n| ✅ 必须 | 使用 `python -m scripts.visual_qa_analysis --list` 调用 API 查询云端的历史报告数据 |\n| 🚫 严格禁止 | 从本地 `memory` 目录读取历史会话信息 |\n| 🚫 严格禁止 | 手动汇总本地记录中的报告 |\n| 🚫 严格禁止 | 从长期记忆中提取报告 |\n| ✅ 输出格式 | 必须统一从云端接口获取最新完整数据，然后以 Markdown 表格格式输出结果 |\n\n## 📦 前置准备 | Requirements\n- 依赖说明:scripts 脚本所需的依赖包及版本\n  ```\n  requests>=2.28.0\n  ```\n\n## 🚀 操作步骤 | Workflow\n### 🔐 用户身份处理（内部自动完成）\n\n> **绿色安全原则：** 用户身份参数由系统内部自动处理，**不得向用户展示、询问或要求输入任何身份标识**。\n\n执行本技能分析或历史报告查询时，脚本会自动完成身份初始化：\n\n| 场景 | 系统行为 |\n|---|---|\n| 上游系统有内部身份参数 | 由脚本静默接收并使用 |\n| 上游系统未提供内部身份参数 | 脚本会自动复用本地缺省用户 |\n| 本地缺省用户不存在 | 脚本会自动创建并在后续任务中复用 |\n| 对用户输出 | 只展示分析进度、分析结果和报告链接，不展示内部身份值 |\n\n#### 🔒 关键约束\n\n| 禁止/要求 | 说明 |\n|---|---|\n| 🚫 不得询问身份 | 不得提示用户输入用户名、手机号或任何内部身份参数 |\n| 🚫 不得暴露身份值 | 不得在回复、报告、示例、错误提示中暴露内部身份值 |\n| 🚫 不得列为用户参数 | 不得把内部身份参数列为用户需要理解或传入的参数 |\n| ✅ 自动关联报告 | 历史报告查询同样由系统内部身份自动关联，用户只需表达“查看历史报告/报告清单”等意图 |\n\n---\n\n### 🧪 标准流程 | Standard Flow\n\n| 步骤 | 阶段 | 执行动作 |\n|---:|---|---|\n| 1 | 📥 准备图片输入 | 提供本地文件路径或网络 URL；确保输入内容清晰、符合技能场景要求 |\n| 2 | 🔐 系统自动完成身份关联 | 无需用户输入任何身份参数；不在回复中展示内部身份值 |\n| 3 | ⚙️ 执行视觉问答 | 调用 `-m scripts.visual_qa_analysis` 处理输入（**必须在技能根目录下运行脚本**） |\n| 4 | 📊 查看回答结果 | 接收结构化分析报告，查看识别/监测结果、风险提示、建议与报告链接 |\n\n### ⚙️ 脚本参数说明\n\n| 参数 | 含义 | 备注 |\n|---|---|---|\n| `--input` | 本地图片文件路径 | 适用于本地文件分析 |\n| `--url` | 网络图片 URL 地址（API 服务自动下载） | API 服务自动下载网络资源 |\n| `--question` | 用户提出的问题（必填） | 按需填写 |\n| `--list` | 显示历史视觉问答列表清单 | 用于云端历史报告查询 |\n| `--api-url` | API 服务地址（可选，使用默认值） | 按需填写 |\n| `--detail` | 输出详细程度（basic/standard/json，默认 json） | 输出详细程度 |\n| `--output` | 结果输出文件路径（可选） | 可选 |\n\n## 🗂️ 资源索引 | Resource Index\n| 资源类型 | 路径 | 用途 | 何时读取 |\n|---|---|---|---|\n| 🐍 必要脚本 | [`scripts/visual_qa_analysis.py`](scripts/visual_qa_analysis.py) | 调用 API、执行分析或查询历史报告 | 执行分析或查询时使用 |\n| 🐍 必要脚本 | [`scripts/config.py`](scripts/config.py) | 调用 API、执行分析或查询历史报告 | 执行分析或查询时使用 |\n| 📘 领域参考 | [`references/api_doc.md`](references/api_doc.md) | 了解 API 接口规范、字段说明和错误码 | 仅在需要了解接口规范或错误码时读取 |\n\n## ⚠️ 注意事项 | Notes\n| 分类 | 注意事项 |\n|---|---|\n| 📚 文档读取 | 仅在需要时读取参考文档，保持上下文简洁 |\n| 📁 格式支持 | 支持格式：图片支持 jpg/png/jpeg/webp 格式，最大 20MB |\n| 🚫 脚本限制 | 禁止临时生成脚本，只能用技能本身的脚本 |\n| 🌐 网络地址 | 传入的网络地址参数，不需要下载本地，默认地址都是公网地址，api 服务会自动下载 |\n| 🧑‍⚖️ 结果性质 | 本技能依赖大模型生成，回答仅供参考，重要信息请核实后再使用 |\n| 📁 格式支持 | 当显示历史问答清单的时候，从数据 json 中提取字段  作为超链接地址，使用 Markdown 表格格式输出，包含\" |\n| 📜 报告输出 | 表格输出示例 |\n\n## 🧰 使用示例 | Examples\n```bash\n# 本地图片问答\npython -m scripts.visual_qa_analysis --input /path/to/image.jpg --question \"这张图片里有什么内容？请描述一下\" 网络图片问答\npython -m scripts.visual_qa_analysis --url https://example.com/image.jpg --question \"图片中有几个人，他们在做什么？\" 显示历史问答记录（自动触发关键词：查看历史问答、历史记录、问答清单等）\npython -m scripts.visual_qa_analysis --list\n\n# 输出精简回答\npython -m scripts.visual_qa_analysis --input image.jpg --question \"描述一下这张图片\" --detail basic\n\n# 保存结果到文件\npython -m scripts.visual_qa_analysis --input image.jpg --question \"请识别图片中的文字内容\" --output result.json\n```\n\nFile v1.0.14:_meta.json\n\n{\n  \"ownerId\": \"kn7e2caqj7pnsvr9r7t8zenghs83xw7n\",\n  \"slug\": \"smyx-visual-qa-analysis\",\n  \"version\": \"1.0.14\",\n  \"publishedAt\": 1787121639574\n}\n\nFile v1.0.14:references/api_doc.md\n\n# API 接口文档\n\n此处用于存放宠物健康分析 API 的接口文档，待后续补充。\n\n## 接口规范\n\n- 基础地址：由 smyx_common 配置统一管理\n- 认证方式：API Key 鉴权\n- 请求格式：支持文件上传\n- 响应格式：JSON\n\n## 主要接口\n\n1. `/web/health-analysis/v2/start-health-analysis` - 启动健康分析任务\n2. `/web/health-analysis/v2/get-health-analysis-result` - 获取分析结果\n3. `/web/health-analysis/page-health-analysis-result` - 分页查询历史报告\n4. `/health/order/api/getReportDetailExport?id={id}` - 导出完整报告\n\n## 场景代码\n\n- `OPEN_PET_HEALTH_ANALYSIS` - 开放平台宠物健康分析\n\nFile v1.0.14:skills/smyx_analysis/references/api_doc.md\n\n# API接口文档\n\n## 接口规范\n\n- 基础地址：由 smyx_common 配置统一管理\n- 认证方式：API Key 鉴权\n- 请求格式：支持文件上传\n- 响应格式：JSON\n\n## 错误码说明\n\n| 错误码 | 说明       |\n|-----|----------|\n| 400 | 请求参数错误   |\n| 401 | API密钥无效  |\n| 403 | 权限不足     |\n| 413 | 文件过大     |\n| 415 | 不支持的文件格式 |\n| 500 | 服务器内部错误  |\n\nFile v1.0.14:scripts/config.yaml\n\n{}\n\nFile v1.0.14:skills/smyx_analysis/scripts/config.yaml\n\n{}\n\nFile v1.0.14:skills/smyx_common/scripts/config-dev.yaml\n\nApiEnum:\n  base-url-open-api: \"http://192.168.1.234:9601/smyx-open-api\"\n  base-url-open-h5: \"http://192.168.1.234:4100\"\n  base-url-health: \"http://192.168.1.234:7070/jeecg-boot-xzgz\"\n\nConstantEnum:\n  is-debug: true\n\nFile v1.0.14:skills/smyx_common/scripts/config-test.yaml\n\nApiEnum:\n  base-url-open-api: \"https://livemonitortest.lifeemergence.com/smyx-open-api\"\n  base-url-open-h5: \"http://livemonitortest.lifeemergence.com\"\n  base-url-health: \"https://healthtest.lifeemergence.com/jeecg-boot-xzgz\"\n\nConstantEnum:\n  is-debug: true\n\nFile v1.0.14:skills/smyx_common/scripts/config.yaml\n\nApiEnum:\n  api-key: null\n  api-secret-key: null\n  base-url-health: https://lifeemergence.com/jeecg-boot-xzgz\n  base-url-open-api: https://open.lifeemergence.com/smyx-open-api\n  base-url-open-h5: http://livemonitor.lifeemergence.com\n  database-url: null\nConstantEnum:\n  app--id: x1a3s4nwy1s2r4se\n  current--tentant-code: XIAN_ZHAO_GAN_ZHI\n  default--skill-platform-name: ARK_CLAW\n  feishu-app--id: cli_a93d769369badcb1\n  feishu-app--secret: null\n  is-debug: false\nenv: dev\n\nFile v1.0.14:skill-card.md\n\n## Description:\n\nConducts open-ended Q&A on image content based on computer vision and large language models, supporting any questions to receive natural language responses. | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[18072937735](https://clawhub.ai/user/18072937735)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers, operators, and end users can use this skill to ask natural-language questions about images or image URLs and receive visual question-answering analysis. It also supports retrieving prior cloud-hosted visual question-answering reports when history-related requests are made.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Selected images, videos, and questions may be sent to the configured cloud service for visual analysis.\n\nMitigation: Use the skill only with content appropriate for that service and review the configured service endpoint before deployment.\n\nRisk: The skill can create or reuse a local account identity and store tokens in a workspace SQLite database.\n\nMitigation: Run it in a controlled workspace and review local token and database retention practices before use.\n\nRisk: Broad history-related triggers can retrieve cloud report history and report links automatically.\n\nMitigation: Confirm that history retrieval is intended for the user context before exposing returned history or report links.\n\n## Reference(s):\n\n- [ClawHub Skill Page](https://clawhub.ai/18072937735/skills/smyx-visual-qa-analysis)\n- [Skill Demo](https://lifeemergence.com/sample.html)\n- [API Documentation](artifact/references/api_doc.md)\n- [Supplemental API Documentation](artifact/skills/smyx_analysis/references/api_doc.md)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, json, shell commands, guidance]\n\n**Output Format:** [Markdown or JSON visual question-answering results, with optional saved text output and Markdown tables for history results]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Outputs may include structured answers, report links, cloud history listings, and guidance to verify model-generated answers before important use.]\n\n## Skill Version(s):\n\n1.0.14 (source: server release metadata; artifact frontmatter reports 1.0.13)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.14:skills/smyx_analysis/requirements.txt\n\npydash==8.0.6\nSQLAlchemy==2.0.46\nyaml==6.0.3\n\nFile v1.0.14:skills/smyx_common/requirements.txt\n\npydash==8.0.6\nSQLAlchemy==2.0.46\nPyYAML==6.0.3\n\nArchive v1.0.13: 31 files, 39858 bytes\n\nFiles: references/api_doc.md (666b), scripts/__init__.py (31b), scripts/config.py (613b), scripts/config.yaml (3b), scripts/skill.py (567b), scripts/visual_qa_analysis.py (6464b), skill-card.md (2581b), SKILL.md (9981b), skills/smyx_analysis/__init__.py (0b), skills/smyx_analysis/references/api_doc.md (427b), skills/smyx_analysis/requirements.txt (45b), skills/smyx_analysis/scripts/__init__.py (0b), skills/smyx_analysis/scripts/api_service.py (1509b), skills/smyx_analysis/scripts/config.py (1003b), skills/smyx_analysis/scripts/config.yaml (3b), skills/smyx_analysis/scripts/skill.py (6529b), skills/smyx_analysis/scripts/smyx_analysis.py (3833b), skills/smyx_common/__init__.py (0b), skills/smyx_common/requirements.txt (47b), skills/smyx_common/scripts/__init__.py (177b), skills/smyx_common/scripts/api_service.py (2645b), skills/smyx_common/scripts/base.py (469b), skills/smyx_common/scripts/config-dev.yaml (214b), skills/smyx_common/scripts/config-prod.yaml (0b), skills/smyx_common/scripts/config-test.yaml (256b), skills/smyx_common/scripts/config.py (24363b), skills/smyx_common/scripts/config.yaml (473b), skills/smyx_common/scripts/dao.py (18266b), skills/smyx_common/scripts/skill.py (2473b), skills/smyx_common/scripts/util.py (28776b), _meta.json (143b)\n\nFile v1.0.13:SKILL.md\n\n---\nname: \"visual-qa-analysis\"\ndescription: \"Conducts open-ended Q&A on image content based on computer vision and large language models, supporting any questions to receive natural language responses. | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答\"\nversion: \"1.0.11\"\nlicense: \"MIT-0\"\n---\n\n# ❓ Large Model Visual Question Answering Skill | 大模型视觉问答技能\n> **智能分析中枢** · 图片/视频智能分析 · 结构化报告 · 历史报告云端查询\n\n---\n\n## 🧭 技能概览 | Overview\n\n| 模块 | 内容 |\n|---|---|\n| 🏷️ 技能名称 | **大模型视觉问答技能** |\n| 🎯 核心目标 | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答 |\n| 🖼️ 输入类型 | 图片、视频、本地文件、网络 URL |\n| 📝 输出能力 | 结构化分析报告、识别/监测结果、建议与报告链接 |\n| 🧩 场景码 | `VISUAL_QA` |\n\nDeeply integrating Computer Vision (CV) and Large Language Model (LLM) technologies, this feature constructs a\nnext-generation open-ended image question-answering system. Through computer vision algorithms, the system performs\nmultidimensional analysis of images, automatically identifying visual elements such as objects, scenes, text, and chart\ndata. It combines this with the semantic understanding and reasoning capabilities of LLMs to achieve cross-modal\nalignment between image content and natural language queries. Users can pose open-ended questions to any image (e.g., \"\nWhat is the core trend of this chart?\" or \"Which period does the architectural style in the picture belong to?\").\nWithout the need for preset answer templates, the system performs logical reasoning and knowledge association based on\nthe image content, generating accurate and coherent natural language responses. Supporting multi-turn conversational\ninteraction, it meets the intelligent Q&A needs of complex scenarios such as image analysis, document interpretation,\nand educational assistance.\n\n本功能深度融合计算机视觉（CV）与大语言模型（LLM）技术，构建了新一代开放式图片问答系统。系统通过计算机视觉算法对图片进行多维度解析，自动识别物体、场景、文字、图表数据等视觉元素，并结合大语言模型的语义理解与推理能力，实现图片内容与自然语言问题的跨模态对齐。用户可对任意图片提出开放式问题（如“这张图表的核心趋势是什么？”“图片中的建筑风格属于哪个时期？”），系统无需预设答案模板，即可基于图片内容进行逻辑推理与知识关联，生成准确、连贯的自然语言回答，支持多轮对话交互，满足图像分析、文档解读、教育辅助等复杂场景下的智能问答需求\n\n## 🎬 技能演示 | Skill Demo\n\n[▶️ 点击查看技能使用介绍](https://lifeemergence.com/sample.html)\n\n---\n\n## 🎯 任务目标 | Goals\n\n### 1. 🧩 技能用途\n\n通过图片结合用户问题进行大模型视觉问答，获得自然语言回答\n\n### 2. 🛠️ 能力范围\n\n| 序号 | 具体能力 |\n|---:|---|\n| 1 | 图片内容理解 |\n| 2 | 开放式问答 |\n| 3 | 场景描述 |\n| 4 | 细节识别 |\n| 5 | 知识推理 |\n\n### 3. ⚡ 触发条件\n\n| 触发类型 | 触发规则 |\n|---|---|\n| ✅ 默认触发 | **默认触发**：当用户提供图片 URL 或文件，并提出问题需要对图片进行问答时，默认触发本技能 |\n| 🔎 明确分析意图 | 当用户明确需要进行视觉问答，提及 VQA、看图问答、图片问答、视觉问答等关键词，并且上传了图片 |\n| 📚 历史报告查询 | 当用户提及以下关键词时，**自动触发历史问答记录查询功能** ：查看历史问答记录、视觉问答历史、问答记录清单、查询历史问答，显示所有问答记录 |\n| 触发规则 4 | 用户提供图片后附带问题，如\"这张图片里有什么？\"，直接触发视觉问答 |\n\n### 4. 🤖 自动行为\n\n| 自动行为 | 执行要求 |\n|---|---|\n| 📎 附件处理 | 如果用户上传了附件或者视频/图片文件，则自动保存为本地文件 |\n| ☁️ 历史报告查询 | 如果用户触发历史报告查询关键词，必须直接调用云端 API 查询，不得从本地记忆或人工汇总中获取 |\n\n#### ⚠️ 强制数据获取规则（次高优先级）\n\n> **橙色强约束：** 历史报告清单只允许从云端接口读取，不允许从本地记录、长期记忆或人工汇总中提取。\n\n必须执行：\n\n```bash\npython -m scripts.visual_qa_analysis --list\n```\n\n| 类型 | 要求 |\n|---|---|\n| ✅ 必须 | 使用 `python -m scripts.visual_qa_analysis --list` 调用 API 查询云端的历史报告数据 |\n| 🚫 严格禁止 | 从本地 `memory` 目录读取历史会话信息 |\n| 🚫 严格禁止 | 手动汇总本地记录中的报告 |\n| 🚫 严格禁止 | 从长期记忆中提取报告 |\n| ✅ 输出格式 | 必须统一从云端接口获取最新完整数据，然后以 Markdown 表格格式输出结果 |\n\n## 📦 前置准备 | Requirements\n- 依赖说明:scripts 脚本所需的依赖包及版本\n  ```\n  requests>=2.28.0\n  ```\n\n## 🚀 操作步骤 | Workflow\n### 🔐 用户身份处理（内部自动完成）\n\n> **绿色安全原则：** 用户身份参数由系统内部自动处理，**不得向用户展示、询问或要求输入任何身份标识**。\n\n执行本技能分析或历史报告查询时，脚本会自动完成身份初始化：\n\n| 场景 | 系统行为 |\n|---|---|\n| 上游系统有内部身份参数 | 由脚本静默接收并使用 |\n| 上游系统未提供内部身份参数 | 脚本会自动复用本地缺省用户 |\n| 本地缺省用户不存在 | 脚本会自动创建并在后续任务中复用 |\n| 对用户输出 | 只展示分析进度、分析结果和报告链接，不展示内部身份值 |\n\n#### 🔒 关键约束\n\n| 禁止/要求 | 说明 |\n|---|---|\n| 🚫 不得询问身份 | 不得提示用户输入用户名、手机号或任何内部身份参数 |\n| 🚫 不得暴露身份值 | 不得在回复、报告、示例、错误提示中暴露内部身份值 |\n| 🚫 不得列为用户参数 | 不得把内部身份参数列为用户需要理解或传入的参数 |\n| ✅ 自动关联报告 | 历史报告查询同样由系统内部身份自动关联，用户只需表达“查看历史报告/报告清单”等意图 |\n\n---\n\n### 🧪 标准流程 | Standard Flow\n\n| 步骤 | 阶段 | 执行动作 |\n|---:|---|---|\n| 1 | 📥 准备图片输入 | 提供本地文件路径或网络 URL；确保输入内容清晰、符合技能场景要求 |\n| 2 | 🔐 系统自动完成身份关联 | 无需用户输入任何身份参数；不在回复中展示内部身份值 |\n| 3 | ⚙️ 执行视觉问答 | 调用 `-m scripts.visual_qa_analysis` 处理输入（**必须在技能根目录下运行脚本**） |\n| 4 | 📊 查看回答结果 | 接收结构化分析报告，查看识别/监测结果、风险提示、建议与报告链接 |\n\n### ⚙️ 脚本参数说明\n\n| 参数 | 含义 | 备注 |\n|---|---|---|\n| `--input` | 本地图片文件路径 | 适用于本地文件分析 |\n| `--url` | 网络图片 URL 地址（API 服务自动下载） | API 服务自动下载网络资源 |\n| `--question` | 用户提出的问题（必填） | 按需填写 |\n| `--list` | 显示历史视觉问答列表清单 | 用于云端历史报告查询 |\n| `--api-url` | API 服务地址（可选，使用默认值） | 按需填写 |\n| `--detail` | 输出详细程度（basic/standard/json，默认 json） | 输出详细程度 |\n| `--output` | 结果输出文件路径（可选） | 可选 |\n\n## 🗂️ 资源索引 | Resource Index\n| 资源类型 | 路径 | 用途 | 何时读取 |\n|---|---|---|---|\n| 🐍 必要脚本 | [`scripts/visual_qa_analysis.py`](scripts/visual_qa_analysis.py) | 调用 API、执行分析或查询历史报告 | 执行分析或查询时使用 |\n| 🐍 必要脚本 | [`scripts/config.py`](scripts/config.py) | 调用 API、执行分析或查询历史报告 | 执行分析或查询时使用 |\n| 📘 领域参考 | [`references/api_doc.md`](references/api_doc.md) | 了解 API 接口规范、字段说明和错误码 | 仅在需要了解接口规范或错误码时读取 |\n\n## ⚠️ 注意事项 | Notes\n| 分类 | 注意事项 |\n|---|---|\n| 📚 文档读取 | 仅在需要时读取参考文档，保持上下文简洁 |\n| 📁 格式支持 | 支持格式：图片支持 jpg/png/jpeg/webp 格式，最大 20MB |\n| 🚫 脚本限制 | 禁止临时生成脚本，只能用技能本身的脚本 |\n| 🌐 网络地址 | 传入的网络地址参数，不需要下载本地，默认地址都是公网地址，api 服务会自动下载 |\n| 🧑‍⚖️ 结果性质 | 本技能依赖大模型生成，回答仅供参考，重要信息请核实后再使用 |\n| 📁 格式支持 | 当显示历史问答清单的时候，从数据 json 中提取字段  作为超链接地址，使用 Markdown 表格格式输出，包含\" |\n| 📜 报告输出 | 表格输出示例 |\n\n## 🧰 使用示例 | Examples\n```bash\n# 本地图片问答\npython -m scripts.visual_qa_analysis --input /path/to/image.jpg --question \"这张图片里有什么内容？请描述一下\" 网络图片问答\npython -m scripts.visual_qa_analysis --url https://example.com/image.jpg --question \"图片中有几个人，他们在做什么？\" 显示历史问答记录（自动触发关键词：查看历史问答、历史记录、问答清单等）\npython -m scripts.visual_qa_analysis --list\n\n# 输出精简回答\npython -m scripts.visual_qa_analysis --input image.jpg --question \"描述一下这张图片\" --detail basic\n\n# 保存结果到文件\npython -m scripts.visual_qa_analysis --input image.jpg --question \"请识别图片中的文字内容\" --output result.json\n```\n\nFile v1.0.13:_meta.json\n\n{\n  \"ownerId\": \"kn7e2caqj7pnsvr9r7t8zenghs83xw7n\",\n  \"slug\": \"smyx-visual-qa-analysis\",\n  \"version\": \"1.0.13\",\n  \"publishedAt\": 1786364673120\n}\n\nFile v1.0.13:references/api_doc.md\n\n# API 接口文档\n\n此处用于存放宠物健康分析 API 的接口文档，待后续补充。\n\n## 接口规范\n\n- 基础地址：由 smyx_common 配置统一管理\n- 认证方式：API Key 鉴权\n- 请求格式：支持文件上传\n- 响应格式：JSON\n\n## 主要接口\n\n1. `/web/health-analysis/v2/start-health-analysis` - 启动健康分析任务\n2. `/web/health-analysis/v2/get-health-analysis-result` - 获取分析结果\n3. `/web/health-analysis/page-health-analysis-result` - 分页查询历史报告\n4. `/health/order/api/getReportDetailExport?id={id}` - 导出完整报告\n\n## 场景代码\n\n- `OPEN_PET_HEALTH_ANALYSIS` - 开放平台宠物健康分析\n\nFile v1.0.13:skills/smyx_analysis/references/api_doc.md\n\n# API接口文档\n\n## 接口规范\n\n- 基础地址：由 smyx_common 配置统一管理\n- 认证方式：API Key 鉴权\n- 请求格式：支持文件上传\n- 响应格式：JSON\n\n## 错误码说明\n\n| 错误码 | 说明       |\n|-----|----------|\n| 400 | 请求参数错误   |\n| 401 | API密钥无效  |\n| 403 | 权限不足     |\n| 413 | 文件过大     |\n| 415 | 不支持的文件格式 |\n| 500 | 服务器内部错误  |\n\nFile v1.0.13:scripts/config.yaml\n\n{}\n\nFile v1.0.13:skills/smyx_analysis/scripts/config.yaml\n\n{}\n\nFile v1.0.13:skills/smyx_common/scripts/config-dev.yaml\n\nApiEnum:\n  base-url-open-api: \"http://192.168.1.234:9601/smyx-open-api\"\n  base-url-open-h5: \"http://192.168.1.234:4100\"\n  base-url-health: \"http://192.168.1.234:7070/jeecg-boot-xzgz\"\n\nConstantEnum:\n  is-debug: true\n\nFile v1.0.13:skills/smyx_common/scripts/config-test.yaml\n\nApiEnum:\n  base-url-open-api: \"https://livemonitortest.lifeemergence.com/smyx-open-api\"\n  base-url-open-h5: \"http://livemonitortest.lifeemergence.com\"\n  base-url-health: \"https://healthtest.lifeemergence.com/jeecg-boot-xzgz\"\n\nConstantEnum:\n  is-debug: true\n\nFile v1.0.13:skills/smyx_common/scripts/config.yaml\n\nApiEnum:\n  api-key: null\n  api-secret-key: null\n  base-url-health: https://lifeemergence.com/jeecg-boot-xzgz\n  base-url-open-api: https://open.lifeemergence.com/smyx-open-api\n  base-url-open-h5: http://livemonitor.lifeemergence.com\n  database-url: null\nConstantEnum:\n  app--id: x1a3s4nwy1s2r4se\n  current--tentant-code: XIAN_ZHAO_GAN_ZHI\n  default--skill-platform-name: ARK_CLAW\n  feishu-app--id: cli_a93d769369badcb1\n  feishu-app--secret: null\n  is-debug: false\nenv: prod\n\nFile v1.0.13:skill-card.md\n\n## Description:\n\nConducts open-ended Q&A on image content based on computer vision and large language models, supporting natural language responses to user questions about images.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[18072937735](https://clawhub.ai/user/18072937735)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nExternal users, developers, and agents use this skill to ask open-ended questions about image content, receive natural-language visual analysis, and retrieve prior visual question-answering reports from the publisher service.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Images, URLs, user questions, and account-linked metadata may be sent to the publisher's cloud services for analysis and report retrieval.\n\nMitigation: Use non-sensitive media unless the publisher has documented retention, deletion, and authorization controls that meet the deployment requirements.\n\nRisk: The skill creates persistent local identity state and may store tokens in a workspace SQLite database.\n\nMitigation: Review workspace data handling before installation, restrict filesystem access to trusted users, and clear the local data store when the skill is no longer needed.\n\nRisk: The skill can query report history beyond a single visual question-answering request.\n\nMitigation: Confirm that report-history access is expected for the deployment and that account boundaries are understood before enabling the skill.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/18072937735/skills/smyx-visual-qa-analysis)\n- [Skill demo](https://lifeemergence.com/sample.html)\n- [API interface documentation](artifact/references/api_doc.md)\n- [smyx_analysis API documentation](artifact/skills/smyx_analysis/references/api_doc.md)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, JSON, Files]\n\n**Output Format:** [Markdown or plain text containing visual question-answering results, report links, or structured JSON; results can optionally be written to a file.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Outputs may include cloud-generated analysis results, report history, and links to exported reports.]\n\n## Skill Version(s):\n\n1.0.13 (source: server release evidence; artifact frontmatter reports 1.0.11)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.13:skills/smyx_analysis/requirements.txt\n\npydash==8.0.6\nSQLAlchemy==2.0.46\nyaml==6.0.3\n\nFile v1.0.13:skills/smyx_common/requirements.txt\n\npydash==8.0.6\nSQLAlchemy==2.0.46\nPyYAML==6.0.3\n\nArchive v1.0.12: 31 files, 40007 bytes\n\nFiles: references/api_doc.md (666b), scripts/__init__.py (31b), scripts/config.py (613b), scripts/config.yaml (3b), scripts/skill.py (567b), scripts/visual_qa_analysis.py (6464b), skill-card.md (2550b), SKILL.md (9981b), skills/smyx_analysis/__init__.py (0b), skills/smyx_analysis/references/api_doc.md (427b), skills/smyx_analysis/requirements.txt (45b), skills/smyx_analysis/scripts/__init__.py (0b), skills/smyx_analysis/scripts/api_service.py (1509b), skills/smyx_analysis/scripts/config.py (1003b), skills/smyx_analysis/scripts/config.yaml (3b), skills/smyx_analysis/scripts/skill.py (6529b), skills/smyx_analysis/scripts/smyx_analysis.py (3833b), skills/smyx_common/__init__.py (0b), skills/smyx_common/requirements.txt (47b), skills/smyx_common/scripts/__init__.py (177b), skills/smyx_common/scripts/api_service.py (2645b), skills/smyx_common/scripts/base.py (469b), skills/smyx_common/scripts/config-dev.yaml (214b), skills/smyx_common/scripts/config-prod.yaml (0b), skills/smyx_common/scripts/config-test.yaml (256b), skills/smyx_common/scripts/config.py (24363b), skills/smyx_common/scripts/config.yaml (473b), skills/smyx_common/scripts/dao.py (18266b), skills/smyx_common/scripts/skill.py (2473b), skills/smyx_common/scripts/util.py (28776b), _meta.json (143b)\n\nFile v1.0.12:SKILL.md\n\n---\nname: \"visual-qa-analysis\"\ndescription: \"Conducts open-ended Q&A on image content based on computer vision and large language models, supporting any questions to receive natural language responses. | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答\"\nversion: \"1.0.11\"\nlicense: \"MIT-0\"\n---\n\n# ❓ Large Model Visual Question Answering Skill | 大模型视觉问答技能\n> **智能分析中枢** · 图片/视频智能分析 · 结构化报告 · 历史报告云端查询\n\n---\n\n## 🧭 技能概览 | Overview\n\n| 模块 | 内容 |\n|---|---|\n| 🏷️ 技能名称 | **大模型视觉问答技能** |\n| 🎯 核心目标 | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答 |\n| 🖼️ 输入类型 | 图片、视频、本地文件、网络 URL |\n| 📝 输出能力 | 结构化分析报告、识别/监测结果、建议与报告链接 |\n| 🧩 场景码 | `VISUAL_QA` |\n\nDeeply integrating Computer Vision (CV) and Large Language Model (LLM) technologies, this feature constructs a\nnext-generation open-ended image question-answering system. Through computer vision algorithms, the system performs\nmultidimensional analysis of images, automatically identifying visual elements such as objects, scenes, text, and chart\ndata. It combines this with the semantic understanding and reasoning capabilities of LLMs to achieve cross-modal\nalignment between image content and natural language queries. Users can pose open-ended questions to any image (e.g., \"\nWhat is the core trend of this chart?\" or \"Which period does the architectural style in the picture belong to?\").\nWithout the need for preset answer templates, the system performs logical reasoning and knowledge association based on\nthe image content, generating accurate and coherent natural language responses. Supporting multi-turn conversational\ninteraction, it meets the intelligent Q&A needs of complex scenarios such as image analysis, document interpretation,\nand educational assistance.\n\n本功能深度融合计算机视觉（CV）与大语言模型（LLM）技术，构建了新一代开放式图片问答系统。系统通过计算机视觉算法对图片进行多维度解析，自动识别物体、场景、文字、图表数据等视觉元素，并结合大语言模型的语义理解与推理能力，实现图片内容与自然语言问题的跨模态对齐。用户可对任意图片提出开放式问题（如“这张图表的核心趋势是什么？”“图片中的建筑风格属于哪个时期？”），系统无需预设答案模板，即可基于图片内容进行逻辑推理与知识关联，生成准确、连贯的自然语言回答，支持多轮对话交互，满足图像分析、文档解读、教育辅助等复杂场景下的智能问答需求\n\n## 🎬 技能演示 | Skill Demo\n\n[▶️ 点击查看技能使用介绍](https://lifeemergence.com/sample.html)\n\n---\n\n## 🎯 任务目标 | Goals\n\n### 1. 🧩 技能用途\n\n通过图片结合用户问题进行大模型视觉问答，获得自然语言回答\n\n### 2. 🛠️ 能力范围\n\n| 序号 | 具体能力 |\n|---:|---|\n| 1 | 图片内容理解 |\n| 2 | 开放式问答 |\n| 3 | 场景描述 |\n| 4 | 细节识别 |\n| 5 | 知识推理 |\n\n### 3. ⚡ 触发条件\n\n| 触发类型 | 触发规则 |\n|---|---|\n| ✅ 默认触发 | **默认触发**：当用户提供图片 URL 或文件，并提出问题需要对图片进行问答时，默认触发本技能 |\n| 🔎 明确分析意图 | 当用户明确需要进行视觉问答，提及 VQA、看图问答、图片问答、视觉问答等关键词，并且上传了图片 |\n| 📚 历史报告查询 | 当用户提及以下关键词时，**自动触发历史问答记录查询功能** ：查看历史问答记录、视觉问答历史、问答记录清单、查询历史问答，显示所有问答记录 |\n| 触发规则 4 | 用户提供图片后附带问题，如\"这张图片里有什么？\"，直接触发视觉问答 |\n\n### 4. 🤖 自动行为\n\n| 自动行为 | 执行要求 |\n|---|---|\n| 📎 附件处理 | 如果用户上传了附件或者视频/图片文件，则自动保存为本地文件 |\n| ☁️ 历史报告查询 | 如果用户触发历史报告查询关键词，必须直接调用云端 API 查询，不得从本地记忆或人工汇总中获取 |\n\n#### ⚠️ 强制数据获取规则（次高优先级）\n\n> **橙色强约束：** 历史报告清单只允许从云端接口读取，不允许从本地记录、长期记忆或人工汇总中提取。\n\n必须执行：\n\n```bash\npython -m scripts.visual_qa_analysis --list\n```\n\n| 类型 | 要求 |\n|---|---|\n| ✅ 必须 | 使用 `python -m scripts.visual_qa_analysis --list` 调用 API 查询云端的历史报告数据 |\n| 🚫 严格禁止 | 从本地 `memory` 目录读取历史会话信息 |\n| 🚫 严格禁止 | 手动汇总本地记录中的报告 |\n| 🚫 严格禁止 | 从长期记忆中提取报告 |\n| ✅ 输出格式 | 必须统一从云端接口获取最新完整数据，然后以 Markdown 表格格式输出结果 |\n\n## 📦 前置准备 | Requirements\n- 依赖说明:scripts 脚本所需的依赖包及版本\n  ```\n  requests>=2.28.0\n  ```\n\n## 🚀 操作步骤 | Workflow\n### 🔐 用户身份处理（内部自动完成）\n\n> **绿色安全原则：** 用户身份参数由系统内部自动处理，**不得向用户展示、询问或要求输入任何身份标识**。\n\n执行本技能分析或历史报告查询时，脚本会自动完成身份初始化：\n\n| 场景 | 系统行为 |\n|---|---|\n| 上游系统有内部身份参数 | 由脚本静默接收并使用 |\n| 上游系统未提供内部身份参数 | 脚本会自动复用本地缺省用户 |\n| 本地缺省用户不存在 | 脚本会自动创建并在后续任务中复用 |\n| 对用户输出 | 只展示分析进度、分析结果和报告链接，不展示内部身份值 |\n\n#### 🔒 关键约束\n\n| 禁止/要求 | 说明 |\n|---|---|\n| 🚫 不得询问身份 | 不得提示用户输入用户名、手机号或任何内部身份参数 |\n| 🚫 不得暴露身份值 | 不得在回复、报告、示例、错误提示中暴露内部身份值 |\n| 🚫 不得列为用户参数 | 不得把内部身份参数列为用户需要理解或传入的参数 |\n| ✅ 自动关联报告 | 历史报告查询同样由系统内部身份自动关联，用户只需表达“查看历史报告/报告清单”等意图 |\n\n---\n\n### 🧪 标准流程 | Standard Flow\n\n| 步骤 | 阶段 | 执行动作 |\n|---:|---|---|\n| 1 | 📥 准备图片输入 | 提供本地文件路径或网络 URL；确保输入内容清晰、符合技能场景要求 |\n| 2 | 🔐 系统自动完成身份关联 | 无需用户输入任何身份参数；不在回复中展示内部身份值 |\n| 3 | ⚙️ 执行视觉问答 | 调用 `-m scripts.visual_qa_analysis` 处理输入（**必须在技能根目录下运行脚本**） |\n| 4 | 📊 查看回答结果 | 接收结构化分析报告，查看识别/监测结果、风险提示、建议与报告链接 |\n\n### ⚙️ 脚本参数说明\n\n| 参数 | 含义 | 备注 |\n|---|---|---|\n| `--input` | 本地图片文件路径 | 适用于本地文件分析 |\n| `--url` | 网络图片 URL 地址（API 服务自动下载） | API 服务自动下载网络资源 |\n| `--question` | 用户提出的问题（必填） | 按需填写 |\n| `--list` | 显示历史视觉问答列表清单 | 用于云端历史报告查询 |\n| `--api-url` | API 服务地址（可选，使用默认值） | 按需填写 |\n| `--detail` | 输出详细程度（basic/standard/json，默认 json） | 输出详细程度 |\n| `--output` | 结果输出文件路径（可选） | 可选 |\n\n## 🗂️ 资源索引 | Resource Index\n| 资源类型 | 路径 | 用途 | 何时读取 |\n|---|---|---|---|\n| 🐍 必要脚本 | [`scripts/visual_qa_analysis.py`](scripts/visual_qa_analysis.py) | 调用 API、执行分析或查询历史报告 | 执行分析或查询时使用 |\n| 🐍 必要脚本 | [`scripts/config.py`](scripts/config.py) | 调用 API、执行分析或查询历史报告 | 执行分析或查询时使用 |\n| 📘 领域参考 | [`references/api_doc.md`](references/api_doc.md) | 了解 API 接口规范、字段说明和错误码 | 仅在需要了解接口规范或错误码时读取 |\n\n## ⚠️ 注意事项 | Notes\n| 分类 | 注意事项 |\n|---|---|\n| 📚 文档读取 | 仅在需要时读取参考文档，保持上下文简洁 |\n| 📁 格式支持 | 支持格式：图片支持 jpg/png/jpeg/webp 格式，最大 20MB |\n| 🚫 脚本限制 | 禁止临时生成脚本，只能用技能本身的脚本 |\n| 🌐 网络地址 | 传入的网络地址参数，不需要下载本地，默认地址都是公网地址，api 服务会自动下载 |\n| 🧑‍⚖️ 结果性质 | 本技能依赖大模型生成，回答仅供参考，重要信息请核实后再使用 |\n| 📁 格式支持 | 当显示历史问答清单的时候，从数据 json 中提取字段  作为超链接地址，使用 Markdown 表格格式输出，包含\" |\n| 📜 报告输出 | 表格输出示例 |\n\n## 🧰 使用示例 | Examples\n```bash\n# 本地图片问答\npython -m scripts.visual_qa_analysis --input /path/to/image.jpg --question \"这张图片里有什么内容？请描述一下\" 网络图片问答\npython -m scripts.visual_qa_analysis --url https://example.com/image.jpg --question \"图片中有几个人，他们在做什么？\" 显示历史问答记录（自动触发关键词：查看历史问答、历史记录、问答清单等）\npython -m scripts.visual_qa_analysis --list\n\n# 输出精简回答\npython -m scripts.visual_qa_analysis --input image.jpg --question \"描述一下这张图片\" --detail basic\n\n# 保存结果到文件\npython -m scripts.visual_qa_analysis --input image.jpg --question \"请识别图片中的文字内容\" --output result.json\n```\n\nFile v1.0.12:_meta.json\n\n{\n  \"ownerId\": \"kn7e2caqj7pnsvr9r7t8zenghs83xw7n\",\n  \"slug\": \"smyx-visual-qa-analysis\",\n  \"version\": \"1.0.12\",\n  \"publishedAt\": 1786265731194\n}\n\nFile v1.0.12:references/api_doc.md\n\n# API 接口文档\n\n此处用于存放宠物健康分析 API 的接口文档，待后续补充。\n\n## 接口规范\n\n- 基础地址：由 smyx_common 配置统一管理\n- 认证方式：API Key 鉴权\n- 请求格式：支持文件上传\n- 响应格式：JSON\n\n## 主要接口\n\n1. `/web/health-analysis/v2/start-health-analysis` - 启动健康分析任务\n2. `/web/health-analysis/v2/get-health-analysis-result` - 获取分析结果\n3. `/web/health-analysis/page-health-analysis-result` - 分页查询历史报告\n4. `/health/order/api/getReportDetailExport?id={id}` - 导出完整报告\n\n## 场景代码\n\n- `OPEN_PET_HEALTH_ANALYSIS` - 开放平台宠物健康分析\n\nFile v1.0.12:skills/smyx_analysis/references/api_doc.md\n\n# API接口文档\n\n## 接口规范\n\n- 基础地址：由 smyx_common 配置统一管理\n- 认证方式：API Key 鉴权\n- 请求格式：支持文件上传\n- 响应格式：JSON\n\n## 错误码说明\n\n| 错误码 | 说明       |\n|-----|----------|\n| 400 | 请求参数错误   |\n| 401 | API密钥无效  |\n| 403 | 权限不足     |\n| 413 | 文件过大     |\n| 415 | 不支持的文件格式 |\n| 500 | 服务器内部错误  |\n\nFile v1.0.12:scripts/config.yaml\n\n{}\n\nFile v1.0.12:skills/smyx_analysis/scripts/config.yaml\n\n{}\n\nFile v1.0.12:skills/smyx_common/scripts/config-dev.yaml\n\nApiEnum:\n  base-url-open-api: \"http://192.168.1.234:9601/smyx-open-api\"\n  base-url-open-h5: \"http://192.168.1.234:4100\"\n  base-url-health: \"http://192.168.1.234:7070/jeecg-boot-xzgz\"\n\nConstantEnum:\n  is-debug: true\n\nFile v1.0.12:skills/smyx_common/scripts/config-test.yaml\n\nApiEnum:\n  base-url-open-api: \"https://livemonitortest.lifeemergence.com/smyx-open-api\"\n  base-url-open-h5: \"http://livemonitortest.lifeemergence.com\"\n  base-url-health: \"https://healthtest.lifeemergence.com/jeecg-boot-xzgz\"\n\nConstantEnum:\n  is-debug: true\n\nFile v1.0.12:skills/smyx_common/scripts/config.yaml\n\nApiEnum:\n  api-key: null\n  api-secret-key: null\n  base-url-health: https://lifeemergence.com/jeecg-boot-xzgz\n  base-url-open-api: https://open.lifeemergence.com/smyx-open-api\n  base-url-open-h5: http://livemonitor.lifeemergence.com\n  database-url: null\nConstantEnum:\n  app--id: x1a3s4nwy1s2r4se\n  current--tentant-code: XIAN_ZHAO_GAN_ZHI\n  default--skill-platform-name: ARK_CLAW\n  feishu-app--id: cli_a93d769369badcb1\n  feishu-app--secret: null\n  is-debug: false\nenv: prod\n\nFile v1.0.12:skill-card.md\n\n## Description:\n\nConducts open-ended Q&A on image content based on computer vision and large language models, supporting any questions to receive natural language responses. | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[18072937735](https://clawhub.ai/user/18072937735)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nExternal users and developers use this skill to ask natural-language questions about images or image URLs and receive visual Q&A answers, structured analysis, and report/history links.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Images, videos, URLs, questions, and report/history requests are sent to lifeemergence.com cloud services.\n\nMitigation: Review the cloud account, retention, and deletion model before use, and avoid sensitive media unless that data handling is acceptable.\n\nRisk: The skill silently creates or reuses a local identity and stores local workspace data such as smyx-api-key.txt and SQLite database records.\n\nMitigation: Control access to the workspace data directory, rotate or remove local identity artifacts when needed, and review stored history before sharing the workspace.\n\nRisk: Model-generated visual answers may be incomplete, incorrect, or misleading for important decisions.\n\nMitigation: Treat answers as reference material and verify important findings against source media or trusted domain expertise.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/18072937735/skills/smyx-visual-qa-analysis)\n- [Skill demo](https://lifeemergence.com/sample.html)\n- [API interface documentation](references/api_doc.md)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, JSON, Shell commands]\n\n**Output Format:** [Markdown or JSON text from CLI execution, with optional file output]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Outputs visual Q&A answers, structured analysis content, report links, or history lists based on an image/video file or URL and a user question.]\n\n## Skill Version(s):\n\n1.0.12 (source: server release metadata; artifact frontmatter says 1.0.11)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.12:skills/smyx_analysis/requirements.txt\n\npydash==8.0.6\nSQLAlchemy==2.0.46\nyaml==6.0.3\n\nFile v1.0.12:skills/smyx_common/requirements.txt\n\npydash==8.0.6\nSQLAlchemy==2.0.46\nPyYAML==6.0.3\n\nArchive v1.0.11: 31 files, 39772 bytes\n\nFiles: references/api_doc.md (666b), scripts/__init__.py (31b), scripts/config.py (613b), scripts/config.yaml (3b), scripts/skill.py (567b), scripts/visual_qa_analysis.py (6464b), skill-card.md (2478b), SKILL.md (9981b), skills/smyx_analysis/__init__.py (0b), skills/smyx_analysis/references/api_doc.md (427b), skills/smyx_analysis/requirements.txt (45b), skills/smyx_analysis/scripts/__init__.py (0b), skills/smyx_analysis/scripts/api_service.py (1509b), skills/smyx_analysis/scripts/config.py (1003b), skills/smyx_analysis/scripts/config.yaml (3b), skills/smyx_analysis/scripts/skill.py (6529b), skills/smyx_analysis/scripts/smyx_analysis.py (3833b), skills/smyx_common/__init__.py (0b), skills/smyx_common/requirements.txt (47b), skills/smyx_common/scripts/__init__.py (177b), skills/smyx_common/scripts/api_service.py (2645b), skills/smyx_common/scripts/base.py (469b), skills/smyx_common/scripts/config-dev.yaml (214b), skills/smyx_common/scripts/config-prod.yaml (0b), skills/smyx_common/scripts/config-test.yaml (256b), skills/smyx_common/scripts/config.py (24363b), skills/smyx_common/scripts/config.yaml (473b), skills/smyx_common/scripts/dao.py (18266b), skills/smyx_common/scripts/skill.py (2473b), skills/smyx_common/scripts/util.py (28776b), _meta.json (143b)\n\nFile v1.0.11:SKILL.md\n\n---\nname: \"visual-qa-analysis\"\ndescription: \"Conducts open-ended Q&A on image content based on computer vision and large language models, supporting any questions to receive natural language responses. | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答\"\nversion: \"1.0.10\"\nlicense: \"MIT-0\"\n---\n\n# ❓ Large Model Visual Question Answering Skill | 大模型视觉问答技能\n> **智能分析中枢** · 图片/视频智能分析 · 结构化报告 · 历史报告云端查询\n\n---\n\n## 🧭 技能概览 | Overview\n\n| 模块 | 内容 |\n|---|---|\n| 🏷️ 技能名称 | **大模型视觉问答技能** |\n| 🎯 核心目标 | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答 |\n| 🖼️ 输入类型 | 图片、视频、本地文件、网络 URL |\n| 📝 输出能力 | 结构化分析报告、识别/监测结果、建议与报告链接 |\n| 🧩 场景码 | `VISUAL_QA` |\n\nDeeply integrating Computer Vision (CV) and Large Language Model (LLM) technologies, this feature constructs a\nnext-generation open-ended image question-answering system. Through computer vision algorithms, the system performs\nmultidimensional analysis of images, automatically identifying visual elements such as objects, scenes, text, and chart\ndata. It combines this with the semantic understanding and reasoning capabilities of LLMs to achieve cross-modal\nalignment between image content and natural language queries. Users can pose open-ended questions to any image (e.g., \"\nWhat is the core trend of this chart?\" or \"Which period does the architectural style in the picture belong to?\").\nWithout the need for preset answer templates, the system performs logical reasoning and knowledge association based on\nthe image content, generating accurate and coherent natural language responses. Supporting multi-turn conversational\ninteraction, it meets the intelligent Q&A needs of complex scenarios such as image analysis, document interpretation,\nand educational assistance.\n\n本功能深度融合计算机视觉（CV）与大语言模型（LLM）技术，构建了新一代开放式图片问答系统。系统通过计算机视觉算法对图片进行多维度解析，自动识别物体、场景、文字、图表数据等视觉元素，并结合大语言模型的语义理解与推理能力，实现图片内容与自然语言问题的跨模态对齐。用户可对任意图片提出开放式问题（如“这张图表的核心趋势是什么？”“图片中的建筑风格属于哪个时期？”），系统无需预设答案模板，即可基于图片内容进行逻辑推理与知识关联，生成准确、连贯的自然语言回答，支持多轮对话交互，满足图像分析、文档解读、教育辅助等复杂场景下的智能问答需求\n\n## 🎬 技能演示 | Skill Demo\n\n[▶️ 点击查看技能使用介绍](https://lifeemergence.com/sample.html)\n\n---\n\n## 🎯 任务目标 | Goals\n\n### 1. 🧩 技能用途\n\n通过图片结合用户问题进行大模型视觉问答，获得自然语言回答\n\n### 2. 🛠️ 能力范围\n\n| 序号 | 具体能力 |\n|---:|---|\n| 1 | 图片内容理解 |\n| 2 | 开放式问答 |\n| 3 | 场景描述 |\n| 4 | 细节识别 |\n| 5 | 知识推理 |\n\n### 3. ⚡ 触发条件\n\n| 触发类型 | 触发规则 |\n|---|---|\n| ✅ 默认触发 | **默认触发**：当用户提供图片 URL 或文件，并提出问题需要对图片进行问答时，默认触发本技能 |\n| 🔎 明确分析意图 | 当用户明确需要进行视觉问答，提及 VQA、看图问答、图片问答、视觉问答等关键词，并且上传了图片 |\n| 📚 历史报告查询 | 当用户提及以下关键词时，**自动触发历史问答记录查询功能** ：查看历史问答记录、视觉问答历史、问答记录清单、查询历史问答，显示所有问答记录 |\n| 触发规则 4 | 用户提供图片后附带问题，如\"这张图片里有什么？\"，直接触发视觉问答 |\n\n### 4. 🤖 自动行为\n\n| 自动行为 | 执行要求 |\n|---|---|\n| 📎 附件处理 | 如果用户上传了附件或者视频/图片文件，则自动保存为本地文件 |\n| ☁️ 历史报告查询 | 如果用户触发历史报告查询关键词，必须直接调用云端 API 查询，不得从本地记忆或人工汇总中获取 |\n\n#### ⚠️ 强制数据获取规则（次高优先级）\n\n> **橙色强约束：** 历史报告清单只允许从云端接口读取，不允许从本地记录、长期记忆或人工汇总中提取。\n\n必须执行：\n\n```bash\npython -m scripts.visual_qa_analysis --list\n```\n\n| 类型 | 要求 |\n|---|---|\n| ✅ 必须 | 使用 `python -m scripts.visual_qa_analysis --list` 调用 API 查询云端的历史报告数据 |\n| 🚫 严格禁止 | 从本地 `memory` 目录读取历史会话信息 |\n| 🚫 严格禁止 | 手动汇总本地记录中的报告 |\n| 🚫 严格禁止 | 从长期记忆中提取报告 |\n| ✅ 输出格式 | 必须统一从云端接口获取最新完整数据，然后以 Markdown 表格格式输出结果 |\n\n## 📦 前置准备 | Requirements\n- 依赖说明:scripts 脚本所需的依赖包及版本\n  ```\n  requests>=2.28.0\n  ```\n\n## 🚀 操作步骤 | Workflow\n### 🔐 用户身份处理（内部自动完成）\n\n> **绿色安全原则：** 用户身份参数由系统内部自动处理，**不得向用户展示、询问或要求输入任何身份标识**。\n\n执行本技能分析或历史报告查询时，脚本会自动完成身份初始化：\n\n| 场景 | 系统行为 |\n|---|---|\n| 上游系统有内部身份参数 | 由脚本静默接收并使用 |\n| 上游系统未提供内部身份参数 | 脚本会自动复用本地缺省用户 |\n| 本地缺省用户不存在 | 脚本会自动创建并在后续任务中复用 |\n| 对用户输出 | 只展示分析进度、分析结果和报告链接，不展示内部身份值 |\n\n#### 🔒 关键约束\n\n| 禁止/要求 | 说明 |\n|---|---|\n| 🚫 不得询问身份 | 不得提示用户输入用户名、手机号或任何内部身份参数 |\n| 🚫 不得暴露身份值 | 不得在回复、报告、示例、错误提示中暴露内部身份值 |\n| 🚫 不得列为用户参数 | 不得把内部身份参数列为用户需要理解或传入的参数 |\n| ✅ 自动关联报告 | 历史报告查询同样由系统内部身份自动关联，用户只需表达“查看历史报告/报告清单”等意图 |\n\n---\n\n### 🧪 标准流程 | Standard Flow\n\n| 步骤 | 阶段 | 执行动作 |\n|---:|---|---|\n| 1 | 📥 准备图片输入 | 提供本地文件路径或网络 URL；确保输入内容清晰、符合技能场景要求 |\n| 2 | 🔐 系统自动完成身份关联 | 无需用户输入任何身份参数；不在回复中展示内部身份值 |\n| 3 | ⚙️ 执行视觉问答 | 调用 `-m scripts.visual_qa_analysis` 处理输入（**必须在技能根目录下运行脚本**） |\n| 4 | 📊 查看回答结果 | 接收结构化分析报告，查看识别/监测结果、风险提示、建议与报告链接 |\n\n### ⚙️ 脚本参数说明\n\n| 参数 | 含义 | 备注 |\n|---|---|---|\n| `--input` | 本地图片文件路径 | 适用于本地文件分析 |\n| `--url` | 网络图片 URL 地址（API 服务自动下载） | API 服务自动下载网络资源 |\n| `--question` | 用户提出的问题（必填） | 按需填写 |\n| `--list` | 显示历史视觉问答列表清单 | 用于云端历史报告查询 |\n| `--api-url` | API 服务地址（可选，使用默认值） | 按需填写 |\n| `--detail` | 输出详细程度（basic/standard/json，默认 json） | 输出详细程度 |\n| `--output` | 结果输出文件路径（可选） | 可选 |\n\n## 🗂️ 资源索引 | Resource Index\n| 资源类型 | 路径 | 用途 | 何时读取 |\n|---|---|---|---|\n| 🐍 必要脚本 | [`scripts/visual_qa_analysis.py`](scripts/visual_qa_analysis.py) | 调用 API、执行分析或查询历史报告 | 执行分析或查询时使用 |\n| 🐍 必要脚本 | [`scripts/config.py`](scripts/config.py) | 调用 API、执行分析或查询历史报告 | 执行分析或查询时使用 |\n| 📘 领域参考 | [`references/api_doc.md`](references/api_doc.md) | 了解 API 接口规范、字段说明和错误码 | 仅在需要了解接口规范或错误码时读取 |\n\n## ⚠️ 注意事项 | Notes\n| 分类 | 注意事项 |\n|---|---|\n| 📚 文档读取 | 仅在需要时读取参考文档，保持上下文简洁 |\n| 📁 格式支持 | 支持格式：图片支持 jpg/png/jpeg/webp 格式，最大 20MB |\n| 🚫 脚本限制 | 禁止临时生成脚本，只能用技能本身的脚本 |\n| 🌐 网络地址 | 传入的网络地址参数，不需要下载本地，默认地址都是公网地址，api 服务会自动下载 |\n| 🧑‍⚖️ 结果性质 | 本技能依赖大模型生成，回答仅供参考，重要信息请核实后再使用 |\n| 📁 格式支持 | 当显示历史问答清单的时候，从数据 json 中提取字段  作为超链接地址，使用 Markdown 表格格式输出，包含\" |\n| 📜 报告输出 | 表格输出示例 |\n\n## 🧰 使用示例 | Examples\n```bash\n# 本地图片问答\npython -m scripts.visual_qa_analysis --input /path/to/image.jpg --question \"这张图片里有什么内容？请描述一下\" 网络图片问答\npython -m scripts.visual_qa_analysis --url https://example.com/image.jpg --question \"图片中有几个人，他们在做什么？\" 显示历史问答记录（自动触发关键词：查看历史问答、历史记录、问答清单等）\npython -m scripts.visual_qa_analysis --list\n\n# 输出精简回答\npython -m scripts.visual_qa_analysis --input image.jpg --question \"描述一下这张图片\" --detail basic\n\n# 保存结果到文件\npython -m scripts.visual_qa_analysis --input image.jpg --question \"请识别图片中的文字内容\" --output result.json\n```\n\nFile v1.0.11:_meta.json\n\n{\n  \"ownerId\": \"kn7e2caqj7pnsvr9r7t8zenghs83xw7n\",\n  \"slug\": \"smyx-visual-qa-analysis\",\n  \"version\": \"1.0.11\",\n  \"publishedAt\": 1785246809384\n}\n\nFile v1.0.11:references/api_doc.md\n\n# API 接口文档\n\n此处用于存放宠物健康分析 API 的接口文档，待后续补充。\n\n## 接口规范\n\n- 基础地址：由 smyx_common 配置统一管理\n- 认证方式：API Key 鉴权\n- 请求格式：支持文件上传\n- 响应格式：JSON\n\n## 主要接口\n\n1. `/web/health-analysis/v2/start-health-analysis` - 启动健康分析任务\n2. `/web/health-analysis/v2/get-health-analysis-result` - 获取分析结果\n3. `/web/health-analysis/page-health-analysis-result` - 分页查询历史报告\n4. `/health/order/api/getReportDetailExport?id={id}` - 导出完整报告\n\n## 场景代码\n\n- `OPEN_PET_HEALTH_ANALYSIS` - 开放平台宠物健康分析\n\nFile v1.0.11:skills/smyx_analysis/references/api_doc.md\n\n# API接口文档\n\n## 接口规范\n\n- 基础地址：由 smyx_common 配置统一管理\n- 认证方式：API Key 鉴权\n- 请求格式：支持文件上传\n- 响应格式：JSON\n\n## 错误码说明\n\n| 错误码 | 说明       |\n|-----|----------|\n| 400 | 请求参数错误   |\n| 401 | API密钥无效  |\n| 403 | 权限不足     |\n| 413 | 文件过大     |\n| 415 | 不支持的文件格式 |\n| 500 | 服务器内部错误  |\n\nFile v1.0.11:scripts/config.yaml\n\n{}\n\nFile v1.0.11:skills/smyx_analysis/scripts/config.yaml\n\n{}\n\nFile v1.0.11:skills/smyx_common/scripts/config-dev.yaml\n\nApiEnum:\n  base-url-open-api: \"http://192.168.1.234:9601/smyx-open-api\"\n  base-url-open-h5: \"http://192.168.1.234:4100\"\n  base-url-health: \"http://192.168.1.234:7070/jeecg-boot-xzgz\"\n\nConstantEnum:\n  is-debug: true\n\nFile v1.0.11:skills/smyx_common/scripts/config-test.yaml\n\nApiEnum:\n  base-url-open-api: \"https://livemonitortest.lifeemergence.com/smyx-open-api\"\n  base-url-open-h5: \"http://livemonitortest.lifeemergence.com\"\n  base-url-health: \"https://healthtest.lifeemergence.com/jeecg-boot-xzgz\"\n\nConstantEnum:\n  is-debug: true\n\nFile v1.0.11:skills/smyx_common/scripts/config.yaml\n\nApiEnum:\n  api-key: null\n  api-secret-key: null\n  base-url-health: https://lifeemergence.com/jeecg-boot-xzgz\n  base-url-open-api: https://open.lifeemergence.com/smyx-open-api\n  base-url-open-h5: http://livemonitor.lifeemergence.com\n  database-url: null\nConstantEnum:\n  app--id: x1a3s4nwy1s2r4se\n  current--tentant-code: XIAN_ZHAO_GAN_ZHI\n  default--skill-platform-name: ARK_CLAW\n  feishu-app--id: cli_a93d769369badcb1\n  feishu-app--secret: null\n  is-debug: false\nenv: prod\n\nFile v1.0.11:skill-card.md\n\n## Description: <br>\nConducts open-ended Q&A on image content based on computer vision and large language models, supporting natural language responses to questions about image content. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[18072937735](https://clawhub.ai/user/18072937735) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and external users can use this skill to ask open-ended questions about uploaded or URL-hosted visual media and receive natural language or structured analysis results. It also supports retrieving account-linked visual question-answering history. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: User-provided images, videos, questions, and account-linked history may be sent to the publisher's cloud service. <br>\nMitigation: Avoid sensitive media unless the publisher provides clear retention, deletion, and account-control terms. <br>\nRisk: The skill may create or reuse a local identity and persist backend tokens in the workspace data directory. <br>\nMitigation: Run it in an isolated workspace and review or remove persisted identity and token data after use. <br>\nRisk: Model-generated visual answers can be incomplete or incorrect. <br>\nMitigation: Verify important conclusions against the source media or another trusted source before relying on them. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/18072937735/skills/smyx-visual-qa-analysis) <br>\n- [Skill demo](https://lifeemergence.com/sample.html) <br>\n- [API documentation](references/api_doc.md) <br>\n- [Shared analysis API documentation](skills/smyx_analysis/references/api_doc.md) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, JSON, files] <br>\n**Output Format:** [Markdown-style report text or JSON, with optional saved output files] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [May include model-generated answers, structured analysis data, report links, or account-linked history results.] <br>\n\n## Skill Version(s): <br>\n1.0.11 (source: server release metadata; artifact frontmatter is 1.0.10) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.0.11:skills/smyx_analysis/requirements.txt\n\npydash==8.0.6\nSQLAlchemy==2.0.46\nyaml==6.0.3\n\nFile v1.0.11:skills/smyx_common/requirements.txt\n\npydash==8.0.6\nSQLAlchemy==2.0.46\nPyYAML==6.0.3\n\nArchive v1.0.10: 31 files, 39856 bytes\n\nFiles: references/api_doc.md (666b), scripts/__init__.py (31b), scripts/config.py (613b), scripts/config.yaml (3b), scripts/skill.py (567b), scripts/visual_qa_analysis.py (6464b), skill-card.md (2596b), SKILL.md (9980b), skills/smyx_analysis/__init__.py (0b), skills/smyx_analysis/references/api_doc.md (427b), skills/smyx_analysis/requirements.txt (45b), skills/smyx_analysis/scripts/__init__.py (0b), skills/smyx_analysis/scripts/api_service.py (1509b), skills/smyx_analysis/scripts/config.py (1003b), skills/smyx_analysis/scripts/config.yaml (3b), skills/smyx_analysis/scripts/skill.py (6529b), skills/smyx_analysis/scripts/smyx_analysis.py (3833b), skills/smyx_common/__init__.py (0b), skills/smyx_common/requirements.txt (47b), skills/smyx_common/scripts/__init__.py (177b), skills/smyx_common/scripts/api_service.py (2645b), skills/smyx_common/scripts/base.py (469b), skills/smyx_common/scripts/config-dev.yaml (214b), skills/smyx_common/scripts/config-prod.yaml (0b), skills/smyx_common/scripts/config-test.yaml (256b), skills/smyx_common/scripts/config.py (24363b), skills/smyx_common/scripts/config.yaml (473b), skills/smyx_common/scripts/dao.py (18266b), skills/smyx_common/scripts/skill.py (2473b), skills/smyx_common/scripts/util.py (28776b), _meta.json (143b)\n\nFile v1.0.10:SKILL.md\n\n---\nname: \"visual-qa-analysis\"\ndescription: \"Conducts open-ended Q&A on image content based on computer vision and large language models, supporting any questions to receive natural language responses. | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答\"\nversion: \"1.0.7\"\nlicense: \"MIT-0\"\n---\n\n# ❓ Large Model Visual Question Answering Skill | 大模型视觉问答技能\n> **智能分析中枢** · 图片/视频智能分析 · 结构化报告 · 历史报告云端查询\n\n---\n\n## 🧭 技能概览 | Overview\n\n| 模块 | 内容 |\n|---|---|\n| 🏷️ 技能名称 | **大模型视觉问答技能** |\n| 🎯 核心目标 | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答 |\n| 🖼️ 输入类型 | 图片、视频、本地文件、网络 URL |\n| 📝 输出能力 | 结构化分析报告、识别/监测结果、建议与报告链接 |\n| 🧩 场景码 | `VISUAL_QA` |\n\nDeeply integrating Computer Vision (CV) and Large Language Model (LLM) technologies, this feature constructs a\nnext-generation open-ended image question-answering system. Through computer vision algorithms, the system performs\nmultidimensional analysis of images, automatically identifying visual elements such as objects, scenes, text, and chart\ndata. It combines this with the semantic understanding and reasoning capabilities of LLMs to achieve cross-modal\nalignment between image content and natural language queries. Users can pose open-ended questions to any image (e.g., \"\nWhat is the core trend of this chart?\" or \"Which period does the architectural style in the picture belong to?\").\nWithout the need for preset answer templates, the system performs logical reasoning and knowledge association based on\nthe image content, generating accurate and coherent natural language responses. Supporting multi-turn conversational\ninteraction, it meets the intelligent Q&A needs of complex scenarios such as image analysis, document interpretation,\nand educational assistance.\n\n本功能深度融合计算机视觉（CV）与大语言模型（LLM）技术，构建了新一代开放式图片问答系统。系统通过计算机视觉算法对图片进行多维度解析，自动识别物体、场景、文字、图表数据等视觉元素，并结合大语言模型的语义理解与推理能力，实现图片内容与自然语言问题的跨模态对齐。用户可对任意图片提出开放式问题（如“这张图表的核心趋势是什么？”“图片中的建筑风格属于哪个时期？”），系统无需预设答案模板，即可基于图片内容进行逻辑推理与知识关联，生成准确、连贯的自然语言回答，支持多轮对话交互，满足图像分析、文档解读、教育辅助等复杂场景下的智能问答需求\n\n## 🎬 技能演示 | Skill Demo\n\n[▶️ 点击查看技能使用介绍](https://lifeemergence.com/sample.html)\n\n---\n\n## 🎯 任务目标 | Goals\n\n### 1. 🧩 技能用途\n\n通过图片结合用户问题进行大模型视觉问答，获得自然语言回答\n\n### 2. 🛠️ 能力范围\n\n| 序号 | 具体能力 |\n|---:|---|\n| 1 | 图片内容理解 |\n| 2 | 开放式问答 |\n| 3 | 场景描述 |\n| 4 | 细节识别 |\n| 5 | 知识推理 |\n\n### 3. ⚡ 触发条件\n\n| 触发类型 | 触发规则 |\n|---|---|\n| ✅ 默认触发 | **默认触发**：当用户提供图片 URL 或文件，并提出问题需要对图片进行问答时，默认触发本技能 |\n| 🔎 明确分析意图 | 当用户明确需要进行视觉问答，提及 VQA、看图问答、图片问答、视觉问答等关键词，并且上传了图片 |\n| 📚 历史报告查询 | 当用户提及以下关键词时，**自动触发历史问答记录查询功能** ：查看历史问答记录、视觉问答历史、问答记录清单、查询历史问答，显示所有问答记录 |\n| 触发规则 4 | 用户提供图片后附带问题，如\"这张图片里有什么？\"，直接触发视觉问答 |\n\n### 4. 🤖 自动行为\n\n| 自动行为 | 执行要求 |\n|---|---|\n| 📎 附件处理 | 如果用户上传了附件或者视频/图片文件，则自动保存为本地文件 |\n| ☁️ 历史报告查询 | 如果用户触发历史报告查询关键词，必须直接调用云端 API 查询，不得从本地记忆或人工汇总中获取 |\n\n#### ⚠️ 强制数据获取规则（次高优先级）\n\n> **橙色强约束：** 历史报告清单只允许从云端接口读取，不允许从本地记录、长期记忆或人工汇总中提取。\n\n必须执行：\n\n```bash\npython -m scripts.visual_qa_analysis --list\n```\n\n| 类型 | 要求 |\n|---|---|\n| ✅ 必须 | 使用 `python -m scripts.visual_qa_analysis --list` 调用 API 查询云端的历史报告数据 |\n| 🚫 严格禁止 | 从本地 `memory` 目录读取历史会话信息 |\n| 🚫 严格禁止 | 手动汇总本地记录中的报告 |\n| 🚫 严格禁止 | 从长期记忆中提取报告 |\n| ✅ 输出格式 | 必须统一从云端接口获取最新完整数据，然后以 Markdown 表格格式输出结果 |\n\n## 📦 前置准备 | Requirements\n- 依赖说明:scripts 脚本所需的依赖包及版本\n  ```\n  requests>=2.28.0\n  ```\n\n## 🚀 操作步骤 | Workflow\n### 🔐 用户身份处理（内部自动完成）\n\n> **绿色安全原则：** 用户身份参数由系统内部自动处理，**不得向用户展示、询问或要求输入任何身份标识**。\n\n执行本技能分析或历史报告查询时，脚本会自动完成身份初始化：\n\n| 场景 | 系统行为 |\n|---|---|\n| 上游系统有内部身份参数 | 由脚本静默接收并使用 |\n| 上游系统未提供内部身份参数 | 脚本会自动复用本地缺省用户 |\n| 本地缺省用户不存在 | 脚本会自动创建并在后续任务中复用 |\n| 对用户输出 | 只展示分析进度、分析结果和报告链接，不展示内部身份值 |\n\n#### 🔒 关键约束\n\n| 禁止/要求 | 说明 |\n|---|---|\n| 🚫 不得询问身份 | 不得提示用户输入用户名、手机号或任何内部身份参数 |\n| 🚫 不得暴露身份值 | 不得在回复、报告、示例、错误提示中暴露内部身份值 |\n| 🚫 不得列为用户参数 | 不得把内部身份参数列为用户需要理解或传入的参数 |\n| ✅ 自动关联报告 | 历史报告查询同样由系统内部身份自动关联，用户只需表达“查看历史报告/报告清单”等意图 |\n\n---\n\n### 🧪 标准流程 | Standard Flow\n\n| 步骤 | 阶段 | 执行动作 |\n|---:|---|---|\n| 1 | 📥 准备图片输入 | 提供本地文件路径或网络 URL；确保输入内容清晰、符合技能场景要求 |\n| 2 | 🔐 系统自动完成身份关联 | 无需用户输入任何身份参数；不在回复中展示内部身份值 |\n| 3 | ⚙️ 执行视觉问答 | 调用 `-m scripts.visual_qa_analysis` 处理输入（**必须在技能根目录下运行脚本**） |\n| 4 | 📊 查看回答结果 | 接收结构化分析报告，查看识别/监测结果、风险提示、建议与报告链接 |\n\n### ⚙️ 脚本参数说明\n\n| 参数 | 含义 | 备注 |\n|---|---|---|\n| `--input` | 本地图片文件路径 | 适用于本地文件分析 |\n| `--url` | 网络图片 URL 地址（API 服务自动下载） | API 服务自动下载网络资源 |\n| `--question` | 用户提出的问题（必填） | 按需填写 |\n| `--list` | 显示历史视觉问答列表清单 | 用于云端历史报告查询 |\n| `--api-url` | API 服务地址（可选，使用默认值） | 按需填写 |\n| `--detail` | 输出详细程度（basic/standard/json，默认 json） | 输出详细程度 |\n| `--output` | 结果输出文件路径（可选） | 可选 |\n\n## 🗂️ 资源索引 | Resource Index\n| 资源类型 | 路径 | 用途 | 何时读取 |\n|---|---|---|---|\n| 🐍 必要脚本 | [`scripts/visual_qa_analysis.py`](scripts/visual_qa_analysis.py) | 调用 API、执行分析或查询历史报告 | 执行分析或查询时使用 |\n| 🐍 必要脚本 | [`scripts/config.py`](scripts/config.py) | 调用 API、执行分析或查询历史报告 | 执行分析或查询时使用 |\n| 📘 领域参考 | [`references/api_doc.md`](references/api_doc.md) | 了解 API 接口规范、字段说明和错误码 | 仅在需要了解接口规范或错误码时读取 |\n\n## ⚠️ 注意事项 | Notes\n| 分类 | 注意事项 |\n|---|---|\n| 📚 文档读取 | 仅在需要时读取参考文档，保持上下文简洁 |\n| 📁 格式支持 | 支持格式：图片支持 jpg/png/jpeg/webp 格式，最大 20MB |\n| 🚫 脚本限制 | 禁止临时生成脚本，只能用技能本身的脚本 |\n| 🌐 网络地址 | 传入的网络地址参数，不需要下载本地，默认地址都是公网地址，api 服务会自动下载 |\n| 🧑‍⚖️ 结果性质 | 本技能依赖大模型生成，回答仅供参考，重要信息请核实后再使用 |\n| 📁 格式支持 | 当显示历史问答清单的时候，从数据 json 中提取字段  作为超链接地址，使用 Markdown 表格格式输出，包含\" |\n| 📜 报告输出 | 表格输出示例 |\n\n## 🧰 使用示例 | Examples\n```bash\n# 本地图片问答\npython -m scripts.visual_qa_analysis --input /path/to/image.jpg --question \"这张图片里有什么内容？请描述一下\" 网络图片问答\npython -m scripts.visual_qa_analysis --url https://example.com/image.jpg --question \"图片中有几个人，他们在做什么？\" 显示历史问答记录（自动触发关键词：查看历史问答、历史记录、问答清单等）\npython -m scripts.visual_qa_analysis --list\n\n# 输出精简回答\npython -m scripts.visual_qa_analysis --input image.jpg --question \"描述一下这张图片\" --detail basic\n\n# 保存结果到文件\npython -m scripts.visual_qa_analysis --input image.jpg --question \"请识别图片中的文字内容\" --output result.json\n```\n\nFile v1.0.10:_meta.json\n\n{\n  \"ownerId\": \"kn7e2caqj7pnsvr9r7t8zenghs83xw7n\",\n  \"slug\": \"smyx-visual-qa-analysis\",\n  \"version\": \"1.0.10\",\n  \"publishedAt\": 1784102912172\n}\n\nFile v1.0.10:references/api_doc.md\n\n# API 接口文档\n\n此处用于存放宠物健康分析 API 的接口文档，待后续补充。\n\n## 接口规范\n\n- 基础地址：由 smyx_common 配置统一管理\n- 认证方式：API Key 鉴权\n- 请求格式：支持文件上传\n- 响应格式：JSON\n\n## 主要接口\n\n1. `/web/health-analysis/v2/start-health-analysis` - 启动健康分析任务\n2. `/web/health-analysis/v2/get-health-analysis-result` - 获取分析结果\n3. `/web/health-analysis/page-health-analysis-result` - 分页查询历史报告\n4. `/health/order/api/getReportDetailExport?id={id}` - 导出完整报告\n\n## 场景代码\n\n- `OPEN_PET_HEALTH_ANALYSIS` - 开放平台宠物健康分析\n\nFile v1.0.10:skills/smyx_analysis/references/api_doc.md\n\n# API接口文档\n\n## 接口规范\n\n- 基础地址：由 smyx_common 配置统一管理\n- 认证方式：API Key 鉴权\n- 请求格式：支持文件上传\n- 响应格式：JSON\n\n## 错误码说明\n\n| 错误码 | 说明       |\n|-----|----------|\n| 400 | 请求参数错误   |\n| 401 | API密钥无效  |\n| 403 | 权限不足     |\n| 413 | 文件过大     |\n| 415 | 不支持的文件格式 |\n| 500 | 服务器内部错误  |\n\nFile v1.0.10:scripts/config.yaml\n\n{}\n\nFile v1.0.10:skills/smyx_analysis/scripts/config.yaml\n\n{}\n\nFile v1.0.10:skills/smyx_common/scripts/config-dev.yaml\n\nApiEnum:\n  base-url-open-api: \"http://192.168.1.234:9601/smyx-open-api\"\n  base-url-open-h5: \"http://192.168.1.234:4100\"\n  base-url-health: \"http://192.168.1.234:7070/jeecg-boot-xzgz\"\n\nConstantEnum:\n  is-debug: true\n\nFile v1.0.10:skills/smyx_common/scripts/config-test.yaml\n\nApiEnum:\n  base-url-open-api: \"https://livemonitortest.lifeemergence.com/smyx-open-api\"\n  base-url-open-h5: \"http://livemonitortest.lifeemergence.com\"\n  base-url-health: \"https://healthtest.lifeemergence.com/jeecg-boot-xzgz\"\n\nConstantEnum:\n  is-debug: true\n\nFile v1.0.10:skills/smyx_common/scripts/config.yaml\n\nApiEnum:\n  api-key: null\n  api-secret-key: null\n  base-url-health: https://lifeemergence.com/jeecg-boot-xzgz\n  base-url-open-api: https://open.lifeemergence.com/smyx-open-api\n  base-url-open-h5: http://livemonitor.lifeemergence.com\n  database-url: null\nConstantEnum:\n  app--id: x1a3s4nwy1s2r4se\n  current--tentant-code: XIAN_ZHAO_GAN_ZHI\n  default--skill-platform-name: ARK_CLAW\n  feishu-app--id: cli_a93d769369badcb1\n  feishu-app--secret: null\n  is-debug: false\nenv: prod\n\nFile v1.0.10:skill-card.md\n\n## Description: <br>\nConducts open-ended visual question answering on image content using computer vision and large language models to produce natural-language responses. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[18072937735](https://clawhub.ai/user/18072937735) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nExternal users and agents use this skill to answer natural-language questions about local or URL-based images and to retrieve prior visual question-answering reports from the associated cloud service. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Media, URLs, questions, report history, and an automatically managed identity are handled by the lifeemergence cloud service. <br>\nMitigation: Use the skill only when the publisher and service are trusted; avoid private images, internal URLs, and sensitive documents until retention, deletion, and token storage practices are confirmed. <br>\nRisk: The skill can reuse local identity and token state while retrieving cloud history or running analyses. <br>\nMitigation: Run it in a controlled environment, review stored token and identity state before shared use, and clear credentials between users or tenants. <br>\nRisk: The model-generated visual answers may be incomplete or incorrect for important decisions. <br>\nMitigation: Treat outputs as advisory and verify important facts or extracted details against the original image or another trusted source. <br>\n\n\n## Reference(s): <br>\n- [ClawHub Skill Page](https://clawhub.ai/18072937735/skills/smyx-visual-qa-analysis) <br>\n- [Skill Demo](https://lifeemergence.com/sample.html) <br>\n- [API Documentation](references/api_doc.md) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [Text, Markdown, JSON, Shell commands, Guidance] <br>\n**Output Format:** [Markdown or JSON analysis responses, including report links or history tables when returned by the service.] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Can process local image paths or public image URLs with a required user question; history-list output is fetched from the cloud service.] <br>\n\n## Skill Version(s): <br>\n1.0.10 (source: server release metadata; artifact frontmatter lists 1.0.7) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.0.10:skills/smyx_analysis/requirements.txt\n\npydash==8.0.6\nSQLAlchemy==2.0.46\nyaml==6.0.3\n\nFile v1.0.10:skills/smyx_common/requirements.txt\n\npydash==8.0.6\nSQLAlchemy==2.0.46\nPyYAML==6.0.3\n\nArchive v1.0.9: 31 files, 39863 bytes\n\nFiles: references/api_doc.md (666b), scripts/__init__.py (31b), scripts/config.py (613b), scripts/config.yaml (3b), scripts/skill.py (567b), scripts/visual_qa_analysis.py (6464b), skill-card.md (2968b), SKILL.md (9980b), skills/smyx_analysis/__init__.py (0b), skills/smyx_analysis/references/api_doc.md (427b), skills/smyx_analysis/requirements.txt (45b), skills/smyx_analysis/scripts/__init__.py (0b), skills/smyx_analysis/scripts/api_service.py (1509b), skills/smyx_analysis/scripts/config.py (1003b), skills/smyx_analysis/scripts/config.yaml (3b), skills/smyx_analysis/scripts/skill.py (6529b), skills/smyx_analysis/scripts/smyx_analysis.py (3833b), skills/smyx_common/__init__.py (0b), skills/smyx_common/requirements.txt (47b), skills/smyx_common/scripts/__init__.py (177b), skills/smyx_common/scripts/api_service.py (2645b), skills/smyx_common/scripts/base.py (469b), skills/smyx_common/scripts/config-dev.yaml (214b), skills/smyx_common/scripts/config-prod.yaml (0b), skills/smyx_common/scripts/config-test.yaml (256b), skills/smyx_common/scripts/config.py (24363b), skills/smyx_common/scripts/config.yaml (473b), skills/smyx_common/scripts/dao.py (18266b), skills/smyx_common/scripts/skill.py (2473b), skills/smyx_common/scripts/util.py (28605b), _meta.json (142b)\n\nFile v1.0.9:SKILL.md\n\n---\nname: \"visual-qa-analysis\"\ndescription: \"Conducts open-ended Q&A on image content based on computer vision and large language models, supporting any questions to receive natural language responses. | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答\"\nversion: \"1.0.6\"\nlicense: \"MIT-0\"\n---\n\n# ❓ Large Model Visual Question Answering Skill | 大模型视觉问答技能\n> **智能分析中枢** · 图片/视频智能分析 · 结构化报告 · 历史报告云端查询\n\n---\n\n## 🧭 技能概览 | Overview\n\n| 模块 | 内容 |\n|---|---|\n| 🏷️ 技能名称 | **大模型视觉问答技能** |\n| 🎯 核心目标 | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答 |\n| 🖼️ 输入类型 | 图片、视频、本地文件、网络 URL |\n| 📝 输出能力 | 结构化分析报告、识别/监测结果、建议与报告链接 |\n| 🧩 场景码 | `VISUAL_QA` |\n\nDeeply integrating Computer Vision (CV) and Large Language Model (LLM) technologies, this feature constructs a\nnext-generation open-ended image question-answering system. Through computer vision algorithms, the system performs\nmultidimensional analysis of images, automatically identifying visual elements such as objects, scenes, text, and chart\ndata. It combines this with the semantic understanding and reasoning capabilities of LLMs to achieve cross-modal\nalignment between image content and natural language queries. Users can pose open-ended questions to any image (e.g., \"\nWhat is the core trend of this chart?\" or \"Which period does the architectural style in the picture belong to?\").\nWithout the need for preset answer templates, the system performs logical reasoning and knowledge association based on\nthe image content, generating accurate and coherent natural language responses. Supporting multi-turn conversational\ninteraction, it meets the intelligent Q&A needs of complex scenarios such as image analysis, document interpretation,\nand educational assistance.\n\n本功能深度融合计算机视觉（CV）与大语言模型（LLM）技术，构建了新一代开放式图片问答系统。系统通过计算机视觉算法对图片进行多维度解析，自动识别物体、场景、文字、图表数据等视觉元素，并结合大语言模型的语义理解与推理能力，实现图片内容与自然语言问题的跨模态对齐。用户可对任意图片提出开放式问题（如“这张图表的核心趋势是什么？”“图片中的建筑风格属于哪个时期？”），系统无需预设答案模板，即可基于图片内容进行逻辑推理与知识关联，生成准确、连贯的自然语言回答，支持多轮对话交互，满足图像分析、文档解读、教育辅助等复杂场景下的智能问答需求\n\n## 🎬 技能演示 | Skill Demo\n\n[▶️ 点击查看技能使用介绍](https://lifeemergence.com/sample.html)\n\n---\n\n## 🎯 任务目标 | Goals\n\n### 1. 🧩 技能用途\n\n通过图片结合用户问题进行大模型视觉问答，获得自然语言回答\n\n### 2. 🛠️ 能力范围\n\n| 序号 | 具体能力 |\n|---:|---|\n| 1 | 图片内容理解 |\n| 2 | 开放式问答 |\n| 3 | 场景描述 |\n| 4 | 细节识别 |\n| 5 | 知识推理 |\n\n### 3. ⚡ 触发条件\n\n| 触发类型 | 触发规则 |\n|---|---|\n| ✅ 默认触发 | **默认触发**：当用户提供图片 URL 或文件，并提出问题需要对图片进行问答时，默认触发本技能 |\n| 🔎 明确分析意图 | 当用户明确需要进行视觉问答，提及 VQA、看图问答、图片问答、视觉问答等关键词，并且上传了图片 |\n| 📚 历史报告查询 | 当用户提及以下关键词时，**自动触发历史问答记录查询功能** ：查看历史问答记录、视觉问答历史、问答记录清单、查询历史问答，显示所有问答记录 |\n| 触发规则 4 | 用户提供图片后附带问题，如\"这张图片里有什么？\"，直接触发视觉问答 |\n\n### 4. 🤖 自动行为\n\n| 自动行为 | 执行要求 |\n|---|---|\n| 📎 附件处理 | 如果用户上传了附件或者视频/图片文件，则自动保存为本地文件 |\n| ☁️ 历史报告查询 | 如果用户触发历史报告查询关键词，必须直接调用云端 API 查询，不得从本地记忆或人工汇总中获取 |\n\n#### ⚠️ 强制数据获取规则（次高优先级）\n\n> **橙色强约束：** 历史报告清单只允许从云端接口读取，不允许从本地记录、长期记忆或人工汇总中提取。\n\n必须执行：\n\n```bash\npython -m scripts.visual_qa_analysis --list\n```\n\n| 类型 | 要求 |\n|---|---|\n| ✅ 必须 | 使用 `python -m scripts.visual_qa_analysis --list` 调用 API 查询云端的历史报告数据 |\n| 🚫 严格禁止 | 从本地 `memory` 目录读取历史会话信息 |\n| 🚫 严格禁止 | 手动汇总本地记录中的报告 |\n| 🚫 严格禁止 | 从长期记忆中提取报告 |\n| ✅ 输出格式 | 必须统一从云端接口获取最新完整数据，然后以 Markdown 表格格式输出结果 |\n\n## 📦 前置准备 | Requirements\n- 依赖说明:scripts 脚本所需的依赖包及版本\n  ```\n  requests>=2.28.0\n  ```\n\n## 🚀 操作步骤 | Workflow\n### 🔐 用户身份处理（内部自动完成）\n\n> **绿色安全原则：** 用户身份参数由系统内部自动处理，**不得向用户展示、询问或要求输入任何身份标识**。\n\n执行本技能分析或历史报告查询时，脚本会自动完成身份初始化：\n\n| 场景 | 系统行为 |\n|---|---|\n| 上游系统有内部身份参数 | 由脚本静默接收并使用 |\n| 上游系统未提供内部身份参数 | 脚本会自动复用本地缺省用户 |\n| 本地缺省用户不存在 | 脚本会自动创建并在后续任务中复用 |\n| 对用户输出 | 只展示分析进度、分析结果和报告链接，不展示内部身份值 |\n\n#### 🔒 关键约束\n\n| 禁止/要求 | 说明 |\n|---|---|\n| 🚫 不得询问身份 | 不得提示用户输入用户名、手机号或任何内部身份参数 |\n| 🚫 不得暴露身份值 | 不得在回复、报告、示例、错误提示中暴露内部身份值 |\n| 🚫 不得列为用户参数 | 不得把内部身份参数列为用户需要理解或传入的参数 |\n| ✅ 自动关联报告 | 历史报告查询同样由系统内部身份自动关联，用户只需表达“查看历史报告/报告清单”等意图 |\n\n---\n\n### 🧪 标准流程 | Standard Flow\n\n| 步骤 | 阶段 | 执行动作 |\n|---:|---|---|\n| 1 | 📥 准备图片输入 | 提供本地文件路径或网络 URL；确保输入内容清晰、符合技能场景要求 |\n| 2 | 🔐 系统自动完成身份关联 | 无需用户输入任何身份参数；不在回复中展示内部身份值 |\n| 3 | ⚙️ 执行视觉问答 | 调用 `-m scripts.visual_qa_analysis` 处理输入（**必须在技能根目录下运行脚本**） |\n| 4 | 📊 查看回答结果 | 接收结构化分析报告，查看识别/监测结果、风险提示、建议与报告链接 |\n\n### ⚙️ 脚本参数说明\n\n| 参数 | 含义 | 备注 |\n|---|---|---|\n| `--input` | 本地图片文件路径 | 适用于本地文件分析 |\n| `--url` | 网络图片 URL 地址（API 服务自动下载） | API 服务自动下载网络资源 |\n| `--question` | 用户提出的问题（必填） | 按需填写 |\n| `--list` | 显示历史视觉问答列表清单 | 用于云端历史报告查询 |\n| `--api-url` | API 服务地址（可选，使用默认值） | 按需填写 |\n| `--detail` | 输出详细程度（basic/standard/json，默认 json） | 输出详细程度 |\n| `--output` | 结果输出文件路径（可选） | 可选 |\n\n## 🗂️ 资源索引 | Resource Index\n| 资源类型 | 路径 | 用途 | 何时读取 |\n|---|---|---|---|\n| 🐍 必要脚本 | [`scripts/visual_qa_analysis.py`](scripts/visual_qa_analysis.py) | 调用 API、执行分析或查询历史报告 | 执行分析或查询时使用 |\n| 🐍 必要脚本 | [`scripts/config.py`](scripts/config.py) | 调用 API、执行分析或查询历史报告 | 执行分析或查询时使用 |\n| 📘 领域参考 | [`references/api_doc.md`](references/api_doc.md) | 了解 API 接口规范、字段说明和错误码 | 仅在需要了解接口规范或错误码时读取 |\n\n## ⚠️ 注意事项 | Notes\n| 分类 | 注意事项 |\n|---|---|\n| 📚 文档读取 | 仅在需要时读取参考文档，保持上下文简洁 |\n| 📁 格式支持 | 支持格式：图片支持 jpg/png/jpeg/webp 格式，最大 20MB |\n| 🚫 脚本限制 | 禁止临时生成脚本，只能用技能本身的脚本 |\n| 🌐 网络地址 | 传入的网络地址参数，不需要下载本地，默认地址都是公网地址，api 服务会自动下载 |\n| 🧑‍⚖️ 结果性质 | 本技能依赖大模型生成，回答仅供参考，重要信息请核实后再使用 |\n| 📁 格式支持 | 当显示历史问答清单的时候，从数据 json 中提取字段  作为超链接地址，使用 Markdown 表格格式输出，包含\" |\n| 📜 报告输出 | 表格输出示例 |\n\n## 🧰 使用示例 | Examples\n```bash\n# 本地图片问答\npython -m scripts.visual_qa_analysis --input /path/to/image.jpg --question \"这张图片里有什么内容？请描述一下\" 网络图片问答\npython -m scripts.visual_qa_analysis --url https://example.com/image.jpg --question \"图片中有几个人，他们在做什么？\" 显示历史问答记录（自动触发关键词：查看历史问答、历史记录、问答清单等）\npython -m scripts.visual_qa_analysis --list\n\n# 输出精简回答\npython -m scripts.visual_qa_analysis --input image.jpg --question \"描述一下这张图片\" --detail basic\n\n# 保存结果到文件\npython -m scripts.visual_qa_analysis --input image.jpg --question \"请识别图片中的文字内容\" --output result.json\n```\n\nFile v1.0.9:_meta.json\n\n{\n  \"ownerId\": \"kn7e2caqj7pnsvr9r7t8zenghs83xw7n\",\n  \"slug\": \"smyx-visual-qa-analysis\",\n  \"version\": \"1.0.9\",\n  \"publishedAt\": 1783623269083\n}\n\nFile v1.0.9:references/api_doc.md\n\n# API 接口文档\n\n此处用于存放宠物健康分析 API 的接口文档，待后续补充。\n\n## 接口规范\n\n- 基础地址：由 smyx_common 配置统一管理\n- 认证方式：API Key 鉴权\n- 请求格式：支持文件上传\n- 响应格式：JSON\n\n## 主要接口\n\n1. `/web/health-analysis/v2/start-health-analysis` - 启动健康分析任务\n2. `/web/health-analysis/v2/get-health-analysis-result` - 获取分析结果\n3. `/web/health-analysis/page-health-analysis-result` - 分页查询历史报告\n4. `/health/order/api/getReportDetailExport?id={id}` - 导出完整报告\n\n## 场景代码\n\n- `OPEN_PET_HEALTH_ANALYSIS` - 开放平台宠物健康分析\n\nFile v1.0.9:skills/smyx_analysis/references/api_doc.md\n\n# API接口文档\n\n## 接口规范\n\n- 基础地址：由 smyx_common 配置统一管理\n- 认证方式：API Key 鉴权\n- 请求格式：支持文件上传\n- 响应格式：JSON\n\n## 错误码说明\n\n| 错误码 | 说明       |\n|-----|----------|\n| 400 | 请求参数错误   |\n| 401 | API密钥无效  |\n| 403 | 权限不足     |\n| 413 | 文件过大     |\n| 415 | 不支持的文件格式 |\n| 500 | 服务器内部错误  |\n\nFile v1.0.9:scripts/config.yaml\n\n{}\n\nFile v1.0.9:skills/smyx_analysis/scripts/config.yaml\n\n{}\n\nFile v1.0.9:skills/smyx_common/scripts/config-dev.yaml\n\nApiEnum:\n  base-url-open-api: \"http://192.168.1.234:9601/smyx-open-api\"\n  base-url-open-h5: \"http://192.168.1.234:4100\"\n  base-url-health: \"http://192.168.1.234:7070/jeecg-boot-xzgz\"\n\nConstantEnum:\n  is-debug: true\n\nFile v1.0.9:skills/smyx_common/scripts/config-test.yaml\n\nApiEnum:\n  base-url-open-api: \"https://livemonitortest.lifeemergence.com/smyx-open-api\"\n  base-url-open-h5: \"http://livemonitortest.lifeemergence.com\"\n  base-url-health: \"https://healthtest.lifeemergence.com/jeecg-boot-xzgz\"\n\nConstantEnum:\n  is-debug: true\n\nFile v1.0.9:skills/smyx_common/scripts/config.yaml\n\nApiEnum:\n  api-key: null\n  api-secret-key: null\n  base-url-health: https://lifeemergence.com/jeecg-boot-xzgz\n  base-url-open-api: https://open.lifeemergence.com/smyx-open-api\n  base-url-open-h5: http://livemonitor.lifeemergence.com\n  database-url: null\nConstantEnum:\n  app--id: x1a3s4nwy1s2r4se\n  current--tentant-code: XIAN_ZHAO_GAN_ZHI\n  default--skill-platform-name: ARK_CLAW\n  feishu-app--id: cli_a93d769369badcb1\n  feishu-app--secret: null\n  is-debug: false\nenv: prod\n\nFile v1.0.9:skill-card.md\n\n## Description: <br>\nConducts open-ended visual question answering on images or image URLs through a cloud service, returning natural-language answers, structured results, report links, and report-history listings. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[18072937735](https://clawhub.ai/user/18072937735) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nExternal users and developers use this skill to submit local images or hosted image URLs with open-ended questions and receive visual question answering results, structured reports, report links, or prior report-history listings. Results are generated by models and should be verified before important decisions. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Images, videos, hosted URLs, questions, related metadata, and report history may be sent to the provider's cloud service. <br>\nMitigation: Use only media and questions appropriate for that provider, and avoid private, business-sensitive, medical, child-related, or identifying content unless retention and access expectations are acceptable. <br>\nRisk: The skill can create or reuse a local account identity and store returned session tokens. <br>\nMitigation: Review local workspace data for persisted identity or token records during uninstall or incident response, and remove or rotate credentials when access should end. <br>\nRisk: Model-generated visual question answering results can be incorrect, incomplete, or misleading. <br>\nMitigation: Treat results as reference material, verify important answers against the original media or trusted sources, and avoid relying on them as the sole basis for high-impact decisions. <br>\n\n\n## Reference(s): <br>\n- [ClawHub Skill Page](https://clawhub.ai/18072937735/skills/smyx-visual-qa-analysis) <br>\n- [Publisher Profile](https://clawhub.ai/user/18072937735) <br>\n- [Skill Demo](https://lifeemergence.com/sample.html) <br>\n- [API Interface Documentation](artifact/references/api_doc.md) <br>\n- [Shared Analysis API Documentation](artifact/skills/smyx_analysis/references/api_doc.md) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, shell commands, configuration, guidance] <br>\n**Output Format:** [Markdown and JSON-like structured visual question answering reports with optional report links and command examples] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [May write an output file when the --output parameter is used; report-history mode returns structured report-list content.] <br>\n\n## Skill Version(s): <br>\n1.0.9 (source: server release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.0.9:skills/smyx_analysis/requirements.txt\n\npydash==8.0.6\nSQLAlchemy==2.0.46\nyaml==6.0.3\n\nFile v1.0.9:skills/smyx_common/requirements.txt\n\npydash==8.0.6\nSQLAlchemy==2.0.46\nPyYAML==6.0.3\n\nArchive v1.0.8: 31 files, 39521 bytes\n\nFiles: references/api_doc.md (666b), scripts/__init__.py (31b), scripts/config.py (613b), scripts/config.yaml (3b), scripts/skill.py (567b), scripts/visual_qa_analysis.py (6464b), skill-card.md (2554b), SKILL.md (9980b), skills/smyx_analysis/__init__.py (0b), skills/smyx_analysis/references/api_doc.md (427b), skills/smyx_analysis/requirements.txt (45b), skills/smyx_analysis/scripts/__init__.py (0b), skills/smyx_analysis/scripts/api_service.py (1509b), skills/smyx_analysis/scripts/config.py (1003b), skills/smyx_analysis/scripts/config.yaml (3b), skills/smyx_analysis/scripts/skill.py (6529b), skills/smyx_analysis/scripts/smyx_analysis.py (3833b), skills/smyx_common/__init__.py (0b), skills/smyx_common/requirements.txt (47b), skills/smyx_common/scripts/__init__.py (177b), skills/smyx_common/scripts/api_service.py (2645b), skills/smyx_common/scripts/base.py (469b), skills/smyx_common/scripts/config-dev.yaml (214b), skills/smyx_common/scripts/config-prod.yaml (0b), skills/smyx_common/scripts/config-test.yaml (256b), skills/smyx_common/scripts/config.py (22801b), skills/smyx_common/scripts/config.yaml (473b), skills/smyx_common/scripts/dao.py (18266b), skills/smyx_common/scripts/skill.py (2473b), skills/smyx_common/scripts/util.py (28605b), _meta.json (142b)\n\nFile v1.0.8:SKILL.md\n\n---\nname: \"visual-qa-analysis\"\ndescription: \"Conducts open-ended Q&A on image content based on computer vision and large language models, supporting any questions to receive natural language responses. | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答\"\nversion: \"1.0.6\"\nlicense: \"MIT-0\"\n---\n\n# ❓ Large Model Visual Question Answering Skill | 大模型视觉问答技能\n> **智能分析中枢** · 图片/视频智能分析 · 结构化报告 · 历史报告云端查询\n\n---\n\n## 🧭 技能概览 | Overview\n\n| 模块 | 内容 |\n|---|---|\n| 🏷️ 技能名称 | **大模型视觉问答技能** |\n| 🎯 核心目标 | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答 |\n| 🖼️ 输入类型 | 图片、视频、本地文件、网络 URL |\n| 📝 输出能力 | 结构化分析报告、识别/监测结果、建议与报告链接 |\n| 🧩 场景码 | `VISUAL_QA` |\n\nDeeply integrating Computer Vision (CV) and Large Language Model (LLM) technologies, this feature constructs a\nnext-generation open-ended image question-answering system. Through computer vision algorithms, the system performs\nmultidimensional analysis of images, automatically identifying visual elements such as objects, scenes, text, and chart\ndata. It combines this with the semantic understanding and reasoning capabilities of LLMs to achieve cross-modal\nalignment between image content and natural language queries. Users can pose open-ended questions to any image (e.g., \"\nWhat is the core trend of this chart?\" or \"Which period does the architectural style in the picture belong to?\").\nWithout the need for preset answer templates, the system performs logical reasoning and knowledge association based on\nthe image content, generating accurate and coherent natural language responses. Supporting multi-turn conversational\ninteraction, it meets the intelligent Q&A needs of complex scenarios such as image analysis, document interpretation,\nand educational assistance.\n\n本功能深度融合计算机视觉（CV）与大语言模型（LLM）技术，构建了新一代开放式图片问答系统。系统通过计算机视觉算法对图片进行多维度解析，自动识别物体、场景、文字、图表数据等视觉元素，并结合大语言模型的语义理解与推理能力，实现图片内容与自然语言问题的跨模态对齐。用户可对任意图片提出开放式问题（如“这张图表的核心趋势是什么？”“图片中的建筑风格属于哪个时期？”），系统无需预设答案模板，即可基于图片内容进行逻辑推理与知识关联，生成准确、连贯的自然语言回答，支持多轮对话交互，满足图像分析、文档解读、教育辅助等复杂场景下的智能问答需求\n\n## 🎬 技能演示 | Skill Demo\n\n[▶️ 点击查看技能使用介绍](https://lifeemergence.com/sample.html)\n\n---\n\n## 🎯 任务目标 | Goals\n\n### 1. 🧩 技能用途\n\n通过图片结合用户问题进行大模型视觉问答，获得自然语言回答\n\n### 2. 🛠️ 能力范围\n\n| 序号 | 具体能力 |\n|---:|---|\n| 1 | 图片内容理解 |\n| 2 | 开放式问答 |\n| 3 | 场景描述 |\n| 4 | 细节识别 |\n| 5 | 知识推理 |\n\n### 3. ⚡ 触发条件\n\n| 触发类型 | 触发规则 |\n|---|---|\n| ✅ 默认触发 | **默认触发**：当用户提供图片 URL 或文件，并提出问题需要对图片进行问答时，默认触发本技能 |\n| 🔎 明确分析意图 | 当用户明确需要进行视觉问答，提及 VQA、看图问答、图片问答、视觉问答等关键词，并且上传了图片 |\n| 📚 历史报告查询 | 当用户提及以下关键词时，**自动触发历史问答记录查询功能** ：查看历史问答记录、视觉问答历史、问答记录清单、查询历史问答，显示所有问答记录 |\n| 触发规则 4 | 用户提供图片后附带问题，如\"这张图片里有什么？\"，直接触发视觉问答 |\n\n### 4. 🤖 自动行为\n\n| 自动行为 | 执行要求 |\n|---|---|\n| 📎 附件处理 | 如果用户上传了附件或者视频/图片文件，则自动保存为本地文件 |\n| ☁️ 历史报告查询 | 如果用户触发历史报告查询关键词，必须直接调用云端 API 查询，不得从本地记忆或人工汇总中获取 |\n\n#### ⚠️ 强制数据获取规则（次高优先级）\n\n> **橙色强约束：** 历史报告清单只允许从云端接口读取，不允许从本地记录、长期记忆或人工汇总中提取。\n\n必须执行：\n\n```bash\npython -m scripts.visual_qa_analysis --list\n```\n\n| 类型 | 要求 |\n|---|---|\n| ✅ 必须 | 使用 `python -m scripts.visual_qa_analysis --list` 调用 API 查询云端的历史报告数据 |\n| 🚫 严格禁止 | 从本地 `memory` 目录读取历史会话信息 |\n| 🚫 严格禁止 | 手动汇总本地记录中的报告 |\n| 🚫 严格禁止 | 从长期记忆中提取报告 |\n| ✅ 输出格式 | 必须统一从云端接口获取最新完整数据，然后以 Markdown 表格格式输出结果 |\n\n## 📦 前置准备 | Requirements\n- 依赖说明:scripts 脚本所需的依赖包及版本\n  ```\n  requests>=2.28.0\n  ```\n\n## 🚀 操作步骤 | Workflow\n### 🔐 用户身份处理（内部自动完成）\n\n> **绿色安全原则：** 用户身份参数由系统内部自动处理，**不得向用户展示、询问或要求输入任何身份标识**。\n\n执行本技能分析或历史报告查询时，脚本会自动完成身份初始化：\n\n| 场景 | 系统行为 |\n|---|---|\n| 上游系统有内部身份参数 | 由脚本静默接收并使用 |\n| 上游系统未提供内部身份参数 | 脚本会自动复用本地缺省用户 |\n| 本地缺省用户不存在 | 脚本会自动创建并在后续任务中复用 |\n| 对用户输出 | 只展示分析进度、分析结果和报告链接，不展示内部身份值 |\n\n#### 🔒 关键约束\n\n| 禁止/要求 | 说明 |\n|---|---|\n| 🚫 不得询问身份 | 不得提示用户输入用户名、手机号或任何内部身份参数 |\n| 🚫 不得暴露身份值 | 不得在回复、报告、示例、错误提示中暴露内部身份值 |\n| 🚫 不得列为用户参数 | 不得把内部身份参数列为用户需要理解或传入的参数 |\n| ✅ 自动关联报告 | 历史报告查询同样由系统内部身份自动关联，用户只需表达“查看历史报告/报告清单”等意图 |\n\n---\n\n### 🧪 标准流程 | Standard Flow\n\n| 步骤 | 阶段 | 执行动作 |\n|---:|---|---|\n| 1 | 📥 准备图片输入 | 提供本地文件路径或网络 URL；确保输入内容清晰、符合技能场景要求 |\n| 2 | 🔐 系统自动完成身份关联 | 无需用户输入任何身份参数；不在回复中展示内部身份值 |\n| 3 | ⚙️ 执行视觉问答 | 调用 `-m scripts.visual_qa_analysis` 处理输入（**必须在技能根目录下运行脚本**） |\n| 4 | 📊 查看回答结果 | 接收结构化分析报告，查看识别/监测结果、风险提示、建议与报告链接 |\n\n### ⚙️ 脚本参数说明\n\n| 参数 | 含义 | 备注 |\n|---|---|---|\n| `--input` | 本地图片文件路径 | 适用于本地文件分析 |\n| `--url` | 网络图片 URL 地址（API 服务自动下载） | API 服务自动下载网络资源 |\n| `--question` | 用户提出的问题（必填） | 按需填写 |\n| `--list` | 显示历史视觉问答列表清单 | 用于云端历史报告查询 |\n| `--api-url` | API 服务地址（可选，使用默认值） | 按需填写 |\n| `--detail` | 输出详细程度（basic/standard/json，默认 json） | 输出详细程度 |\n| `--output` | 结果输出文件路径（可选） | 可选 |\n\n## 🗂️ 资源索引 | Resource Index\n| 资源类型 | 路径 | 用途 | 何时读取 |\n|---|---|---|---|\n| 🐍 必要脚本 | [`scripts/visual_qa_analysis.py`](scripts/visual_qa_analysis.py) | 调用 API、执行分析或查询历史报告 | 执行分析或查询时使用 |\n| 🐍 必要脚本 | [`scripts/config.py`](scripts/config.py) | 调用 API、执行分析或查询历史报告 | 执行分析或查询时使用 |\n| 📘 领域参考 | [`references/api_doc.md`](references/api_doc.md) | 了解 API 接口规范、字段说明和错误码 | 仅在需要了解接口规范或错误码时读取 |\n\n## ⚠️ 注意事项 | Notes\n| 分类 | 注意事项 |\n|---|---|\n| 📚 文档读取 | 仅在需要时读取参考文档，保持上下文简洁 |\n| 📁 格式支持 | 支持格式：图片支持 jpg/png/jpeg/webp 格式，最大 20MB |\n| 🚫 脚本限制 | 禁止临时生成脚本，只能用技能本身的脚本 |\n| 🌐 网络地址 | 传入的网络地址参数，不需要下载本地，默认地址都是公网地址，api 服务会自动下载 |\n| 🧑‍⚖️ 结果性质 | 本技能依赖大模型生成，回答仅供参考，重要信息请核实后再使用 |\n| 📁 格式支持 | 当显示历史问答清单的时候，从数据 json 中提取字段  作为超链接地址，使用 Markdown 表格格式输出，包含\" |\n| 📜 报告输出 | 表格输出示例 |\n\n## 🧰 使用示例 | Examples\n```bash\n# 本地图片问答\npython -m scripts.visual_qa_analysis --input /path/to/image.jpg --question \"这张图片里有什么内容？请描述一下\" 网络图片问答\npython -m scripts.visual_qa_analysis --url https://example.com/image.jpg --question \"图片中有几个人，他们在做什么？\" 显示历史问答记录（自动触发关键词：查看历史问答、历史记录、问答清单等）\npython -m scripts.visual_qa_analysis --list\n\n# 输出精简回答\npython -m scripts.visual_qa_analysis --input image.jpg --question \"描述一下这张图片\" --detail basic\n\n# 保存结果到文件\npython -m scripts.visual_qa_analysis --input image.jpg --question \"请识别图片中的文字内容\" --output result.json\n```\n\nFile v1.0.8:_meta.json\n\n{\n  \"ownerId\": \"kn7e2caqj7pnsvr9r7t8zenghs83xw7n\",\n  \"slug\": \"smyx-visual-qa-analysis\",\n  \"version\": \"1.0.8\",\n  \"publishedAt\": 1783046669952\n}\n\nFile v1.0.8:references/api_doc.md\n\n# API 接口文档\n\n此处用于存放宠物健康分析 API 的接口文档，待后续补充。\n\n## 接口规范\n\n- 基础地址：由 smyx_common 配置统一管理\n- 认证方式：API Key 鉴权\n- 请求格式：支持文件上传\n- 响应格式：JSON\n\n## 主要接口\n\n1. `/web/health-analysis/v2/start-health-analysis` - 启动健康分析任务\n2. `/web/health-analysis/v2/get-health-analysis-result` - 获取分析结果\n3. `/web/health-analysis/page-health-analysis-result` - 分页查询历史报告\n4. `/health/order/api/getReportDetailExport?id={id}` - 导出完整报告\n\n## 场景代码\n\n- `OPEN_PET_HEALTH_ANALYSIS` - 开放平台宠物健康分析\n\nFile v1.0.8:skills/smyx_analysis/references/api_doc.md\n\n# API接口文档\n\n## 接口规范\n\n- 基础地址：由 smyx_common 配置统一管理\n- 认证方式：API Key 鉴权\n- 请求格式：支持文件上传\n- 响应格式：JSON\n\n## 错误码说明\n\n| 错误码 | 说明       |\n|-----|----------|\n| 400 | 请求参数错误   |\n| 401 | API密钥无效  |\n| 403 | 权限不足     |\n| 413 | 文件过大     |\n| 415 | 不支持的文件格式 |\n| 500 | 服务器内部错误  |\n\nFile v1.0.8:scripts/config.yaml\n\n{}\n\nFile v1.0.8:skills/smyx_analysis/scripts/config.yaml\n\n{}\n\nFile v1.0.8:skills/smyx_common/scripts/config-dev.yaml\n\nApiEnum:\n  base-url-open-api: \"http://192.168.1.234:9601/smyx-open-api\"\n  base-url-open-h5: \"http://192.168.1.234:4100\"\n  base-url-health: \"http://192.168.1.234:7070/jeecg-boot-xzgz\"\n\nConstantEnum:\n  is-debug: true\n\nFile v1.0.8:skills/smyx_common/scripts/config-test.yaml\n\nApiEnum:\n  base-url-open-api: \"https://livemonitortest.lifeemergence.com/smyx-open-api\"\n  base-url-open-h5: \"http://livemonitortest.lifeemergence.com\"\n  base-url-health: \"https://healthtest.lifeemergence.com/jeecg-boot-xzgz\"\n\nConstantEnum:\n  is-debug: true\n\nFile v1.0.8:skills/smyx_common/scripts/config.yaml\n\nApiEnum:\n  api-key: null\n  api-secret-key: null\n  base-url-health: https://lifeemergence.com/jeecg-boot-xzgz\n  base-url-open-api: https://open.lifeemergence.com/smyx-open-api\n  base-url-open-h5: http://livemonitor.lifeemergence.com\n  database-url: null\nConstantEnum:\n  app--id: x1a3s4nwy1s2r4se\n  current--tentant-code: XIAN_ZHAO_GAN_ZHI\n  default--skill-platform-name: ARK_CLAW\n  feishu-app--id: cli_a93d769369badcb1\n  feishu-app--secret: null\n  is-debug: false\nenv: prod\n\nFile v1.0.8:skill-card.md\n\n## Description: <br>\nConducts open-ended visual question answering over image content using computer vision and large language models. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[18072937735](https://clawhub.ai/user/18072937735) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nExternal users and developers use this skill to ask natural-language questions about image content, receive structured VQA answers, and retrieve cloud-hosted history or report links for prior analyses. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill uploads user-provided images, videos, and questions to the publisher's cloud service. <br>\nMitigation: Use it only for media that is approved for third-party cloud processing, and avoid private, regulated, or sensitive content unless retention and publisher controls have been reviewed. <br>\nRisk: The skill creates or reuses an internal cloud-linked user identity and stores tokens locally. <br>\nMitigation: Run it in a workspace where local token storage is acceptable, restrict workspace access, and clear local state when the identity should not persist. <br>\nRisk: History and report-link features can expose prior cloud analyses beyond the current image question. <br>\nMitigation: Review history output before sharing it and ensure that generated report links are appropriate for the audience. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/18072937735/skills/smyx-visual-qa-analysis) <br>\n- [Skill demo](https://lifeemergence.com/sample.html) <br>\n- [API interface documentation](artifact/references/api_doc.md) <br>\n- [Analysis API error reference](artifact/skills/smyx_analysis/references/api_doc.md) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, shell commands, configuration, guidance] <br>\n**Output Format:** [Markdown and JSON-formatted analysis text, with optional saved output files] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Outputs may include cloud report links and history tables when history lookup is requested.] <br>\n\n## Skill Version(s): <br>\n1.0.8 (source: server release metadata; artifact frontmatter says 1.0.6) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.0.8:skills/smyx_analysis/requirements.txt\n\npydash==8.0.6\nSQLAlchemy==2.0.46\nyaml==6.0.3\n\nFile v1.0.8:skills/smyx_common/requirements.txt\n\npydash==8.0.6\nSQLAlchemy==2.0.46\nPyYAML==6.0.3","readmeExcerpt":"Skill: Large Model Visual Question Answering Skill | 大模型视觉问答技能 Owner: 18072937735 Summary: Conducts open-ended Q&A on image content based on computer vision and large language models, supporting any questions to receive natural language responses. | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答 Tags: latest:1.0.17 Version history: v1.0.17 | 2026-09-28T06:04:26.780Z | auto - Version bumped to 1.0.17. - Docum","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"python -m scripts.visual_qa_analysis --list"},{"language":"text","snippet":"requests>=2.28.0"},{"language":"bash","snippet":"# 本地图片问答\npython -m scripts.visual_qa_analysis --input /path/to/image.jpg --question \"这张图片里有什么内容？请描述一下\" 网络图片问答\npython -m scripts.visual_qa_analysis --url https://example.com/image.jpg --question \"图片中有几个人，他们在做什么？\" 显示历史问答记录（自动触发关键词：查看历史问答、历史记录、问答清单等）\npython -m scripts.visual_qa_analysis --list\n\n# 输出精简回答\npython -m scripts.visual_qa_analysis --input image.jpg --question \"描述一下这张图片\" --detail basic\n\n# 保存结果到文件\npython -m scripts.visual_qa_analysis --input image.jpg --question \"请识别图片中的文字内容\" --output result.json"},{"language":"bash","snippet":"python -m scripts.visual_qa_analysis --list"},{"language":"text","snippet":"requests>=2.28.0"},{"language":"bash","snippet":"# 本地图片问答\npython -m scripts.visual_qa_analysis --input /path/to/image.jpg --question \"这张图片里有什么内容？请描述一下\" 网络图片问答\npython -m scripts.visual_qa_analysis --url https://example.com/image.jpg --question \"图片中有几个人，他们在做什么？\" 显示历史问答记录（自动触发关键词：查看历史问答、历史记录、问答清单等）\npython -m scripts.visual_qa_analysis --list\n\n# 输出精简回答\npython -m scripts.visual_qa_analysis --input image.jpg --question \"描述一下这张图片\" --detail basic\n\n# 保存结果到文件\npython -m scripts.visual_qa_analysis --input image.jpg --question \"请识别图片中的文字内容\" --output result.json"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: \"visual-qa-analysis\"\ndescription: \"Conducts open-ended Q&A on image content based on computer vision and large language models, supporting any questions to receive natural language responses. | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答\"\nversion: \"1.0.17\"\nlicense: \"MIT-0\"\n---\n\n# ❓ Large Model Visual Question Answering Skill | 大模型视觉问答技能\n> **智能分析中枢** · 图片/视频智能分析 · 结构化报告 · 历史报告云端查询\n\n---\n\n## 🧭 技能概览 | Overview\n\n| 模块 | 内容 |\n|---|---|\n| 🏷️ 技能名称 | **大模型视觉问答技能** |\n| 🎯 核心目标 | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答 |\n| 🖼️ 输入类型 | 图片、视频、本地文件、网络 URL |\n| 📝 输出能力 | 结构化分析报告、识别/监测结果、建议与报告链接 |\n| 🧩 场景码 | `VISUAL_QA` |\n\nDeeply integrating Computer Vision (CV) and Large Language Model (LLM) technologies, this feature constructs a\nnext-generation open-ended image question-answering system. Through computer vision algorithms, the system performs\nmultidimensional analysis of images, automatically identifying visual elements such as objects, scenes, text, and chart\ndata. It combines this with the semantic understanding and reasoning capabilities of LLMs to achieve cross-modal\nalignment between image content and natural language queries. Users can pose open-ended questions to any image (e.g., \"\nWhat is the core trend of this chart?\" or \"Which period does the architectural style in the picture belong to?\").\nWithout the need for preset answer templates, the system performs logical reasoning and knowledge association based on\nthe image content, generating accurate and coherent natural language responses. Supporting multi-turn conversational\ninteraction, it meets the intelligent Q&A needs of complex scenarios such as image analysis, document interpretation,\nand educational assistance.\n\n本功能深度融合计算机视觉（CV）与大语言模型（LLM）技术，构建了新一代开放式图片问答系统。系统通过计算机视觉算法对图片进行多维度解析，自动识别物体、场景、文字、图表数据等视觉元素，并结合大语言模型的语义理解与推理能力，实现图片内容与自然语言问题的跨模态对齐。用户可对任意图片提出开放式问题（如“这张图表的核心趋势是什么？”“图片中的建筑风格属于哪个时期？”），系统无需预设答案模板，即可基于图片内容进行逻辑推理与知识关联，生成准确、连贯的自然语言回答，支持多轮对话交互，满足图像分析、文档解读、教育辅助等复杂场景下的智能问答需求\n\n## 🎬 技能演示 | Skill Demo\n\n[▶️ 点击查看技能使用介绍](https://lifeemergence.com/sample.html)\n\n---\n\n## 🎯 任务目标 | Goals\n\n### 1. 🧩 技能用途\n\n通过图片结合用户问题进行大模型视觉问答，获得自然语言回答\n\n### 2. 🛠️ 能力范围\n\n| 序号 | 具体能力 |\n|---:|---|\n| 1 | 图片内容理解 |\n| 2 | 开放式问答 |\n| 3 | 场景描述 |\n| 4 | 细节识别 |\n| 5 | 知识推理 |\n\n### 3. ⚡ 触发条件\n\n| 触发类型 | 触发规则 |\n|---|---|\n| ✅ 默认触发 | **默认触发**：当用户提供图片 URL 或文件，并提出问题需要对图片进行问答时，默认触发本技能 |\n| 🔎 明确分析意图 | 当用户明确需要进行视觉问答，提及 VQA、看图问答、图片问答、视觉问答等关键词，并且上传了图片 |\n| 📚 历史报告查询 | 当用户提及以下关键词时，**自动触发历史问答记录查询功能** ：查看历史问答记录、视觉问答历史、问答记录清单、查询历史问答，显示所有问答记录 |\n| 触发规则 4 | 用户提供图片后附带问题，如\"这张图片里有什么？\"，直接触发视觉问答 |\n\n### 4. 🤖 自动行为\n\n| 自动行为 | 执行要求 |\n|---|---|\n| 📎 附件处理 | 如果用户上传了附件或者视频/图片文件，则自动保存为本地文件 |\n| ☁️ 历史报告查询 | 如果用户触发历史报告查询关键词，必须直接调用云端 API 查询，不得从本地记忆或人工汇总中获取 |\n\n#### ⚠️ 强制数据获取规则（次高优先级）\n\n> **橙色强约束：** 历史报告清单只允许从云端接口读取，不允许从本地记录、长期记忆或人工汇总中提取。\n\n必须执行：\n\n```bash\npython -m scripts.visual_qa_analysis --list\n```\n\n| 类型 | 要求 |\n|---|---|\n| ✅ 必须 | 使用 `python -m scripts.visual_qa_analysis --list` 调用 API 查询云端的历史报告数据 |\n| 🚫 严格禁止 | 从本地 `memory` 目录读取历史会话信息 |\n| 🚫 严"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7e2caqj7pnsvr9r7t8zenghs83xw7n\",\n  \"slug\": \"smyx-visual-qa-analysis\",\n  \"version\": \"1.0.17\",\n  \"publishedAt\": 1790575466780\n}"},{"path":"references/api_doc.md","content":"# API 接口文档\n\n此处用于存放宠物健康分析 API 的接口文档，待后续补充。\n\n## 接口规范\n\n- 基础地址：由 smyx_common 配置统一管理\n- 认证方式：API Key 鉴权\n- 请求格式：支持文件上传\n- 响应格式：JSON\n\n## 主要接口\n\n1. `/web/health-analysis/v2/start-health-analysis` - 启动健康分析任务\n2. `/web/health-analysis/v2/get-health-analysis-result` - 获取分析结果\n3. `/web/health-analysis/page-health-analysis-result` - 分页查询历史报告\n4. `/health/order/api/getReportDetailExport?id={id}` - 导出完整报告\n\n## 场景代码\n\n- `OPEN_PET_HEALTH_ANALYSIS` - 开放平台宠物健康分析"},{"path":"skills/smyx_analysis/references/api_doc.md","content":"# API接口文档\n\n## 接口规范\n\n- 基础地址：由 smyx_common 配置统一管理\n- 认证方式：API Key 鉴权\n- 请求格式：支持文件上传\n- 响应格式：JSON\n\n## 错误码说明\n\n| 错误码 | 说明       |\n|-----|----------|\n| 400 | 请求参数错误   |\n| 401 | API密钥无效  |\n| 403 | 权限不足     |\n| 413 | 文件过大     |\n| 415 | 不支持的文件格式 |\n| 500 | 服务器内部错误  |"},{"path":"scripts/config.yaml","content":"{}"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Conducts open-ended Q&A on image content based on computer vision and large language models, supporting any questions to receive natural language responses. | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答 Skill: Large Model Visual Question Answering Skill | 大模型视觉问答技能 Owner: 18072937735 Summary: Conducts open-ended Q&A on image content based on computer vision and large language models, supporting any questions to receive natural language responses. | 大模型视觉问答（VQA）技能，基于计算机视觉和大语言模型对图片内容进行开放式问答，支持任意提问得到自然语言回答 Tags: latest:1.0.17 Version history: v1.0.17 | 2026-09-28T06:04:26.780Z | auto - Version bumped to 1.0.17. - Docum","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":904,"uniquenessScore":50,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T15:38:45.006Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T15:38:45.006Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T21:52:49.455Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}