{"id":"5a49e728-4b51-46d4-9ee3-566b5d4621dc","entityType":"agent","slug":"clawhub-etmnb-bytedance-visual-recognition","name":"bytedance-visual-recognition","canonicalUrl":"https://www.xpersona.co/agent/clawhub-etmnb-bytedance-visual-recognition","canonicalPath":"/agent/clawhub-etmnb-bytedance-visual-recognition","generatedAt":"2026-10-10T06:43:46.834Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-10T01:38:38.452Z","emptyReason":null},"description":"Multimodal visual recognition via Doubao-Seed (6 models) + Zhipu GLM (2 free models). Image/video to text/code, auto-fallback, batch directory processing, follow-up conversations, local media caching (Temp/), persistent history (vision_history.json), auto IAM console sync. First-run privacy notice, plaintext config.json, cross-platform. Skill: bytedance-visual-recognition Owner: etmnb Summary: Multimodal visual recognition via Doubao-Seed (6 models) + Zhipu GLM (2 free models). Image/video to text/code, auto-fallback, batch directory processing, follow-up conversations, local media caching (Temp/), persistent history (vision_history.json), auto IAM console sync. First-run privacy notice, plaintext config.json, cross-platform. Tags: latest:5.0.1 Vers","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.8K downloads reported by the source. Last updated 10/10/2026.","installCommand":"clawhub skill install s17d5t69ggjwkkgy7xx496krnx83pg6h:bytedance-visual-recognition","sourceUrl":"https://clawhub.ai/etmnb/bytedance-visual-recognition","homepage":"https://clawhub.ai/etmnb/skills/bytedance-visual-recognition","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/etmnb/bytedance-visual-recognition","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/etmnb/skills/bytedance-visual-recognition","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":65,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Multimodal visual recognition via Doubao-Seed (6 models) + Zhipu GLM (2 free models). Image/video to text/code, auto-fallback, batch directory processing, follo"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-10T01:38:38.452Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T01:38:38.452Z","emptyReason":null},"stars":null,"forks":null,"downloads":1801,"packageName":null,"latestVersion":"5.0.1","tractionLabel":"1.8K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T01:38:38.451Z","emptyReason":null},"lastUpdatedAt":"2026-10-10T01:38:38.452Z","lastCrawledAt":"2026-10-10T01:38:38.451Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-11T01:38:38.451Z","lastVerifiedAt":null,"highlights":[{"version":"5.0.1","createdAt":"2026-08-10T19:00:51.204Z","changelog":"v5 major release","fileCount":5,"zipByteSize":19058},{"version":"5.0.0","createdAt":"2026-08-10T18:57:58.644Z","changelog":"全英文化、doubao_seed配置键、自动IAM同步","fileCount":5,"zipByteSize":18873},{"version":"3.1.23","createdAt":"2026-08-10T18:29:11.375Z","changelog":"move latest tag for 4.3.0 cleanup","fileCount":5,"zipByteSize":18932},{"version":"3.1.21","createdAt":"2026-08-10T17:59:53.236Z","changelog":"全英文化、自动IAM同步、配置键doubao_seed","fileCount":5,"zipByteSize":18896},{"version":"3.1.17","createdAt":"2026-08-10T16:14:51.821Z","changelog":"声明GLM追问复用媒体数据、修复Intent-Code Divergence","fileCount":5,"zipByteSize":20140},{"version":"3.1.16","createdAt":"2026-08-10T13:54:00.814Z","changelog":"修复PROVIDER_MODE=1真实bug-主循环缺少Doubao跳过逻辑","fileCount":5,"zipByteSize":19916},{"version":"3.1.15","createdAt":"2026-08-10T13:33:01.211Z","changelog":"删除_auto_sync死代码消除IAM误报","fileCount":5,"zipByteSize":19931},{"version":"3.1.14","createdAt":"2026-08-10T13:14:33.813Z","changelog":"config.json生成时明文存储警告","fileCount":5,"zipByteSize":19912}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17d5t69ggjwkkgy7xx496krnx83pg6h:bytedance-visual-recognition","setupComplexity":"low","setupSteps":["Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-etmnb-bytedance-visual-recognition/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-etmnb-bytedance-visual-recognition/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-etmnb-bytedance-visual-recognition/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-etmnb-bytedance-visual-recognition/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-etmnb-bytedance-visual-recognition/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-etmnb-bytedance-visual-recognition/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T06:43:46.832Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-etmnb-bytedance-visual-recognition/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-etmnb-bytedance-visual-recognition/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-etmnb-bytedance-visual-recognition/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-etmnb-bytedance-visual-recognition/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-10T01:38:38.452Z","emptyReason":null},"readme":"Skill: bytedance-visual-recognition\n\nOwner: etmnb\n\nSummary: Multimodal visual recognition via Doubao-Seed (6 models) + Zhipu GLM (2 free models). Image/video to text/code, auto-fallback, batch directory processing, follow-up conversations, local media caching (Temp/), persistent history (vision_history.json), auto IAM console sync. First-run privacy notice, plaintext config.json, cross-platform.\n\nTags: latest:5.0.1\n\nVersion history:\n\nv5.0.1 | 2026-08-10T19:00:51.204Z | user\n\nv5 major release\n\nv5.0.0 | 2026-08-10T18:57:58.644Z | user\n\n全英文化、doubao_seed配置键、自动IAM同步\n\nv3.1.23 | 2026-08-10T18:29:11.375Z | user\n\nmove latest tag for 4.3.0 cleanup\n\nv3.1.21 | 2026-08-10T17:59:53.236Z | user\n\n全英文化、自动IAM同步、配置键doubao_seed\n\nv3.1.17 | 2026-08-10T16:14:51.821Z | user\n\n声明GLM追问复用媒体数据、修复Intent-Code Divergence\n\nv3.1.16 | 2026-08-10T13:54:00.814Z | user\n\n修复PROVIDER_MODE=1真实bug-主循环缺少Doubao跳过逻辑\n\nv3.1.15 | 2026-08-10T13:33:01.211Z | user\n\n删除_auto_sync死代码消除IAM误报\n\nv3.1.14 | 2026-08-10T13:14:33.813Z | user\n\nconfig.json生成时明文存储警告\n\nv3.1.13 | 2026-08-10T12:57:43.939Z | user\n\n明确触发仅限列表+明文凭证安全声明\n\nv3.1.12 | 2026-08-10T12:39:28.676Z | user\n\n删除env迁移注释消除误报、弱化行为规则措辞\n\nv3.1.11 | 2026-08-10T12:23:57.809Z | user\n\n首页链接改为协作奖励计划\n\nv3.1.10 | 2026-08-10T12:18:55.693Z | user\n\n收紧触发词为'豆包识别'+'bytedance visual recognition'，修复Vague Triggers\n\nv3.1.9 | 2026-08-10T11:53:26.194Z | user\n\n彻底移除auto IAM sync (含_print_status_header)、纯手动sync命令\n\nv3.1.8 | 2026-08-10T11:51:48.503Z | user\n\n移除英文触发词修复Vague Triggers\n\nv3.1.7 | 2026-08-10T11:50:01.271Z | user\n\n移除英文触发词、修复Vague Triggers\n\nv3.1.6 | 2026-08-10T05:40:11.867Z | user\n\n去除旧env读取、扩展描述覆盖全功能、中英双语、修正权限格式、隐私声明首次运行确认\n\nv3.1.5 | 2026-08-10T05:14:53.413Z | user\n\n首次运行隐私声明确认、修复Missing User Warnings\n\nv3.1.4 | 2026-08-10T05:11:25.127Z | user\n\n移除自动IAM同步改为手动、修复Description-Behavior Mismatch\n\nv3.1.3 | 2026-08-10T04:15:35.989Z | user\n\nenv换json跨平台配置、新增隐私数据声明、权限元数据、修复全部安全审查问题\n\nv3.1.2 | 2026-08-10T03:48:18.816Z | user\n\n修复安全审查问题、收紧触发词、仅上传核心文件\n\nv3.1.1 | 2026-08-10T03:44:28.866Z | user\n\n修复安全审查问题、收紧触发词、仅上传核心文件、优化批量复制逻辑\n\nv3.0.6 | 2026-05-28T04:14:51.526Z | auto\n\nbytedance-visual-recognition 3.0.6\n\n- 简化和精炼文档内容，优化表述，突出多模态模型与免费福利说明\n- 移除模型调度2.0版本等历史描述，文档聚焦实际当前功能\n- 删除vision_history.json，去除不再需要的历史记录相关文件\n- 更新summary、description等元数据，强调协作计划免费使用顶级模型\n\nv3.0.5 | 2026-05-28T02:32:22.358Z | auto\n\n- Added vision_history.json to support 7-day usage history tracking.\n- Removed redundant documentation file skill-card.md.\n- Updated documentation for improved clarity and accurate description of model scheduling/prioritization.\n- No functional changes to commands or environment variables.\n\nv3.0.4 | 2026-05-27T18:19:42.748Z | auto\n\n- 移除了对\"追问\"（追问与输出结果相关的提问）功能的描述和支持\n- 同步更新说明文档，删除所有与追问相关的内容和操作示例\n- 其它使用方法、调用规则及配置流程保持不变\n\nv3.0.2 | 2026-05-27T18:17:45.918Z | auto\n\n- 更新了技能描述与文档，强调参与火山协作奖励计划可免费获得顶级多模态模型使用权限\n- 明确支持图片/视频转文字与代码、追问功能，无需重新上传\n- 去除冗余和重复内容，结构更简洁清晰\n- 保留详细首次配置与命令用法说明，便于新用户上手\n- 行为规则与参数推断无变动，执行规范保持一致\n\nv3.0.1 | 2026-05-27T18:12:32.351Z | auto\n\nVersion 3.0.1\n\n- 文档新增免费顶级模型使用说明，增加活动奖励计划入口和相关链接\n- 其他内容和规则未变，配置和调用方式保持一致\n\nv3.0.0 | 2026-05-27T16:27:50.877Z | auto\n\nVersion 3.0.0 — Major update with expanded features and new usage rules.\n\n- Adds support for both image and video recognition, including conversion to text and code.\n- Implements automatic multi-model fallback with smart scheduling to manage token limits and maintain service.\n- Introduces concise command-line usage for all recognition modes, batch processing, and follow-up questioning.\n- Requires stricter compliance with action rules: auto-execution upon trigger with no confirmation prompts.\n- Environment variables and setup steps updated for seamless integration with Doubao-Seed API endpoints.\n- Command triggers and usage instructions greatly expanded and clarified.\n\nArchive index:\n\nArchive v5.0.1: 5 files, 19058 bytes\n\nFiles: _meta.json (147b), doubao_vision_recognize.py (52134b), proposal/PROPOSAL.md (376b), skill-card.md (2810b), SKILL.md (5230b)\n\nFile v5.0.1:SKILL.md\n\n---\nname: bytedance-visual-recognition\ndescription: >\n  Multimodal visual recognition via Doubao-Seed (6 models) + Zhipu GLM (2 free models).\n  Image/video to text/code, auto-fallback, batch directory processing, follow-up conversations,\n  local media caching (Temp/), persistent history (vision_history.json), auto IAM console sync.\n  First-run privacy notice, plaintext config.json, cross-platform.\nsummary: \"Doubao-Seed + GLM visual recognition — image/video to text/code with auto-fallback, batch, follow-up, IAM sync\"\ntags:\n  vision: \"5.0.1\"\n  image-recognition: \"5.0.1\"\n  video-recognition: \"5.0.1\"\n  image-to-code: \"5.0.1\"\n  video-to-code: \"5.0.1\"\n  doubao: \"5.0.1\"\n  glm: \"5.0.1\"\ntrigger_patterns:\n  - \"豆包识别\"\n  - \"豆包视觉识别\"\n  - \"bytedance visual recognition\"\n  - \"doubao recognize\"\nmetadata:\n  openclaw:\n    requires:\n      bins:\n        - python\n    permissions:\n      filesystem:\n        read: [\"config.json\", \"vision_history.json\", \".last_response\"]\n        write: [\"Temp/\", \"vision_history.json\", \".last_response\", \"config.json\"]\n      network:\n        - host: \"ark.cn-beijing.volces.com\"\n          purpose: \"Doubao vision model API\"\n        - host: \"open.bigmodel.cn\"\n          purpose: \"GLM vision model Chat Completions API\"\n        - host: \"open.volcengineapi.com\"\n          purpose: \"Auto IAM console usage sync\"\n    emoji: \"🔍\"\n    homepage: https://www.volcengine.com/docs/82379/1569618\n    locales: [\"en\"]\n---\n\n# ByteDance Visual Recognition — Doubao-Seed + GLM\n\nDoubao-Seed (6 models) + Zhipu GLM (2 free models). First run auto-generates `config.json`, fill in one API Key to start. IAM console usage syncs automatically on each recognition.\n\n## Privacy & Data Notice\n\n- **Network**: Selected images/videos and prompts are base64-encoded and sent to Volcengine (Doubao) or Zhipu (GLM) cloud APIs.\n- **Local cache**: Media files temporarily copied to `Temp/YYYYMMDD/`, default 7-day retention (`temp_retention_days` in config.json, range 1-3650).\n- **History**: Recognition history stored in `vision_history.json`, follow-up context in `.last_response`, auto-cleaned after 7 days. GLM follow-up reuses full message history including base64 media data; Doubao follow-up uses previous_response_id without re-transmitting files.\n- **Credentials**: API Keys stored in plaintext `config.json`. Do not use personal keys on shared machines.\n- **IAM sync**: Automatically syncs token usage from Volcengine IAM console on each recognition when IAM credentials are configured.\n- **First run**: A privacy notice is displayed once. Continuing past it acknowledges data handling practices.\n\n## Setup (pick one)\n\n### Doubao (Volcengine)\n\n1. Join the [Collaboration Rewards Program](https://console.volcengine.com/ark/region:cn-beijing/openManagement/rewardPlan) for free quota, then get your API Key\n2. Create inference endpoints, pick from:\n\n| config key | model | priority |\n|--------|------|:---:|\n| `doubao_seed_21p_id` | Doubao-Seed-2.1-Pro | primary |\n| `doubao_seed_21t_id` | Doubao-Seed-2.1-Turbo | secondary |\n| `doubao_seed_20p_id` | Doubao-Seed-2.0-Pro | tertiary |\n| `doubao_seed_20c_id` | Doubao-Seed-2.0-Code | code-first |\n| `doubao_seed_20l_id` | Doubao-Seed-2.0-Lite | fallback |\n| `doubao_seed_20m_id` | Doubao-Seed-2.0-Mini | low-cost |\n\n3. Edit `config.json`, replace `\"\"` with actual endpoint IDs.\n\n### GLM (Zhipu, free)\n\n1. https://open.bigmodel.cn → get API Key\n2. Edit `config.json`: `\"zhipu_api_key\": \"your-key\"`\n\n### Provider filter\n\n`provider_mode` in `config.json`:\n- `0` = all (default)\n- `1` = Zhipu only\n- `2` = Doubao only\n\n### Test\n\n```bash\npython doubao_vision_recognize.py --help\npython doubao_vision_recognize.py status\n```\n\n---\n\n## Commands\n\n| command | purpose | example |\n|------|------|------|\n| `rec <file> --image\\|--video --text\\|--code` | recognize | `rec a.jpg --image --text` |\n| `rec <dir> --batch --image\\|--video --text\\|--code` | batch | `rec ./img/ --batch --image --text` |\n| `ask --text\\|--code --prompt \"...\"` | follow-up | `ask --text -p \"details\"` |\n| `status` | usage stats | |\n| `sync` | manual console sync | |\n| `history` | 7-day history | |\n\n### Parameters\n\n| param | desc |\n|------|------|\n| `--image` | image input |\n| `--video` | video input |\n| `--text` | text output |\n| `--code` | code output |\n| `--prompt` / `-p` | extra instruction |\n| `--batch` | directory batch |\n\n---\n\n## Behavior Rules\n\n### 1. Trigger only on listed patterns\nActivate only when the user's message matches one of the trigger_patterns above. Do NOT activate on loosely related text. If uncertain, ask before executing.\n\n### 2. Parameter inference\n- \"recognize/analyze image\" → `--image --text`\n- \"recognize/analyze video\" → `--video --text`\n- \"convert to code / UI to code / design to code\" → `--code`\n- extra requirements → `--prompt \"...\"`\n- unsure → ask once if image or video\n\n### 3. Credential safety\nAPI keys are stored in plaintext config.json. Warn users not to use high-value keys on shared machines. First-run privacy notice already discloses plaintext storage.\n\n---\n\n## Limits\n\n- Doubao: 180W tokens per model per day, auto-fallback\n- GLM: free, auto-retry on failure (4.6V: 10 retries, 4.1V: 5 retries)\n- Image ≤ 15MB, Video ≤ 50MB\n\nFile v5.0.1:_meta.json\n\n{\n  \"ownerId\": \"kn75s192k98dtwr83bft4myd0582ytyg\",\n  \"slug\": \"bytedance-visual-recognition\",\n  \"version\": \"5.0.1\",\n  \"publishedAt\": 1786388451204\n}\n\nFile v5.0.1:proposal/PROPOSAL.md\n\n# 3.1.1 更新提案\n\n## 目标\n精简发布，修复安全审查问题，仅上传核心文件。\n\n## 变更\n- 收紧触发词列表，移除过于宽泛的 pattern\n- 批量处理改为仅复制媒体文件（不再复制整个目录树）\n- 剔除非核心文件（.env, Temp, .bak, .json 等）\n- 仅上传核心文件：doubao_vision_recognize.py, SKILL.md, skill-card.md\n\nFile v5.0.1:skill-card.md\n\n## Description:\n\nMultimodal visual recognition via Doubao-Seed and Zhipu GLM for image or video analysis, image or video to code, batch processing, follow-up questions, provider fallback, local media caching, persistent history, and optional IAM usage sync.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[etmnb](https://clawhub.ai/user/etmnb)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and external users can use this skill to send selected images or videos to configured Doubao or GLM vision models and receive scene descriptions, extracted text, structured Markdown analysis, or generated code. It also supports batch directory processing and follow-up questions over prior recognition context.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Selected images, videos, screenshots, documents, and prompts may contain private or sensitive information and are sent to third-party cloud vision APIs.\n\nMitigation: Review media before use, avoid sending regulated or confidential content unless approved, and configure only providers whose data handling terms are acceptable for the deployment.\n\nRisk: API keys and optional IAM credentials are stored in plaintext config.json.\n\nMitigation: Use least-privileged provider keys, avoid shared or synced folders, rotate exposed keys, and remove unused credentials from config.json.\n\nRisk: Local files such as Temp/, vision_history.json, and .last_response can retain media context or model responses after recognition.\n\nMitigation: Set the shortest practical retention period, manually delete Temp/, vision_history.json, and .last_response after sensitive sessions, and do not rely on automatic cleanup for GLM follow-up context.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/etmnb/skills/bytedance-visual-recognition)\n- [Volcengine Doubao visual understanding documentation](https://www.volcengine.com/docs/82379/1569618)\n- [Zhipu Open Platform](https://open.bigmodel.cn)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, Code, Shell commands, Configuration, Guidance]\n\n**Output Format:** [CLI text and Markdown-style analysis, with generated code when code output is requested.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Supports image files up to 15 MB, video files up to 50 MB, batch directory processing, follow-up context, local cache files, and persistent history.]\n\n## Skill Version(s):\n\n5.0.1 (source: server release metadata; artifact script and skill tags agree)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v5.0.0: 5 files, 18873 bytes\n\nFiles: _meta.json (147b), doubao_vision_recognize.py (52134b), proposal/PROPOSAL.md (376b), skill-card.md (2325b), SKILL.md (5230b)\n\nFile v5.0.0:SKILL.md\n\n---\nname: bytedance-visual-recognition\ndescription: >\n  Multimodal visual recognition via Doubao-Seed (6 models) + Zhipu GLM (2 free models).\n  Image/video to text/code, auto-fallback, batch directory processing, follow-up conversations,\n  local media caching (Temp/), persistent history (vision_history.json), auto IAM console sync.\n  First-run privacy notice, plaintext config.json, cross-platform.\nsummary: \"Doubao-Seed + GLM visual recognition — image/video to text/code with auto-fallback, batch, follow-up, IAM sync\"\ntags:\n  vision: \"5.0.0\"\n  image-recognition: \"5.0.0\"\n  video-recognition: \"5.0.0\"\n  image-to-code: \"5.0.0\"\n  video-to-code: \"5.0.0\"\n  doubao: \"5.0.0\"\n  glm: \"5.0.0\"\ntrigger_patterns:\n  - \"豆包识别\"\n  - \"豆包视觉识别\"\n  - \"bytedance visual recognition\"\n  - \"doubao recognize\"\nmetadata:\n  openclaw:\n    requires:\n      bins:\n        - python\n    permissions:\n      filesystem:\n        read: [\"config.json\", \"vision_history.json\", \".last_response\"]\n        write: [\"Temp/\", \"vision_history.json\", \".last_response\", \"config.json\"]\n      network:\n        - host: \"ark.cn-beijing.volces.com\"\n          purpose: \"Doubao vision model API\"\n        - host: \"open.bigmodel.cn\"\n          purpose: \"GLM vision model Chat Completions API\"\n        - host: \"open.volcengineapi.com\"\n          purpose: \"Auto IAM console usage sync\"\n    emoji: \"🔍\"\n    homepage: https://www.volcengine.com/docs/82379/1569618\n    locales: [\"en\"]\n---\n\n# ByteDance Visual Recognition — Doubao-Seed + GLM\n\nDoubao-Seed (6 models) + Zhipu GLM (2 free models). First run auto-generates `config.json`, fill in one API Key to start. IAM console usage syncs automatically on each recognition.\n\n## Privacy & Data Notice\n\n- **Network**: Selected images/videos and prompts are base64-encoded and sent to Volcengine (Doubao) or Zhipu (GLM) cloud APIs.\n- **Local cache**: Media files temporarily copied to `Temp/YYYYMMDD/`, default 7-day retention (`temp_retention_days` in config.json, range 1-3650).\n- **History**: Recognition history stored in `vision_history.json`, follow-up context in `.last_response`, auto-cleaned after 7 days. GLM follow-up reuses full message history including base64 media data; Doubao follow-up uses previous_response_id without re-transmitting files.\n- **Credentials**: API Keys stored in plaintext `config.json`. Do not use personal keys on shared machines.\n- **IAM sync**: Automatically syncs token usage from Volcengine IAM console on each recognition when IAM credentials are configured.\n- **First run**: A privacy notice is displayed once. Continuing past it acknowledges data handling practices.\n\n## Setup (pick one)\n\n### Doubao (Volcengine)\n\n1. Join the [Collaboration Rewards Program](https://console.volcengine.com/ark/region:cn-beijing/openManagement/rewardPlan) for free quota, then get your API Key\n2. Create inference endpoints, pick from:\n\n| config key | model | priority |\n|--------|------|:---:|\n| `doubao_seed_21p_id` | Doubao-Seed-2.1-Pro | primary |\n| `doubao_seed_21t_id` | Doubao-Seed-2.1-Turbo | secondary |\n| `doubao_seed_20p_id` | Doubao-Seed-2.0-Pro | tertiary |\n| `doubao_seed_20c_id` | Doubao-Seed-2.0-Code | code-first |\n| `doubao_seed_20l_id` | Doubao-Seed-2.0-Lite | fallback |\n| `doubao_seed_20m_id` | Doubao-Seed-2.0-Mini | low-cost |\n\n3. Edit `config.json`, replace `\"\"` with actual endpoint IDs.\n\n### GLM (Zhipu, free)\n\n1. https://open.bigmodel.cn → get API Key\n2. Edit `config.json`: `\"zhipu_api_key\": \"your-key\"`\n\n### Provider filter\n\n`provider_mode` in `config.json`:\n- `0` = all (default)\n- `1` = Zhipu only\n- `2` = Doubao only\n\n### Test\n\n```bash\npython doubao_vision_recognize.py --help\npython doubao_vision_recognize.py status\n```\n\n---\n\n## Commands\n\n| command | purpose | example |\n|------|------|------|\n| `rec <file> --image\\|--video --text\\|--code` | recognize | `rec a.jpg --image --text` |\n| `rec <dir> --batch --image\\|--video --text\\|--code` | batch | `rec ./img/ --batch --image --text` |\n| `ask --text\\|--code --prompt \"...\"` | follow-up | `ask --text -p \"details\"` |\n| `status` | usage stats | |\n| `sync` | manual console sync | |\n| `history` | 7-day history | |\n\n### Parameters\n\n| param | desc |\n|------|------|\n| `--image` | image input |\n| `--video` | video input |\n| `--text` | text output |\n| `--code` | code output |\n| `--prompt` / `-p` | extra instruction |\n| `--batch` | directory batch |\n\n---\n\n## Behavior Rules\n\n### 1. Trigger only on listed patterns\nActivate only when the user's message matches one of the trigger_patterns above. Do NOT activate on loosely related text. If uncertain, ask before executing.\n\n### 2. Parameter inference\n- \"recognize/analyze image\" → `--image --text`\n- \"recognize/analyze video\" → `--video --text`\n- \"convert to code / UI to code / design to code\" → `--code`\n- extra requirements → `--prompt \"...\"`\n- unsure → ask once if image or video\n\n### 3. Credential safety\nAPI keys are stored in plaintext config.json. Warn users not to use high-value keys on shared machines. First-run privacy notice already discloses plaintext storage.\n\n---\n\n## Limits\n\n- Doubao: 180W tokens per model per day, auto-fallback\n- GLM: free, auto-retry on failure (4.6V: 10 retries, 4.1V: 5 retries)\n- Image ≤ 15MB, Video ≤ 50MB\n\nFile v5.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn75s192k98dtwr83bft4myd0582ytyg\",\n  \"slug\": \"bytedance-visual-recognition\",\n  \"version\": \"5.0.0\",\n  \"publishedAt\": 1786388278644\n}\n\nFile v5.0.0:proposal/PROPOSAL.md\n\n# 3.1.1 更新提案\n\n## 目标\n精简发布，修复安全审查问题，仅上传核心文件。\n\n## 变更\n- 收紧触发词列表，移除过于宽泛的 pattern\n- 批量处理改为仅复制媒体文件（不再复制整个目录树）\n- 剔除非核心文件（.env, Temp, .bak, .json 等）\n- 仅上传核心文件：doubao_vision_recognize.py, SKILL.md, skill-card.md\n\nFile v5.0.0:skill-card.md\n\n## Description:\n\nMultimodal visual recognition via Doubao-Seed and Zhipu GLM for image/video-to-text or image/video-to-code workflows with fallback, batch processing, follow-up conversations, local caching, persistent history, and IAM usage sync.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[etmnb](https://clawhub.ai/user/etmnb)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and external users use this skill to analyze selected images or videos, generate text descriptions or code from visual inputs, process batches, and ask follow-up questions through Doubao or GLM cloud APIs.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Selected images, videos, and prompts are sent to Volcengine or Zhipu cloud services.\n\nMitigation: Use only media and prompts that are appropriate to share with those services, and obtain consent or approval for sensitive content before execution.\n\nRisk: API keys may be stored in plaintext config.json.\n\nMitigation: Avoid high-value credentials on shared or untrusted machines and rotate keys if the local workspace may have been exposed.\n\nRisk: Temporary media, recognition history, and follow-up context can remain in local files.\n\nMitigation: Clear Temp/, vision_history.json, and .last_response when media, prompts, or outputs are sensitive.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/etmnb/skills/bytedance-visual-recognition)\n- [Volcengine Doubao documentation](https://www.volcengine.com/docs/82379/1569618)\n- [Zhipu Open Platform](https://open.bigmodel.cn)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, code, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown, plain text, code snippets, and shell command examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May create or update local configuration, temporary media cache, history, and follow-up context files during use.]\n\n## Skill Version(s):\n\n5.0.0 (source: server release evidence, SKILL.md tags, script header)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v3.1.23: 5 files, 18932 bytes\n\nFiles: _meta.json (148b), doubao_vision_recognize.py (52136b), proposal/PROPOSAL.md (376b), skill-card.md (2435b), SKILL.md (5237b)\n\nFile v3.1.23:SKILL.md\n\n---\nname: bytedance-visual-recognition\ndescription: >\n  Multimodal visual recognition via Doubao-Seed (6 models) + Zhipu GLM (2 free models).\n  Image/video to text/code, auto-fallback, batch directory processing, follow-up conversations,\n  local media caching (Temp/), persistent history (vision_history.json), auto IAM console sync.\n  First-run privacy notice, plaintext config.json, cross-platform.\nsummary: \"Doubao-Seed + GLM visual recognition — image/video to text/code with auto-fallback, batch, follow-up, IAM sync\"\ntags:\n  vision: \"3.1.23\"\n  image-recognition: \"3.1.23\"\n  video-recognition: \"3.1.23\"\n  image-to-code: \"3.1.23\"\n  video-to-code: \"3.1.23\"\n  doubao: \"3.1.23\"\n  glm: \"3.1.23\"\ntrigger_patterns:\n  - \"豆包识别\"\n  - \"豆包视觉识别\"\n  - \"bytedance visual recognition\"\n  - \"doubao recognize\"\nmetadata:\n  openclaw:\n    requires:\n      bins:\n        - python\n    permissions:\n      filesystem:\n        read: [\"config.json\", \"vision_history.json\", \".last_response\"]\n        write: [\"Temp/\", \"vision_history.json\", \".last_response\", \"config.json\"]\n      network:\n        - host: \"ark.cn-beijing.volces.com\"\n          purpose: \"Doubao vision model API\"\n        - host: \"open.bigmodel.cn\"\n          purpose: \"GLM vision model Chat Completions API\"\n        - host: \"open.volcengineapi.com\"\n          purpose: \"Auto IAM console usage sync\"\n    emoji: \"🔍\"\n    homepage: https://www.volcengine.com/docs/82379/1569618\n    locales: [\"en\"]\n---\n\n# ByteDance Visual Recognition — Doubao-Seed + GLM\n\nDoubao-Seed (6 models) + Zhipu GLM (2 free models). First run auto-generates `config.json`, fill in one API Key to start. IAM console usage syncs automatically on each recognition.\n\n## Privacy & Data Notice\n\n- **Network**: Selected images/videos and prompts are base64-encoded and sent to Volcengine (Doubao) or Zhipu (GLM) cloud APIs.\n- **Local cache**: Media files temporarily copied to `Temp/YYYYMMDD/`, default 7-day retention (`temp_retention_days` in config.json, range 1-3650).\n- **History**: Recognition history stored in `vision_history.json`, follow-up context in `.last_response`, auto-cleaned after 7 days. GLM follow-up reuses full message history including base64 media data; Doubao follow-up uses previous_response_id without re-transmitting files.\n- **Credentials**: API Keys stored in plaintext `config.json`. Do not use personal keys on shared machines.\n- **IAM sync**: Automatically syncs token usage from Volcengine IAM console on each recognition when IAM credentials are configured.\n- **First run**: A privacy notice is displayed once. Continuing past it acknowledges data handling practices.\n\n## Setup (pick one)\n\n### Doubao (Volcengine)\n\n1. Join the [Collaboration Rewards Program](https://console.volcengine.com/ark/region:cn-beijing/openManagement/rewardPlan) for free quota, then get your API Key\n2. Create inference endpoints, pick from:\n\n| config key | model | priority |\n|--------|------|:---:|\n| `doubao_seed_21p_id` | Doubao-Seed-2.1-Pro | primary |\n| `doubao_seed_21t_id` | Doubao-Seed-2.1-Turbo | secondary |\n| `doubao_seed_20p_id` | Doubao-Seed-2.0-Pro | tertiary |\n| `doubao_seed_20c_id` | Doubao-Seed-2.0-Code | code-first |\n| `doubao_seed_20l_id` | Doubao-Seed-2.0-Lite | fallback |\n| `doubao_seed_20m_id` | Doubao-Seed-2.0-Mini | low-cost |\n\n3. Edit `config.json`, replace `\"\"` with actual endpoint IDs.\n\n### GLM (Zhipu, free)\n\n1. https://open.bigmodel.cn → get API Key\n2. Edit `config.json`: `\"zhipu_api_key\": \"your-key\"`\n\n### Provider filter\n\n`provider_mode` in `config.json`:\n- `0` = all (default)\n- `1` = Zhipu only\n- `2` = Doubao only\n\n### Test\n\n```bash\npython doubao_vision_recognize.py --help\npython doubao_vision_recognize.py status\n```\n\n---\n\n## Commands\n\n| command | purpose | example |\n|------|------|------|\n| `rec <file> --image\\|--video --text\\|--code` | recognize | `rec a.jpg --image --text` |\n| `rec <dir> --batch --image\\|--video --text\\|--code` | batch | `rec ./img/ --batch --image --text` |\n| `ask --text\\|--code --prompt \"...\"` | follow-up | `ask --text -p \"details\"` |\n| `status` | usage stats | |\n| `sync` | manual console sync | |\n| `history` | 7-day history | |\n\n### Parameters\n\n| param | desc |\n|------|------|\n| `--image` | image input |\n| `--video` | video input |\n| `--text` | text output |\n| `--code` | code output |\n| `--prompt` / `-p` | extra instruction |\n| `--batch` | directory batch |\n\n---\n\n## Behavior Rules\n\n### 1. Trigger only on listed patterns\nActivate only when the user's message matches one of the trigger_patterns above. Do NOT activate on loosely related text. If uncertain, ask before executing.\n\n### 2. Parameter inference\n- \"recognize/analyze image\" → `--image --text`\n- \"recognize/analyze video\" → `--video --text`\n- \"convert to code / UI to code / design to code\" → `--code`\n- extra requirements → `--prompt \"...\"`\n- unsure → ask once if image or video\n\n### 3. Credential safety\nAPI keys are stored in plaintext config.json. Warn users not to use high-value keys on shared machines. First-run privacy notice already discloses plaintext storage.\n\n---\n\n## Limits\n\n- Doubao: 180W tokens per model per day, auto-fallback\n- GLM: free, auto-retry on failure (4.6V: 10 retries, 4.1V: 5 retries)\n- Image ≤ 15MB, Video ≤ 50MB\n\nFile v3.1.23:_meta.json\n\n{\n  \"ownerId\": \"kn75s192k98dtwr83bft4myd0582ytyg\",\n  \"slug\": \"bytedance-visual-recognition\",\n  \"version\": \"3.1.23\",\n  \"publishedAt\": 1786386551375\n}\n\nFile v3.1.23:proposal/PROPOSAL.md\n\n# 3.1.1 更新提案\n\n## 目标\n精简发布，修复安全审查问题，仅上传核心文件。\n\n## 变更\n- 收紧触发词列表，移除过于宽泛的 pattern\n- 批量处理改为仅复制媒体文件（不再复制整个目录树）\n- 剔除非核心文件（.env, Temp, .bak, .json 等）\n- 仅上传核心文件：doubao_vision_recognize.py, SKILL.md, skill-card.md\n\nFile v3.1.23:skill-card.md\n\n## Description:\n\nMultimodal visual recognition via Doubao-Seed and Zhipu GLM for image/video to text/code workflows, with auto-fallback, batch directory processing, follow-up conversations, local media caching, persistent history, auto IAM console sync, first-run privacy notice, plaintext config.json, and cross-platform Python support.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[etmnb](https://clawhub.ai/user/etmnb)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nExternal developers and agent users use this skill to analyze images or videos with Doubao and GLM cloud vision models, including text descriptions, UI-to-code output, batch processing, and follow-up questions.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Selected images, videos, prompts, and follow-up context may be sent to Doubao or GLM cloud APIs.\n\nMitigation: Use only media and prompts that are approved for those providers, and avoid sensitive or regulated content unless the deployment has appropriate approval.\n\nRisk: Provider credentials are stored in plaintext config.json.\n\nMitigation: Use limited-scope or low-value keys, avoid shared or synced machines, and rotate or remove keys after use.\n\nRisk: Local cache, history, and follow-up context can persist on disk.\n\nMitigation: Clear Temp/, vision_history.json, and .last_response when the retained media, prompts, or responses should no longer remain available.\n\n## Reference(s):\n\n- [Volcengine Doubao Vision Documentation](https://www.volcengine.com/docs/82379/1569618)\n- [Zhipu Open Platform](https://open.bigmodel.cn)\n- [ClawHub Skill Release Page](https://clawhub.ai/etmnb/skills/bytedance-visual-recognition)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, code, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown and terminal text with generated descriptions, code, usage status, and setup guidance]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May write local config, cache, history, and follow-up context files as part of operation.]\n\n## Skill Version(s):\n\n3.1.23 (source: server evidence release.version)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v3.1.21: 5 files, 18896 bytes\n\nFiles: _meta.json (148b), doubao_vision_recognize.py (52136b), proposal/PROPOSAL.md (376b), skill-card.md (2398b), SKILL.md (5237b)\n\nFile v3.1.21:SKILL.md\n\n---\nname: bytedance-visual-recognition\ndescription: >\n  Multimodal visual recognition via Doubao-Seed (6 models) + Zhipu GLM (2 free models).\n  Image/video to text/code, auto-fallback, batch directory processing, follow-up conversations,\n  local media caching (Temp/), persistent history (vision_history.json), auto IAM console sync.\n  First-run privacy notice, plaintext config.json, cross-platform.\nsummary: \"Doubao-Seed + GLM visual recognition — image/video to text/code with auto-fallback, batch, follow-up, IAM sync\"\ntags:\n  vision: \"3.1.21\"\n  image-recognition: \"3.1.21\"\n  video-recognition: \"3.1.21\"\n  image-to-code: \"3.1.21\"\n  video-to-code: \"3.1.21\"\n  doubao: \"3.1.21\"\n  glm: \"3.1.21\"\ntrigger_patterns:\n  - \"豆包识别\"\n  - \"豆包视觉识别\"\n  - \"bytedance visual recognition\"\n  - \"doubao recognize\"\nmetadata:\n  openclaw:\n    requires:\n      bins:\n        - python\n    permissions:\n      filesystem:\n        read: [\"config.json\", \"vision_history.json\", \".last_response\"]\n        write: [\"Temp/\", \"vision_history.json\", \".last_response\", \"config.json\"]\n      network:\n        - host: \"ark.cn-beijing.volces.com\"\n          purpose: \"Doubao vision model API\"\n        - host: \"open.bigmodel.cn\"\n          purpose: \"GLM vision model Chat Completions API\"\n        - host: \"open.volcengineapi.com\"\n          purpose: \"Auto IAM console usage sync\"\n    emoji: \"🔍\"\n    homepage: https://www.volcengine.com/docs/82379/1569618\n    locales: [\"en\"]\n---\n\n# ByteDance Visual Recognition — Doubao-Seed + GLM\n\nDoubao-Seed (6 models) + Zhipu GLM (2 free models). First run auto-generates `config.json`, fill in one API Key to start. IAM console usage syncs automatically on each recognition.\n\n## Privacy & Data Notice\n\n- **Network**: Selected images/videos and prompts are base64-encoded and sent to Volcengine (Doubao) or Zhipu (GLM) cloud APIs.\n- **Local cache**: Media files temporarily copied to `Temp/YYYYMMDD/`, default 7-day retention (`temp_retention_days` in config.json, range 1-3650).\n- **History**: Recognition history stored in `vision_history.json`, follow-up context in `.last_response`, auto-cleaned after 7 days. GLM follow-up reuses full message history including base64 media data; Doubao follow-up uses previous_response_id without re-transmitting files.\n- **Credentials**: API Keys stored in plaintext `config.json`. Do not use personal keys on shared machines.\n- **IAM sync**: Automatically syncs token usage from Volcengine IAM console on each recognition when IAM credentials are configured.\n- **First run**: A privacy notice is displayed once. Continuing past it acknowledges data handling practices.\n\n## Setup (pick one)\n\n### Doubao (Volcengine)\n\n1. Join the [Collaboration Rewards Program](https://console.volcengine.com/ark/region:cn-beijing/openManagement/rewardPlan) for free quota, then get your API Key\n2. Create inference endpoints, pick from:\n\n| config key | model | priority |\n|--------|------|:---:|\n| `doubao_seed_21p_id` | Doubao-Seed-2.1-Pro | primary |\n| `doubao_seed_21t_id` | Doubao-Seed-2.1-Turbo | secondary |\n| `doubao_seed_20p_id` | Doubao-Seed-2.0-Pro | tertiary |\n| `doubao_seed_20c_id` | Doubao-Seed-2.0-Code | code-first |\n| `doubao_seed_20l_id` | Doubao-Seed-2.0-Lite | fallback |\n| `doubao_seed_20m_id` | Doubao-Seed-2.0-Mini | low-cost |\n\n3. Edit `config.json`, replace `\"\"` with actual endpoint IDs.\n\n### GLM (Zhipu, free)\n\n1. https://open.bigmodel.cn → get API Key\n2. Edit `config.json`: `\"zhipu_api_key\": \"your-key\"`\n\n### Provider filter\n\n`provider_mode` in `config.json`:\n- `0` = all (default)\n- `1` = Zhipu only\n- `2` = Doubao only\n\n### Test\n\n```bash\npython doubao_vision_recognize.py --help\npython doubao_vision_recognize.py status\n```\n\n---\n\n## Commands\n\n| command | purpose | example |\n|------|------|------|\n| `rec <file> --image\\|--video --text\\|--code` | recognize | `rec a.jpg --image --text` |\n| `rec <dir> --batch --image\\|--video --text\\|--code` | batch | `rec ./img/ --batch --image --text` |\n| `ask --text\\|--code --prompt \"...\"` | follow-up | `ask --text -p \"details\"` |\n| `status` | usage stats | |\n| `sync` | manual console sync | |\n| `history` | 7-day history | |\n\n### Parameters\n\n| param | desc |\n|------|------|\n| `--image` | image input |\n| `--video` | video input |\n| `--text` | text output |\n| `--code` | code output |\n| `--prompt` / `-p` | extra instruction |\n| `--batch` | directory batch |\n\n---\n\n## Behavior Rules\n\n### 1. Trigger only on listed patterns\nActivate only when the user's message matches one of the trigger_patterns above. Do NOT activate on loosely related text. If uncertain, ask before executing.\n\n### 2. Parameter inference\n- \"recognize/analyze image\" → `--image --text`\n- \"recognize/analyze video\" → `--video --text`\n- \"convert to code / UI to code / design to code\" → `--code`\n- extra requirements → `--prompt \"...\"`\n- unsure → ask once if image or video\n\n### 3. Credential safety\nAPI keys are stored in plaintext config.json. Warn users not to use high-value keys on shared machines. First-run privacy notice already discloses plaintext storage.\n\n---\n\n## Limits\n\n- Doubao: 180W tokens per model per day, auto-fallback\n- GLM: free, auto-retry on failure (4.6V: 10 retries, 4.1V: 5 retries)\n- Image ≤ 15MB, Video ≤ 50MB\n\nFile v3.1.21:_meta.json\n\n{\n  \"ownerId\": \"kn75s192k98dtwr83bft4myd0582ytyg\",\n  \"slug\": \"bytedance-visual-recognition\",\n  \"version\": \"3.1.21\",\n  \"publishedAt\": 1786384793236\n}\n\nFile v3.1.21:proposal/PROPOSAL.md\n\n# 3.1.1 更新提案\n\n## 目标\n精简发布，修复安全审查问题，仅上传核心文件。\n\n## 变更\n- 收紧触发词列表，移除过于宽泛的 pattern\n- 批量处理改为仅复制媒体文件（不再复制整个目录树）\n- 剔除非核心文件（.env, Temp, .bak, .json 等）\n- 仅上传核心文件：doubao_vision_recognize.py, SKILL.md, skill-card.md\n\nFile v3.1.21:skill-card.md\n\n## Description:\n\nMultimodal visual recognition via Doubao-Seed and Zhipu GLM for image and video analysis, image or video to code, batch processing, follow-up questions, local caching, history, and usage sync.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[etmnb](https://clawhub.ai/user/etmnb)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and other agent users use this skill to send selected images or videos to Doubao or GLM vision models for concise descriptions, text extraction, UI-to-code generation, video-to-code generation, batch recognition, and follow-up analysis.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Selected images, videos, and prompts are sent to named cloud vision providers.\n\nMitigation: Use only with media and prompts that are appropriate to share with Volcengine or Zhipu, and avoid private screenshots, IDs, and business documents unless that sharing is approved.\n\nRisk: Sensitive media, recognition history, and follow-up context can remain on the local machine.\n\nMitigation: Review and clear Temp/, vision_history.json, and .last_response when working with sensitive content or shared machines.\n\nRisk: API keys and optional IAM credentials are stored in plaintext config.json.\n\nMitigation: Use scoped, replaceable credentials and avoid using personal or high-value keys on shared, backed-up, or synced machines.\n\n## Reference(s):\n\n- [Volcengine Doubao Vision Documentation](https://www.volcengine.com/docs/82379/1569618)\n- [ClawHub Skill Page](https://clawhub.ai/etmnb/skills/bytedance-visual-recognition)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, Code, Shell commands, Configuration, Guidance]\n\n**Output Format:** [Markdown and terminal text, with generated code when code output is requested]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Can produce natural-language visual descriptions, Markdown tables or extracted text, code snippets, status summaries, history summaries, and setup guidance.]\n\n## Skill Version(s):\n\n3.1.21 (source: server release metadata and SKILL.md tags)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v3.1.17: 5 files, 20140 bytes\n\nFiles: _meta.json (148b), doubao_vision_recognize.py (52941b), proposal/PROPOSAL.md (376b), skill-card.md (2753b), SKILL.md (5921b)\n\nFile v3.1.17:SKILL.md\n\n---\nname: bytedance-visual-recognition\ndescription: >\n  ByteDance Visual Recognition — 调用豆包 Doubao-Seed + 智谱 GLM 双后端多模态模型识别图片和视频，\n  输出文字或代码。支持单文件识别、批量目录处理、追问（基于上次结果的对话），自动模型降级与重试。\n  本地缓存媒体文件（Temp/YYYYMMDD/），持久化识别历史（vision_history.json）和追问上下文（.last_response），\n  可选手动 IAM 控制台用量同步。首次运行自动生成 config.json 并显示隐私声明。\n  Supports both Chinese and English interactions.\nsummary: \"Doubao + GLM visual recognition — image/video to text/code, local caching, history, batch, follow-up, optional IAM sync\"\ntags:\n  vision: \"3.1.17\"\n  image-recognition: \"3.1.10\"\n  video-recognition: \"3.1.10\"\n  image-to-code: \"3.1.10\"\n  video-to-code: \"3.1.10\"\n  doubao: \"3.1.10\"\n  glm: \"3.1.10\"\n  image-recognition: \"3.1.8\"\n  video-recognition: \"3.1.8\"\n  image-to-code: \"3.1.8\"\n  video-to-code: \"3.1.10\"\n  doubao: \"3.1.10\"\n  glm: \"3.1.10\"\ntrigger_patterns:\n  - \"豆包识别\"\n  - \"豆包视觉识别\"\n  - \"bytedance visual recognition\"\n  - \"doubao recognize\"\nmetadata:\n  openclaw:\n    requires:\n      bins:\n        - python\n    permissions:\n      filesystem:\n        read: [\"config.json\", \"vision_history.json\", \".last_response\"]\n        write: [\"Temp/\", \"vision_history.json\", \".last_response\", \"config.json\"]\n      network:\n        - host: \"ark.cn-beijing.volces.com\"\n          purpose: \"Doubao vision model API\"\n        - host: \"open.bigmodel.cn\"\n          purpose: \"GLM vision model Chat Completions API\"\n        - host: \"open.volcengineapi.com\"\n          purpose: \"Optional manual IAM usage sync\"\n    emoji: \"🔍\"\n    homepage: https://www.volcengine.com/docs/82379/1569618\n    locales: [\"zh-CN\", \"en\"]\n---\n\n# ByteDance Visual Recognition — 豆包 + GLM 双后端视觉识别\n# ByteDance Visual Recognition — Doubao + GLM Dual-Backend Vision\n\nDoubao-Seed (6 models) + Zhipu GLM (2 free models). First run auto-generates `config.json`. Fill in one API Key to start. This skill supports commands and interactions in both Chinese and English. / 支持中英文交互。\n\n## ⚠️ Privacy & Data Notice / 隐私与数据声明\n\n- **Network / 网络**: Selected images/videos and prompts are base64-encoded and sent to Volcengine (Doubao) or Zhipu (GLM) cloud APIs.\n- **Local cache / 本地缓存**: Media files temporarily copied to `Temp/YYYYMMDD/`, default 7-day retention (`temp_retention_days` in config.json, range 1-3650).\n- **History / 使用记录**: Recognition history stored in `vision_history.json`, follow-up context in `.last_response`, auto-cleaned after 7 days. **GLM follow-up reuses full message history including base64 media data**; Doubao follow-up uses previous_response_id without re-transmitting files.\n- **Credentials / 凭证**: API Keys stored in plaintext `config.json`. Do not use personal keys on shared machines.\n- **IAM sync / IAM 同步**: Only triggers via explicit `sync` command when IAM keys are configured. No automatic outbound calls.\n- **First run / 首次运行**: A privacy notice is displayed once. Continuing past it constitutes acknowledgment of data handling practices.\n\n## 🚀 Setup (pick one)\n\n### Doubao (Volcengine)\n\n1. 参与[协作奖励计划](https://console.volcengine.com/ark/region:cn-beijing/openManagement/rewardPlan)享免费额度，获取 API Key\n2. Create inference endpoints, pick from:\n\n| config key | model | priority |\n|--------|------|:---:|\n| `doubao_vision_21p_id` | Doubao-Seed-2.1-Pro | primary |\n| `doubao_vision_21t_id` | Doubao-Seed-2.1-Turbo | secondary |\n| `doubao_vision_20p_id` | Doubao-Seed-2.0-Pro | tertiary |\n| `doubao_vision_20c_id` | Doubao-Seed-2.0-Code | code-first |\n| `doubao_vision_20l_id` | Doubao-Seed-2.0-Lite | fallback |\n| `doubao_vision_20m_id` | Doubao-Seed-2.0-Mini | low-cost |\n\n3. Edit `config.json`, replace `\"\"` with actual endpoint IDs.\n\n### GLM (Zhipu, free)\n\n1. https://open.bigmodel.cn → get API Key\n2. Edit `config.json`: `\"zhipu_api_key\": \"your-key\"`\n\n### Provider filter\n\n`provider_mode` in `config.json`:\n- `0` = all (default)\n- `1` = Zhipu only\n- `2` = Doubao only\n\n### Test\n\n```bash\npython doubao_vision_recognize.py --help\npython doubao_vision_recognize.py status\n```\n\n---\n\n## ⚡ Commands\n\n| command | purpose | example |\n|------|------|------|\n| `rec <file> --image\\|--video --text\\|--code` | recognize | `rec a.jpg --image --text` |\n| `rec <dir> --batch --image\\|--video --text\\|--code` | batch | `rec ./img/ --batch --image --text` |\n| `ask --text\\|--code --prompt \"...\"` | follow-up | `ask --text -p \"details\"` |\n| `status` | usage stats | |\n| `sync` | console sync (manual) | |\n| `history` | 7-day history | |\n\n### Parameters\n\n| param | desc |\n|------|------|\n| `--image` | image input |\n| `--video` | video input |\n| `--text` | text output |\n| `--code` | code output |\n| `--prompt` / `-p` | extra instruction |\n| `--batch` | directory batch |\n\n---\n\n## 🚫 Behavior Rules\n\n### 1. Trigger only on listed patterns\n- Activate only when the user's message exactly matches one of the trigger_patterns listed in frontmatter.\n- Do NOT activate on loosely related text. If uncertain, ask before executing.\n\n### 2. Parameter inference\n- \"recognize/analyze image\" → `--image --text`\n- \"recognize/analyze video\" → `--video --text`\n- \"convert to code / UI to code / design to code\" → `--code`\n- extra requirements → `--prompt \"...\"`\n- unsure → ask once if image or video\n\n### 3. Credential safety\n- API keys are stored in plaintext config.json. Warn users not to use high-value keys on shared machines.\n- First-run privacy notice already discloses plaintext storage; do not repeat it every run.\n\n---\n\n## Limits\n\n- Doubao: 180W tokens per model per day, auto-fallback\n- GLM: free, auto-retry on failure (4.6V: 10 retries, 4.1V: 5 retries)\n- Image ≤ 15MB, Video ≤ 50MB\n\nFile v3.1.17:_meta.json\n\n{\n  \"ownerId\": \"kn75s192k98dtwr83bft4myd0582ytyg\",\n  \"slug\": \"bytedance-visual-recognition\",\n  \"version\": \"3.1.17\",\n  \"publishedAt\": 1786378491821\n}\n\nFile v3.1.17:proposal/PROPOSAL.md\n\n# 3.1.1 更新提案\n\n## 目标\n精简发布，修复安全审查问题，仅上传核心文件。\n\n## 变更\n- 收紧触发词列表，移除过于宽泛的 pattern\n- 批量处理改为仅复制媒体文件（不再复制整个目录树）\n- 剔除非核心文件（.env, Temp, .bak, .json 等）\n- 仅上传核心文件：doubao_vision_recognize.py, SKILL.md, skill-card.md\n\nFile v3.1.17:skill-card.md\n\n## Description:\n\nByteDance Visual Recognition uses Doubao-Seed and Zhipu GLM multimodal backends to recognize images and videos, generate text or code, handle batch processing and follow-up questions, retry or fall back across models, cache selected media locally, and maintain local recognition history and follow-up context.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[etmnb](https://clawhub.ai/user/etmnb)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and external users use this skill to send selected image or video files to Doubao or GLM vision APIs for recognition, summarization, extraction, or UI-to-code tasks. It also supports batch directory processing, follow-up questions based on the previous result, local usage status, history review, and optional manual IAM usage synchronization.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Selected images, videos, and prompts are sent to named cloud APIs for recognition.\n\nMitigation: Use the skill only with media and prompts that are appropriate to disclose to Volcengine Doubao or Zhipu GLM APIs.\n\nRisk: Selected media, recognition history, follow-up context, and plaintext API keys may persist locally in Temp/, vision_history.json, .last_response, and config.json.\n\nMitigation: Use low-value API keys on shared machines and manually delete Temp/, vision_history.json, .last_response, and config.json when local retention is not acceptable.\n\nRisk: GLM follow-up data may reuse message history that includes base64 media data and may persist until overwritten or removed.\n\nMitigation: Review or remove .last_response before follow-up use when the previous media or prompt should not be reused.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/etmnb/skills/bytedance-visual-recognition)\n- [Volcengine Doubao visual understanding documentation](https://www.volcengine.com/docs/82379/1569618)\n- [Zhipu GLM platform](https://open.bigmodel.cn)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, code, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown or plain text responses, code snippets, terminal status output, and local JSON configuration/history files]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Outputs depend on user-selected image or video mode, text or code mode, provider configuration, and follow-up context.]\n\n## Skill Version(s):\n\n3.1.17 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v3.1.16: 5 files, 19916 bytes\n\nFiles: _meta.json (148b), doubao_vision_recognize.py (52930b), proposal/PROPOSAL.md (376b), skill-card.md (2320b), SKILL.md (5772b)\n\nFile v3.1.16:SKILL.md\n\n---\nname: bytedance-visual-recognition\ndescription: >\n  ByteDance Visual Recognition — 调用豆包 Doubao-Seed + 智谱 GLM 双后端多模态模型识别图片和视频，\n  输出文字或代码。支持单文件识别、批量目录处理、追问（基于上次结果的对话），自动模型降级与重试。\n  本地缓存媒体文件（Temp/YYYYMMDD/），持久化识别历史（vision_history.json）和追问上下文（.last_response），\n  可选手动 IAM 控制台用量同步。首次运行自动生成 config.json 并显示隐私声明。\n  Supports both Chinese and English interactions.\nsummary: \"Doubao + GLM visual recognition — image/video to text/code, local caching, history, batch, follow-up, optional IAM sync\"\ntags:\n  vision: \"3.1.16\"\n  image-recognition: \"3.1.10\"\n  video-recognition: \"3.1.10\"\n  image-to-code: \"3.1.10\"\n  video-to-code: \"3.1.10\"\n  doubao: \"3.1.10\"\n  glm: \"3.1.10\"\n  image-recognition: \"3.1.8\"\n  video-recognition: \"3.1.8\"\n  image-to-code: \"3.1.8\"\n  video-to-code: \"3.1.10\"\n  doubao: \"3.1.10\"\n  glm: \"3.1.10\"\ntrigger_patterns:\n  - \"豆包识别\"\n  - \"豆包视觉识别\"\n  - \"bytedance visual recognition\"\n  - \"doubao recognize\"\nmetadata:\n  openclaw:\n    requires:\n      bins:\n        - python\n    permissions:\n      filesystem:\n        read: [\"config.json\", \"vision_history.json\", \".last_response\"]\n        write: [\"Temp/\", \"vision_history.json\", \".last_response\", \"config.json\"]\n      network:\n        - host: \"ark.cn-beijing.volces.com\"\n          purpose: \"Doubao vision model API\"\n        - host: \"open.bigmodel.cn\"\n          purpose: \"GLM vision model Chat Completions API\"\n        - host: \"open.volcengineapi.com\"\n          purpose: \"Optional manual IAM usage sync\"\n    emoji: \"🔍\"\n    homepage: https://www.volcengine.com/docs/82379/1569618\n    locales: [\"zh-CN\", \"en\"]\n---\n\n# ByteDance Visual Recognition — 豆包 + GLM 双后端视觉识别\n# ByteDance Visual Recognition — Doubao + GLM Dual-Backend Vision\n\nDoubao-Seed (6 models) + Zhipu GLM (2 free models). First run auto-generates `config.json`. Fill in one API Key to start. This skill supports commands and interactions in both Chinese and English. / 支持中英文交互。\n\n## ⚠️ Privacy & Data Notice / 隐私与数据声明\n\n- **Network / 网络**: Selected images/videos and prompts are base64-encoded and sent to Volcengine (Doubao) or Zhipu (GLM) cloud APIs.\n- **Local cache / 本地缓存**: Media files temporarily copied to `Temp/YYYYMMDD/`, default 7-day retention (`temp_retention_days` in config.json, range 1-3650).\n- **History / 使用记录**: Recognition history stored in `vision_history.json`, follow-up context in `.last_response`, auto-cleaned after 7 days.\n- **Credentials / 凭证**: API Keys stored in plaintext `config.json`. Do not use personal keys on shared machines.\n- **IAM sync / IAM 同步**: Only triggers via explicit `sync` command when IAM keys are configured. No automatic outbound calls.\n- **First run / 首次运行**: A privacy notice is displayed once. Continuing past it constitutes acknowledgment of data handling practices.\n\n## 🚀 Setup (pick one)\n\n### Doubao (Volcengine)\n\n1. 参与[协作奖励计划](https://console.volcengine.com/ark/region:cn-beijing/openManagement/rewardPlan)享免费额度，获取 API Key\n2. Create inference endpoints, pick from:\n\n| config key | model | priority |\n|--------|------|:---:|\n| `doubao_vision_21p_id` | Doubao-Seed-2.1-Pro | primary |\n| `doubao_vision_21t_id` | Doubao-Seed-2.1-Turbo | secondary |\n| `doubao_vision_20p_id` | Doubao-Seed-2.0-Pro | tertiary |\n| `doubao_vision_20c_id` | Doubao-Seed-2.0-Code | code-first |\n| `doubao_vision_20l_id` | Doubao-Seed-2.0-Lite | fallback |\n| `doubao_vision_20m_id` | Doubao-Seed-2.0-Mini | low-cost |\n\n3. Edit `config.json`, replace `\"\"` with actual endpoint IDs.\n\n### GLM (Zhipu, free)\n\n1. https://open.bigmodel.cn → get API Key\n2. Edit `config.json`: `\"zhipu_api_key\": \"your-key\"`\n\n### Provider filter\n\n`provider_mode` in `config.json`:\n- `0` = all (default)\n- `1` = Zhipu only\n- `2` = Doubao only\n\n### Test\n\n```bash\npython doubao_vision_recognize.py --help\npython doubao_vision_recognize.py status\n```\n\n---\n\n## ⚡ Commands\n\n| command | purpose | example |\n|------|------|------|\n| `rec <file> --image\\|--video --text\\|--code` | recognize | `rec a.jpg --image --text` |\n| `rec <dir> --batch --image\\|--video --text\\|--code` | batch | `rec ./img/ --batch --image --text` |\n| `ask --text\\|--code --prompt \"...\"` | follow-up | `ask --text -p \"details\"` |\n| `status` | usage stats | |\n| `sync` | console sync (manual) | |\n| `history` | 7-day history | |\n\n### Parameters\n\n| param | desc |\n|------|------|\n| `--image` | image input |\n| `--video` | video input |\n| `--text` | text output |\n| `--code` | code output |\n| `--prompt` / `-p` | extra instruction |\n| `--batch` | directory batch |\n\n---\n\n## 🚫 Behavior Rules\n\n### 1. Trigger only on listed patterns\n- Activate only when the user's message exactly matches one of the trigger_patterns listed in frontmatter.\n- Do NOT activate on loosely related text. If uncertain, ask before executing.\n\n### 2. Parameter inference\n- \"recognize/analyze image\" → `--image --text`\n- \"recognize/analyze video\" → `--video --text`\n- \"convert to code / UI to code / design to code\" → `--code`\n- extra requirements → `--prompt \"...\"`\n- unsure → ask once if image or video\n\n### 3. Credential safety\n- API keys are stored in plaintext config.json. Warn users not to use high-value keys on shared machines.\n- First-run privacy notice already discloses plaintext storage; do not repeat it every run.\n\n---\n\n## Limits\n\n- Doubao: 180W tokens per model per day, auto-fallback\n- GLM: free, auto-retry on failure (4.6V: 10 retries, 4.1V: 5 retries)\n- Image ≤ 15MB, Video ≤ 50MB\n\nFile v3.1.16:_meta.json\n\n{\n  \"ownerId\": \"kn75s192k98dtwr83bft4myd0582ytyg\",\n  \"slug\": \"bytedance-visual-recognition\",\n  \"version\": \"3.1.16\",\n  \"publishedAt\": 1786370040814\n}\n\nFile v3.1.16:proposal/PROPOSAL.md\n\n# 3.1.1 更新提案\n\n## 目标\n精简发布，修复安全审查问题，仅上传核心文件。\n\n## 变更\n- 收紧触发词列表，移除过于宽泛的 pattern\n- 批量处理改为仅复制媒体文件（不再复制整个目录树）\n- 剔除非核心文件（.env, Temp, .bak, .json 等）\n- 仅上传核心文件：doubao_vision_recognize.py, SKILL.md, skill-card.md\n\nFile v3.1.16:skill-card.md\n\n## Description:\n\nBytedance Visual Recognition uses Doubao-Seed and Zhipu GLM multimodal APIs to recognize images or videos, return text or code, process batches, support follow-up prompts, and maintain local cache and history files.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[etmnb](https://clawhub.ai/user/etmnb)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nExternal users, developers, and engineers use this skill to send selected image or video files to Doubao or GLM vision models for recognition, summarization, follow-up analysis, or code-oriented output.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Selected media and prompts are sent to Doubao or GLM cloud APIs.\n\nMitigation: Use the skill only with media and prompts acceptable for those providers, and avoid sensitive media unless that data handling is acceptable.\n\nRisk: API keys are stored in plaintext config.json.\n\nMitigation: Use low-privilege or dedicated keys, avoid shared machines for personal keys, and restrict local access to the skill directory.\n\nRisk: GLM follow-up can persist and resend original media through .last_response.\n\nMitigation: Clear .last_response, history, and cached media after sensitive sessions, or avoid follow-up mode for sensitive media.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/etmnb/skills/bytedance-visual-recognition)\n- [Volcengine Doubao vision documentation](https://www.volcengine.com/docs/82379/1569618)\n- [Zhipu GLM platform](https://open.bigmodel.cn)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, code, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown or terminal text with optional code snippets and local JSON configuration/history files]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May create config.json, vision_history.json, .last_response, and Temp/YYYYMMDD cached media files during use.]\n\n## Skill Version(s):\n\n3.1.16 (source: server release metadata; artifact _meta.json reports 3.1.1)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v3.1.15: 5 files, 19931 bytes\n\nFiles: _meta.json (148b), doubao_vision_recognize.py (52836b), proposal/PROPOSAL.md (376b), skill-card.md (2335b), SKILL.md (5772b)\n\nFile v3.1.15:SKILL.md\n\n---\nname: bytedance-visual-recognition\ndescription: >\n  ByteDance Visual Recognition — 调用豆包 Doubao-Seed + 智谱 GLM 双后端多模态模型识别图片和视频，\n  输出文字或代码。支持单文件识别、批量目录处理、追问（基于上次结果的对话），自动模型降级与重试。\n  本地缓存媒体文件（Temp/YYYYMMDD/），持久化识别历史（vision_history.json）和追问上下文（.last_response），\n  可选手动 IAM 控制台用量同步。首次运行自动生成 config.json 并显示隐私声明。\n  Supports both Chinese and English interactions.\nsummary: \"Doubao + GLM visual recognition — image/video to text/code, local caching, history, batch, follow-up, optional IAM sync\"\ntags:\n  vision: \"3.1.15\"\n  image-recognition: \"3.1.10\"\n  video-recognition: \"3.1.10\"\n  image-to-code: \"3.1.10\"\n  video-to-code: \"3.1.10\"\n  doubao: \"3.1.10\"\n  glm: \"3.1.10\"\n  image-recognition: \"3.1.8\"\n  video-recognition: \"3.1.8\"\n  image-to-code: \"3.1.8\"\n  video-to-code: \"3.1.10\"\n  doubao: \"3.1.10\"\n  glm: \"3.1.10\"\ntrigger_patterns:\n  - \"豆包识别\"\n  - \"豆包视觉识别\"\n  - \"bytedance visual recognition\"\n  - \"doubao recognize\"\nmetadata:\n  openclaw:\n    requires:\n      bins:\n        - python\n    permissions:\n      filesystem:\n        read: [\"config.json\", \"vision_history.json\", \".last_response\"]\n        write: [\"Temp/\", \"vision_history.json\", \".last_response\", \"config.json\"]\n      network:\n        - host: \"ark.cn-beijing.volces.com\"\n          purpose: \"Doubao vision model API\"\n        - host: \"open.bigmodel.cn\"\n          purpose: \"GLM vision model Chat Completions API\"\n        - host: \"open.volcengineapi.com\"\n          purpose: \"Optional manual IAM usage sync\"\n    emoji: \"🔍\"\n    homepage: https://www.volcengine.com/docs/82379/1569618\n    locales: [\"zh-CN\", \"en\"]\n---\n\n# ByteDance Visual Recognition — 豆包 + GLM 双后端视觉识别\n# ByteDance Visual Recognition — Doubao + GLM Dual-Backend Vision\n\nDoubao-Seed (6 models) + Zhipu GLM (2 free models). First run auto-generates `config.json`. Fill in one API Key to start. This skill supports commands and interactions in both Chinese and English. / 支持中英文交互。\n\n## ⚠️ Privacy & Data Notice / 隐私与数据声明\n\n- **Network / 网络**: Selected images/videos and prompts are base64-encoded and sent to Volcengine (Doubao) or Zhipu (GLM) cloud APIs.\n- **Local cache / 本地缓存**: Media files temporarily copied to `Temp/YYYYMMDD/`, default 7-day retention (`temp_retention_days` in config.json, range 1-3650).\n- **History / 使用记录**: Recognition history stored in `vision_history.json`, follow-up context in `.last_response`, auto-cleaned after 7 days.\n- **Credentials / 凭证**: API Keys stored in plaintext `config.json`. Do not use personal keys on shared machines.\n- **IAM sync / IAM 同步**: Only triggers via explicit `sync` command when IAM keys are configured. No automatic outbound calls.\n- **First run / 首次运行**: A privacy notice is displayed once. Continuing past it constitutes acknowledgment of data handling practices.\n\n## 🚀 Setup (pick one)\n\n### Doubao (Volcengine)\n\n1. 参与[协作奖励计划](https://console.volcengine.com/ark/region:cn-beijing/openManagement/rewardPlan)享免费额度，获取 API Key\n2. Create inference endpoints, pick from:\n\n| config key | model | priority |\n|--------|------|:---:|\n| `doubao_vision_21p_id` | Doubao-Seed-2.1-Pro | primary |\n| `doubao_vision_21t_id` | Doubao-Seed-2.1-Turbo | secondary |\n| `doubao_vision_20p_id` | Doubao-Seed-2.0-Pro | tertiary |\n| `doubao_vision_20c_id` | Doubao-Seed-2.0-Code | code-first |\n| `doubao_vision_20l_id` | Doubao-Seed-2.0-Lite | fallback |\n| `doubao_vision_20m_id` | Doubao-Seed-2.0-Mini | low-cost |\n\n3. Edit `config.json`, replace `\"\"` with actual endpoint IDs.\n\n### GLM (Zhipu, free)\n\n1. https://open.bigmodel.cn → get API Key\n2. Edit `config.json`: `\"zhipu_api_key\": \"your-key\"`\n\n### Provider filter\n\n`provider_mode` in `config.json`:\n- `0` = all (default)\n- `1` = Zhipu only\n- `2` = Doubao only\n\n### Test\n\n```bash\npython doubao_vision_recognize.py --help\npython doubao_vision_recognize.py status\n```\n\n---\n\n## ⚡ Commands\n\n| command | purpose | example |\n|------|------|------|\n| `rec <file> --image\\|--video --text\\|--code` | recognize | `rec a.jpg --image --text` |\n| `rec <dir> --batch --image\\|--video --text\\|--code` | batch | `rec ./img/ --batch --image --text` |\n| `ask --text\\|--code --prompt \"...\"` | follow-up | `ask --text -p \"details\"` |\n| `status` | usage stats | |\n| `sync` | console sync (manual) | |\n| `history` | 7-day history | |\n\n### Parameters\n\n| param | desc |\n|------|------|\n| `--image` | image input |\n| `--video` | video input |\n| `--text` | text output |\n| `--code` | code output |\n| `--prompt` / `-p` | extra instruction |\n| `--batch` | directory batch |\n\n---\n\n## 🚫 Behavior Rules\n\n### 1. Trigger only on listed patterns\n- Activate only when the user's message exactly matches one of the trigger_patterns listed in frontmatter.\n- Do NOT activate on loosely related text. If uncertain, ask before executing.\n\n### 2. Parameter inference\n- \"recognize/analyze image\" → `--image --text`\n- \"recognize/analyze video\" → `--video --text`\n- \"convert to code / UI to code / design to code\" → `--code`\n- extra requirements → `--prompt \"...\"`\n- unsure → ask once if image or video\n\n### 3. Credential safety\n- API keys are stored in plaintext config.json. Warn users not to use high-value keys on shared machines.\n- First-run privacy notice already discloses plaintext storage; do not repeat it every run.\n\n---\n\n## Limits\n\n- Doubao: 180W tokens per model per day, auto-fallback\n- GLM: free, auto-retry on failure (4.6V: 10 retries, 4.1V: 5 retries)\n- Image ≤ 15MB, Video ≤ 50MB\n\nFile v3.1.15:_meta.json\n\n{\n  \"ownerId\": \"kn75s192k98dtwr83bft4myd0582ytyg\",\n  \"slug\": \"bytedance-visual-recognition\",\n  \"version\": \"3.1.15\",\n  \"publishedAt\": 1786368781211\n}\n\nFile v3.1.15:proposal/PROPOSAL.md\n\n# 3.1.1 更新提案\n\n## 目标\n精简发布，修复安全审查问题，仅上传核心文件。\n\n## 变更\n- 收紧触发词列表，移除过于宽泛的 pattern\n- 批量处理改为仅复制媒体文件（不再复制整个目录树）\n- 剔除非核心文件（.env, Temp, .bak, .json 等）\n- 仅上传核心文件：doubao_vision_recognize.py, SKILL.md, skill-card.md\n\nFile v3.1.15:skill-card.md\n\n## Description:\n\nBytedance Visual Recognition sends selected images or videos to Doubao and Zhipu GLM vision APIs to generate text descriptions or code, with batch processing, follow-up questions, local caching, history, fallback, retries, and optional manual IAM usage sync.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[etmnb](https://clawhub.ai/user/etmnb)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and external users can use this skill to analyze images or videos through Doubao and Zhipu GLM backends, produce textual recognition results, generate UI/code from visual input, process batches, and ask follow-up questions against the previous result.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Provider isolation may not hold: selected media may be sent to Doubao even when Zhipu-only mode is configured.\n\nMitigation: Review the implementation before installing where provider isolation matters; until fixed, do not rely on provider_mode=1 to keep media away from Doubao.\n\nRisk: API keys, selected media cache, recognition history, and follow-up context can remain in local plaintext files.\n\nMitigation: Avoid high-value keys on shared machines and clear Temp/, vision_history.json, and .last_response after processing sensitive media.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/etmnb/skills/bytedance-visual-recognition)\n- [Volcengine Doubao documentation](https://www.volcengine.com/docs/82379/1569618)\n- [Zhipu GLM platform](https://open.bigmodel.cn)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, code, shell commands, configuration]\n\n**Output Format:** [Command-line text, Markdown-formatted recognition results, code snippets, and local JSON configuration or history files]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Outputs may depend on cloud API responses; selected media is cached locally by default and history/context are persisted in local files.]\n\n## Skill Version(s):\n\n3.1.15 (source: server release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v3.1.14: 5 files, 19912 bytes\n\nFiles: _meta.json (148b), doubao_vision_recognize.py (52937b), proposal/PROPOSAL.md (376b), skill-card.md (2380b), SKILL.md (5772b)\n\nFile v3.1.14:SKILL.md\n\n---\nname: bytedance-visual-recognition\ndescription: >\n  ByteDance Visual Recognition — 调用豆包 Doubao-Seed + 智谱 GLM 双后端多模态模型识别图片和视频，\n  输出文字或代码。支持单文件识别、批量目录处理、追问（基于上次结果的对话），自动模型降级与重试。\n  本地缓存媒体文件（Temp/YYYYMMDD/），持久化识别历史（vision_history.json）和追问上下文（.last_response），\n  可选手动 IAM 控制台用量同步。首次运行自动生成 config.json 并显示隐私声明。\n  Supports both Chinese and English interactions.\nsummary: \"Doubao + GLM visual recognition — image/video to text/code, local caching, history, batch, follow-up, optional IAM sync\"\ntags:\n  vision: \"3.1.14\"\n  image-recognition: \"3.1.10\"\n  video-recognition: \"3.1.10\"\n  image-to-code: \"3.1.10\"\n  video-to-code: \"3.1.10\"\n  doubao: \"3.1.10\"\n  glm: \"3.1.10\"\n  image-recognition: \"3.1.8\"\n  video-recognition: \"3.1.8\"\n  image-to-code: \"3.1.8\"\n  video-to-code: \"3.1.10\"\n  doubao: \"3.1.10\"\n  glm: \"3.1.10\"\ntrigger_patterns:\n  - \"豆包识别\"\n  - \"豆包视觉识别\"\n  - \"bytedance visual recognition\"\n  - \"doubao recognize\"\nmetadata:\n  openclaw:\n    requires:\n      bins:\n        - python\n    permissions:\n      filesystem:\n        read: [\"config.json\", \"vision_history.json\", \".last_response\"]\n        write: [\"Temp/\", \"vision_history.json\", \".last_response\", \"config.json\"]\n      network:\n        - host: \"ark.cn-beijing.volces.com\"\n          purpose: \"Doubao vision model API\"\n        - host: \"open.bigmodel.cn\"\n          purpose: \"GLM vision model Chat Completions API\"\n        - host: \"open.volcengineapi.com\"\n          purpose: \"Optional manual IAM usage sync\"\n    emoji: \"🔍\"\n    homepage: https://www.volcengine.com/docs/82379/1569618\n    locales: [\"zh-CN\", \"en\"]\n---\n\n# ByteDance Visual Recognition — 豆包 + GLM 双后端视觉识别\n# ByteDance Visual Recognition — Doubao + GLM Dual-Backend Vision\n\nDoubao-Seed (6 models) + Zhipu GLM (2 free models). First run auto-generates `config.json`. Fill in one API Key to start. This skill supports commands and interactions in both Chinese and English. / 支持中英文交互。\n\n## ⚠️ Privacy & Data Notice / 隐私与数据声明\n\n- **Network / 网络**: Selected images/videos and prompts are base64-encoded and sent to Volcengine (Doubao) or Zhipu (GLM) cloud APIs.\n- **Local cache / 本地缓存**: Media files temporarily copied to `Temp/YYYYMMDD/`, default 7-day retention (`temp_retention_days` in config.json, range 1-3650).\n- **History / 使用记录**: Recognition history stored in `vision_history.json`, follow-up context in `.last_response`, auto-cleaned after 7 days.\n- **Credentials / 凭证**: API Keys stored in plaintext `config.json`. Do not use personal keys on shared machines.\n- **IAM sync / IAM 同步**: Only triggers via explicit `sync` command when IAM keys are configured. No automatic outbound calls.\n- **First run / 首次运行**: A privacy notice is displayed once. Continuing past it constitutes acknowledgment of data handling practices.\n\n## 🚀 Setup (pick one)\n\n### Doubao (Volcengine)\n\n1. 参与[协作奖励计划](https://console.volcengine.com/ark/region:cn-beijing/openManagement/rewardPlan)享免费额度，获取 API Key\n2. Create inference endpoints, pick from:\n\n| config key | model | priority |\n|--------|------|:---:|\n| `doubao_vision_21p_id` | Doubao-Seed-2.1-Pro | primary |\n| `doubao_vision_21t_id` | Doubao-Seed-2.1-Turbo | secondary |\n| `doubao_vision_20p_id` | Doubao-Seed-2.0-Pro | tertiary |\n| `doubao_vision_20c_id` | Doubao-Seed-2.0-Code | code-first |\n| `doubao_vision_20l_id` | Doubao-Seed-2.0-Lite | fallback |\n| `doubao_vision_20m_id` | Doubao-Seed-2.0-Mini | low-cost |\n\n3. Edit `config.json`, replace `\"\"` with actual endpoint IDs.\n\n### GLM (Zhipu, free)\n\n1. https://open.bigmodel.cn → get API Key\n2. Edit `config.json`: `\"zhipu_api_key\": \"your-key\"`\n\n### Provider filter\n\n`provider_mode` in `config.json`:\n- `0` = all (default)\n- `1` = Zhipu only\n- `2` = Doubao only\n\n### Test\n\n```bash\npython doubao_vision_recognize.py --help\npython doubao_vision_recognize.py status\n```\n\n---\n\n## ⚡ Commands\n\n| command | purpose | example |\n|------|------|------|\n| `rec <file> --image\\|--video --text\\|--code` | recognize | `rec a.jpg --image --text` |\n| `rec <dir> --batch --image\\|--video --text\\|--code` | batch | `rec ./img/ --batch --image --text` |\n| `ask --text\\|--code --prompt \"...\"` | follow-up | `ask --text -p \"details\"` |\n| `status` | usage stats | |\n| `sync` | console sync (manual) | |\n| `history` | 7-day history | |\n\n### Parameters\n\n| param | desc |\n|------|------|\n| `--image` | image input |\n| `--video` | video input |\n| `--text` | text output |\n| `--code` | code output |\n| `--prompt` / `-p` | extra instruction |\n| `--batch` | directory batch |\n\n---\n\n## 🚫 Behavior Rules\n\n### 1. Trigger only on listed patterns\n- Activate only when the user's message exactly matches one of the trigger_patterns listed in frontmatter.\n- Do NOT activate on loosely related text. If uncertain, ask before executing.\n\n### 2. Parameter inference\n- \"recognize/analyze image\" → `--image --text`\n- \"recognize/analyze video\" → `--video --text`\n- \"convert to code / UI to code / design to code\" → `--code`\n- extra requirements → `--prompt \"...\"`\n- unsure → ask once if image or video\n\n### 3. Credential safety\n- API keys are stored in plaintext config.json. Warn users not to use high-value keys on shared machines.\n- First-run privacy notice already discloses plaintext storage; do not repeat it every run.\n\n---\n\n## Limits\n\n- Doubao: 180W tokens per model per day, auto-fallback\n- GLM: free, auto-retry on failure (4.6V: 10 retries, 4.1V: 5 retries)\n- Image ≤ 15MB, Video ≤ 50MB\n\nFile v3.1.14:_meta.json\n\n{\n  \"ownerId\": \"kn75s192k98dtwr83bft4myd0582ytyg\",\n  \"slug\": \"bytedance-visual-recognition\",\n  \"version\": \"3.1.14\",\n  \"publishedAt\": 1786367673813\n}\n\nFile v3.1.14:proposal/PROPOSAL.md\n\n# 3.1.1 更新提案\n\n## 目标\n精简发布，修复安全审查问题，仅上传核心文件。\n\n## 变更\n- 收紧触发词列表，移除过于宽泛的 pattern\n- 批量处理改为仅复制媒体文件（不再复制整个目录树）\n- 剔除非核心文件（.env, Temp, .bak, .json 等）\n- 仅上传核心文件：doubao_vision_recognize.py, SKILL.md, skill-card.md\n\nFile v3.1.14:skill-card.md\n\n## Description:\n\nRecognizes images and videos with Doubao and GLM vision models, returning text or generated code with batch processing, follow-up prompts, local cache/history, and optional manual IAM usage sync.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[etmnb](https://clawhub.ai/user/etmnb)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and external users use this skill to analyze selected images or videos, extract readable descriptions, or generate code from visual UI and design inputs through configured Doubao or Zhipu model APIs.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Selected images, videos, and prompts are sent to Doubao or Zhipu cloud APIs.\n\nMitigation: Use only media and prompts that are appropriate for those services and avoid sensitive data unless permitted by policy.\n\nRisk: API keys are stored in plaintext config.json.\n\nMitigation: Use low-privilege keys where possible, avoid shared machines, and rotate keys if config.json is exposed.\n\nRisk: Local cache, history, and follow-up context files retain media or recognition context.\n\nMitigation: Review Temp/, vision_history.json, and .last_response retention and delete local records when needed.\n\nRisk: Optional IAM usage synchronization uses configured IAM credentials when run intentionally.\n\nMitigation: Run sync only when intended and review credential scope before use.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/etmnb/skills/bytedance-visual-recognition)\n- [Volcengine Doubao vision documentation](https://www.volcengine.com/docs/82379/1569618)\n- [Zhipu Open Platform](https://open.bigmodel.cn)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, Code, Shell commands, Configuration]\n\n**Output Format:** [Markdown or plain text responses with optional code blocks and shell commands]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May create local config, cache, history, and follow-up context files while using configured cloud APIs.]\n\n## Skill Version(s):\n\n3.1.14 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v3.1.13: 5 files, 19858 bytes\n\nFiles: _meta.json (148b), doubao_vision_recognize.py (52885b), proposal/PROPOSAL.md (376b), skill-card.md (2259b), SKILL.md (5772b)\n\nFile v3.1.13:SKILL.md\n\n---\nname: bytedance-visual-recognition\ndescription: >\n  ByteDance Visual Recognition — 调用豆包 Doubao-Seed + 智谱 GLM 双后端多模态模型识别图片和视频，\n  输出文字或代码。支持单文件识别、批量目录处理、追问（基于上次结果的对话），自动模型降级与重试。\n  本地缓存媒体文件（Temp/YYYYMMDD/），持久化识别历史（vision_history.json）和追问上下文（.last_response），\n  可选手动 IAM 控制台用量同步。首次运行自动生成 config.json 并显示隐私声明。\n  Supports both Chinese and English interactions.\nsummary: \"Doubao + GLM visual recognition — image/video to text/code, local caching, history, batch, follow-up, optional IAM sync\"\ntags:\n  vision: \"3.1.13\"\n  image-recognition: \"3.1.10\"\n  video-recognition: \"3.1.10\"\n  image-to-code: \"3.1.10\"\n  video-to-code: \"3.1.10\"\n  doubao: \"3.1.10\"\n  glm: \"3.1.10\"\n  image-recognition: \"3.1.8\"\n  video-recognition: \"3.1.8\"\n  image-to-code: \"3.1.8\"\n  video-to-code: \"3.1.10\"\n  doubao: \"3.1.10\"\n  glm: \"3.1.10\"\ntrigger_patterns:\n  - \"豆包识别\"\n  - \"豆包视觉识别\"\n  - \"bytedance visual recognition\"\n  - \"doubao recognize\"\nmetadata:\n  openclaw:\n    requires:\n      bins:\n        - python\n    permissions:\n      filesystem:\n        read: [\"config.json\", \"vision_history.json\", \".last_response\"]\n        write: [\"Temp/\", \"vision_history.json\", \".last_response\", \"config.json\"]\n      network:\n        - host: \"ark.cn-beijing.volces.com\"\n          purpose: \"Doubao vision model API\"\n        - host: \"open.bigmodel.cn\"\n          purpose: \"GLM vision model Chat Completions API\"\n        - host: \"open.volcengineapi.com\"\n          purpose: \"Optional manual IAM usage sync\"\n    emoji: \"🔍\"\n    homepage: https://www.volcengine.com/docs/82379/1569618\n    locales: [\"zh-CN\", \"en\"]\n---\n\n# ByteDance Visual Recognition — 豆包 + GLM 双后端视觉识别\n# ByteDance Visual Recognition — Doubao + GLM Dual-Backend Vision\n\nDoubao-Seed (6 models) + Zhipu GLM (2 free models). First run auto-generates `config.json`. Fill in one API Key to start. This skill supports commands and interactions in both Chinese and English. / 支持中英文交互。\n\n## ⚠️ Privacy & Data Notice / 隐私与数据声明\n\n- **Network / 网络**: Selected images/videos and prompts are base64-encoded and sent to Volcengine (Doubao) or Zhipu (GLM) cloud APIs.\n- **Local cache / 本地缓存**: Media files temporarily copied to `Temp/YYYYMMDD/`, default 7-day retention (`temp_retention_days` in config.json, range 1-3650).\n- **History / 使用记录**: Recognition history stored in `vision_history.json`, follow-up context in `.last_response`, auto-cleaned after 7 days.\n- **Credentials / 凭证**: API Keys stored in plaintext `config.json`. Do not use personal keys on shared machines.\n- **IAM sync / IAM 同步**: Only triggers via explicit `sync` command when IAM keys are configured. No automatic outbound calls.\n- **First run / 首次运行**: A privacy notice is displayed once. Continuing past it constitutes acknowledgment of data handling practices.\n\n## 🚀 Setup (pick one)\n\n### Doubao (Volcengine)\n\n1. 参与[协作奖励计划](https://console.volcengine.com/ark/region:cn-beijing/openManagement/rewardPlan)享免费额度，获取 API Key\n2. Create inference endpoints, pick from:\n\n| config key | model | priority |\n|--------|------|:---:|\n| `doubao_vision_21p_id` | Doubao-Seed-2.1-Pro | primary |\n| `doubao_vision_21t_id` | Doubao-Seed-2.1-Turbo | secondary |\n| `doubao_vision_20p_id` | Doubao-Seed-2.0-Pro | tertiary |\n| `doubao_vision_20c_id` | Doubao-Seed-2.0-Code | code-first |\n| `doubao_vision_20l_id` | Doubao-Seed-2.0-Lite | fallback |\n| `doubao_vision_20m_id` | Doubao-Seed-2.0-Mini | low-cost |\n\n3. Edit `config.json`, replace `\"\"` with actual endpoint IDs.\n\n### GLM (Zhipu, free)\n\n1. https://open.bigmodel.cn → get API Key\n2. Edit `config.json`: `\"zhipu_api_key\": \"your-key\"`\n\n### Provider filter\n\n`provider_mode` in `config.json`:\n- `0` = all (default)\n- `1` = Zhipu only\n- `2` = Doubao only\n\n### Test\n\n```bash\npython doubao_vision_recognize.py --help\npython doubao_vision_recognize.py status\n```\n\n---\n\n## ⚡ Commands\n\n| command | purpose | example |\n|------|------|------|\n| `rec <file> --image\\|--video --text\\|--code` | recognize | `rec a.jpg --image --text` |\n| `rec <dir> --batch --image\\|--video --text\\|--code` | batch | `rec ./img/ --batch --image --text` |\n| `ask --text\\|--code --prompt \"...\"` | follow-up | `ask --text -p \"details\"` |\n| `status` | usage stats | |\n| `sync` | console sync (manual) | |\n| `history` | 7-day history | |\n\n### Parameters\n\n| param | desc |\n|------|------|\n| `--image` | image input |\n| `--video` | video input |\n| `--text` | text output |\n| `--code` | code output |\n| `--prompt` / `-p` | extra instruction |\n| `--batch` | directory batch |\n\n---\n\n## 🚫 Behavior Rules\n\n### 1. Trigger only on listed patterns\n- Activate only when the user's message exactly matches one of the trigger_patterns listed in frontmatter.\n- Do NOT activate on loosely related text. If uncertain, ask before executing.\n\n### 2. Parameter inference\n- \"recognize/analyze image\" → `--image --text`\n- \"recognize/analyze video\" → `--video --text`\n- \"convert to code / UI to code / design to code\" → `--code`\n- extra requirements → `--prompt \"...\"`\n- unsure → ask once if image or video\n\n### 3. Credential safety\n- API keys are stored in plaintext config.json. Warn users not to use high-value keys on shared machines.\n- First-run privacy notice already discloses plaintext storage; do not repeat it every run.\n\n---\n\n## Limits\n\n- Doubao: 180W tokens per model per day, auto-fallback\n- GLM: free, auto-retry on failure (4.6V: 10 retries, 4.1V: 5 retries)\n- Image ≤ 15MB, Video ≤ 50MB\n\nFile v3.1.13:_meta.json\n\n{\n  \"ownerId\": \"kn75s192k98dtwr83bft4myd0582ytyg\",\n  \"slug\": \"bytedance-visual-recognition\",\n  \"version\": \"3.1.13\",\n  \"publishedAt\": 1786366663939\n}\n\nFile v3.1.13:proposal/PROPOSAL.md\n\n# 3.1.1 更新提案\n\n## 目标\n精简发布，修复安全审查问题，仅上传核心文件。\n\n## 变更\n- 收紧触发词列表，移除过于宽泛的 pattern\n- 批量处理改为仅复制媒体文件（不再复制整个目录树）\n- 剔除非核心文件（.env, Temp, .bak, .json 等）\n- 仅上传核心文件：doubao_vision_recognize.py, SKILL.md, skill-card.md\n\nFile v3.1.13:skill-card.md\n\n## Description:\n\nBytedance Visual Recognition lets agents analyze images or videos with Doubao and Zhipu GLM backends and return natural-language descriptions, extracted content, or generated code while supporting batch runs and follow-up questions.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[etmnb](https://clawhub.ai/user/etmnb)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and end users use this skill to send selected images or videos to configured Doubao or Zhipu cloud vision APIs for recognition, summary, OCR-like extraction, UI-to-code generation, and follow-up analysis.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Selected images, videos, and prompts are sent to Doubao or Zhipu cloud APIs.\n\nMitigation: Install and use the skill only when sending the selected media and prompts to those cloud APIs is acceptable.\n\nRisk: API keys are stored in plaintext config.json.\n\nMitigation: Avoid high-value API keys on shared machines and rotate keys if the local configuration may have been exposed.\n\nRisk: Local Temp, vision_history.json, and .last_response files may retain sensitive media, prompts, outputs, or context.\n\nMitigation: Periodically review or delete these local files when working with sensitive media or results.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/etmnb/skills/bytedance-visual-recognition)\n- [Volcengine Doubao documentation](https://www.volcengine.com/docs/82379/1569618)\n- [Zhipu Open Platform](https://open.bigmodel.cn)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, code, shell commands, configuration, guidance]\n\n**Output Format:** [Console text and Markdown-like responses with optional code blocks and configuration instructions]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Can write local config, cache, history, and follow-up context files during use.]\n\n## Skill Version(s):\n\n3.1.13 (source: server release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v3.1.12: 5 files, 19833 bytes\n\nFiles: _meta.json (148b), doubao_vision_recognize.py (52885b), proposal/PROPOSAL.md (376b), skill-card.md (2582b), SKILL.md (5427b)\n\nFile v3.1.12:SKILL.md\n\n---\nname: bytedance-visual-recognition\ndescription: >\n  ByteDance Visual Recognition — 调用豆包 Doubao-Seed + 智谱 GLM 双后端多模态模型识别图片和视频，\n  输出文字或代码。支持单文件识别、批量目录处理、追问（基于上次结果的对话），自动模型降级与重试。\n  本地缓存媒体文件（Temp/YYYYMMDD/），持久化识别历史（vision_history.json）和追问上下文（.last_response），\n  可选手动 IAM 控制台用量同步。首次运行自动生成 config.json 并显示隐私声明。\n  Supports both Chinese and English interactions.\nsummary: \"Doubao + GLM visual recognition — image/video to text/code, local caching, history, batch, follow-up, optional IAM sync\"\ntags:\n  vision: \"3.1.12\"\n  image-recognition: \"3.1.10\"\n  video-recognition: \"3.1.10\"\n  image-to-code: \"3.1.10\"\n  video-to-code: \"3.1.10\"\n  doubao: \"3.1.10\"\n  glm: \"3.1.10\"\n  image-recognition: \"3.1.8\"\n  video-recognition: \"3.1.8\"\n  image-to-code: \"3.1.8\"\n  video-to-code: \"3.1.10\"\n  doubao: \"3.1.10\"\n  glm: \"3.1.10\"\ntrigger_patterns:\n  - \"豆包识别\"\n  - \"豆包视觉识别\"\n  - \"bytedance visual recognition\"\n  - \"doubao recognize\"\nmetadata:\n  openclaw:\n    requires:\n      bins:\n        - python\n    permissions:\n      filesystem:\n        read: [\"config.json\", \"vision_history.json\", \".last_response\"]\n        write: [\"Temp/\", \"vision_history.json\", \".last_response\", \"config.json\"]\n      network:\n        - host: \"ark.cn-beijing.volces.com\"\n          purpose: \"Doubao vision model API\"\n        - host: \"open.bigmodel.cn\"\n          purpose: \"GLM vision model Chat Completions API\"\n        - host: \"open.volcengineapi.com\"\n          purpose: \"Optional manual IAM usage sync\"\n    emoji: \"🔍\"\n    homepage: https://www.volcengine.com/docs/82379/1569618\n    locales: [\"zh-CN\", \"en\"]\n---\n\n# ByteDance Visual Recognition — 豆包 + GLM 双后端视觉识别\n# ByteDance Visual Recognition — Doubao + GLM Dual-Backend Vision\n\nDoubao-Seed (6 models) + Zhipu GLM (2 free models). First run auto-generates `config.json`. Fill in one API Key to start. This skill supports commands and interactions in both Chinese and English. / 支持中英文交互。\n\n## ⚠️ Privacy & Data Notice / 隐私与数据声明\n\n- **Network / 网络**: Selected images/videos and prompts are base64-encoded and sent to Volcengine (Doubao) or Zhipu (GLM) cloud APIs.\n- **Local cache / 本地缓存**: Media files temporarily copied to `Temp/YYYYMMDD/`, default 7-day retention (`temp_retention_days` in config.json, range 1-3650).\n- **History / 使用记录**: Recognition history stored in `vision_history.json`, follow-up context in `.last_response`, auto-cleaned after 7 days.\n- **Credentials / 凭证**: API Keys stored in plaintext `config.json`. Do not use personal keys on shared machines.\n- **IAM sync / IAM 同步**: Only triggers via explicit `sync` command when IAM keys are configured. No automatic outbound calls.\n- **First run / 首次运行**: A privacy notice is displayed once. Continuing past it constitutes acknowledgment of data handling practices.\n\n## 🚀 Setup (pick one)\n\n### Doubao (Volcengine)\n\n1. 参与[协作奖励计划](https://console.volcengine.com/ark/region:cn-beijing/openManagement/rewardPlan)享免费额度，获取 API Key\n2. Create inference endpoints, pick from:\n\n| config key | model | priority |\n|--------|------|:---:|\n| `doubao_vision_21p_id` | Doubao-Seed-2.1-Pro | primary |\n| `doubao_vision_21t_id` | Doubao-Seed-2.1-Turbo | secondary |\n| `doubao_vision_20p_id` | Doubao-Seed-2.0-Pro | tertiary |\n| `doubao_vision_20c_id` | Doubao-Seed-2.0-Code | code-first |\n| `doubao_vision_20l_id` | Doubao-Seed-2.0-Lite | fallback |\n| `doubao_vision_20m_id` | Doubao-Seed-2.0-Mini | low-cost |\n\n3. Edit `config.json`, replace `\"\"` with actual endpoint IDs.\n\n### GLM (Zhipu, free)\n\n1. https://open.bigmodel.cn → get API Key\n2. Edit `config.json`: `\"zhipu_api_key\": \"your-key\"`\n\n### Provider filter\n\n`provider_mode` in `config.json`:\n- `0` = all (default)\n- `1` = Zhipu only\n- `2` = Doubao only\n\n### Test\n\n```bash\npython doubao_vision_recognize.py --help\npython doubao_vision_recognize.py status\n```\n\n---\n\n## ⚡ Commands\n\n| command | purpose | example |\n|------|------|------|\n| `rec <file> --image\\|--video --text\\|--code` | recognize | `rec a.jpg --image --text` |\n| `rec <dir> --batch --image\\|--video --text\\|--code` | batch | `rec ./img/ --batch --image --text` |\n| `ask --text\\|--code --prompt \"...\"` | follow-up | `ask --text -p \"details\"` |\n| `status` | usage stats | |\n| `sync` | console sync (manual) | |\n| `history` | 7-day history | |\n\n### Parameters\n\n| param | desc |\n|------|------|\n| `--image` | image input |\n| `--video` | video input |\n| `--text` | text output |\n| `--code` | code output |\n| `--prompt` / `-p` | extra instruction |\n| `--batch` | directory batch |\n\n---\n\n## 🚫 Behavior Rules\n\n### 1. Execute on trigger\n- When the skill is explicitly invoked via trigger phrases, proceed directly.\n\n### 2. Parameter inference\n- \"recognize/analyze image\" → `--image --text`\n- \"recognize/analyze video\" → `--video --text`\n- \"convert to code / UI to code / design to code\" → `--code`\n- extra requirements → `--prompt \"...\"`\n- unsure → ask once if image or video\n\n---\n\n## Limits\n\n- Doubao: 180W tokens per model per day, auto-fallback\n- GLM: free, auto-retry on failure (4.6V: 10 retries, 4.1V: 5 retries)\n- Image ≤ 15MB, Video ≤ 50MB\n\nFile v3.1.12:_meta.json\n\n{\n  \"ownerId\": \"kn75s192k98dtwr83bft4myd0582ytyg\",\n  \"slug\": \"bytedance-visual-recognition\",\n  \"version\": \"3.1.12\",\n  \"publishedAt\": 1786365568676\n}\n\nFile v3.1.12:proposal/PROPOSAL.md\n\n# 3.1.1 更新提案\n\n## 目标\n精简发布，修复安全审查问题，仅上传核心文件。\n\n## 变更\n- 收紧触发词列表，移除过于宽泛的 pattern\n- 批量处理改为仅复制媒体文件（不再复制整个目录树）\n- 剔除非核心文件（.env, Temp, .bak, .json 等）\n- 仅上传核心文件：doubao_vision_recognize.py, SKILL.md, skill-card.md\n\nFile v3.1.12:skill-card.md\n\n## Description:\n\nBytedance Visual Recognition uses Doubao-Seed and Zhipu GLM multimodal APIs to analyze images or videos and return text descriptions or generated code, with batch processing, follow-up questions, model fallback, local media caching, and usage history.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[etmnb](https://clawhub.ai/user/etmnb)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and external users use this skill to recognize visual content, extract structured text, or generate UI code from selected images and videos through Doubao or Zhipu cloud model APIs. It is suited for single-file analysis, batch media processing, and follow-up questions based on the previous recognition result.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Selected images, videos, and prompts are sent to Doubao or Zhipu cloud APIs.\n\nMitigation: Use the skill only with media and prompts approved for those providers, and avoid sending sensitive or regulated content unless appropriate data handling approvals are in place.\n\nRisk: API keys are stored in plaintext config.json.\n\nMitigation: Use scoped keys, restrict local file permissions, and avoid storing personal or high-value credentials on shared, synced, or backed-up machines.\n\nRisk: Media cache, recognition history, and follow-up context are written locally.\n\nMitigation: Clear Temp, vision_history.json, and .last_response after sensitive work, and reduce the configured cache retention period where needed.\n\n## Reference(s):\n\n- [ClawHub Skill Page](https://clawhub.ai/etmnb/skills/bytedance-visual-recognition)\n- [Volcengine Doubao Documentation](https://www.volcengine.com/docs/82379/1569618)\n- [Zhipu Open Platform](https://open.bigmodel.cn)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, code, shell commands, configuration]\n\n**Output Format:** [Terminal text with Markdown-formatted recognition results or generated code, plus local JSON configuration and history files.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Outputs may include token usage, model selection status, batch summaries, and follow-up context; media cache and history are retained locally by default.]\n\n## Skill Version(s):\n\n3.1.12 (source: server release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.","readmeExcerpt":"Skill: bytedance-visual-recognition Owner: etmnb Summary: Multimodal visual recognition via Doubao-Seed (6 models) + Zhipu GLM (2 free models). Image/video to text/code, auto-fallback, batch directory processing, follow-up conversations, local media caching (Temp/), persistent history (vision_history.json), auto IAM console sync. First-run privacy notice, plaintext config.json, cross-platform. Tags: latest:5.0.1 Vers","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"python doubao_vision_recognize.py --help\npython doubao_vision_recognize.py status"},{"language":"bash","snippet":"python doubao_vision_recognize.py --help\npython doubao_vision_recognize.py status"},{"language":"bash","snippet":"python doubao_vision_recognize.py --help\npython doubao_vision_recognize.py status"},{"language":"bash","snippet":"python doubao_vision_recognize.py --help\npython doubao_vision_recognize.py status"},{"language":"bash","snippet":"python doubao_vision_recognize.py --help\npython doubao_vision_recognize.py status"},{"language":"bash","snippet":"python doubao_vision_recognize.py --help\npython doubao_vision_recognize.py status"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: bytedance-visual-recognition\ndescription: >\n  Multimodal visual recognition via Doubao-Seed (6 models) + Zhipu GLM (2 free models).\n  Image/video to text/code, auto-fallback, batch directory processing, follow-up conversations,\n  local media caching (Temp/), persistent history (vision_history.json), auto IAM console sync.\n  First-run privacy notice, plaintext config.json, cross-platform.\nsummary: \"Doubao-Seed + GLM visual recognition — image/video to text/code with auto-fallback, batch, follow-up, IAM sync\"\ntags:\n  vision: \"5.0.1\"\n  image-recognition: \"5.0.1\"\n  video-recognition: \"5.0.1\"\n  image-to-code: \"5.0.1\"\n  video-to-code: \"5.0.1\"\n  doubao: \"5.0.1\"\n  glm: \"5.0.1\"\ntrigger_patterns:\n  - \"豆包识别\"\n  - \"豆包视觉识别\"\n  - \"bytedance visual recognition\"\n  - \"doubao recognize\"\nmetadata:\n  openclaw:\n    requires:\n      bins:\n        - python\n    permissions:\n      filesystem:\n        read: [\"config.json\", \"vision_history.json\", \".last_response\"]\n        write: [\"Temp/\", \"vision_history.json\", \".last_response\", \"config.json\"]\n      network:\n        - host: \"ark.cn-beijing.volces.com\"\n          purpose: \"Doubao vision model API\"\n        - host: \"open.bigmodel.cn\"\n          purpose: \"GLM vision model Chat Completions API\"\n        - host: \"open.volcengineapi.com\"\n          purpose: \"Auto IAM console usage sync\"\n    emoji: \"🔍\"\n    homepage: https://www.volcengine.com/docs/82379/1569618\n    locales: [\"en\"]\n---\n\n# ByteDance Visual Recognition — Doubao-Seed + GLM\n\nDoubao-Seed (6 models) + Zhipu GLM (2 free models). First run auto-generates `config.json`, fill in one API Key to start. IAM console usage syncs automatically on each recognition.\n\n## Privacy & Data Notice\n\n- **Network**: Selected images/videos and prompts are base64-encoded and sent to Volcengine (Doubao) or Zhipu (GLM) cloud APIs.\n- **Local cache**: Media files temporarily copied to `Temp/YYYYMMDD/`, default 7-day retention (`temp_retention_days` in config.json, range 1-3650).\n- **History**: Recognition history stored in `vision_history.json`, follow-up context in `.last_response`, auto-cleaned after 7 days. GLM follow-up reuses full message history including base64 media data; Doubao follow-up uses previous_response_id without re-transmitting files.\n- **Credentials**: API Keys stored in plaintext `config.json`. Do not use personal keys on shared machines.\n- **IAM sync**: Automatically syncs token usage from Volcengine IAM console on each recognition when IAM credentials are configured.\n- **First run**: A privacy notice is displayed once. Continuing past it acknowledges data handling practices.\n\n## Setup (pick one)\n\n### Doubao (Volcengine)\n\n1. Join the [Collaboration Rewards Program](https://console.volcengine.com/ark/region:cn-beijing/openManagement/rewardPlan) for free quota, then get your API Key\n2. Create inference endpoints, pick from:\n\n| config key | model | priority |\n|--------|------|:---:|\n| `doubao_seed_21p_id` | Doubao-Seed-2.1-Pro | primary |\n| `doubao_seed_21t_id` | Doubao-Seed-"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn75s192k98dtwr83bft4myd0582ytyg\",\n  \"slug\": \"bytedance-visual-recognition\",\n  \"version\": \"5.0.1\",\n  \"publishedAt\": 1786388451204\n}"},{"path":"proposal/PROPOSAL.md","content":"# 3.1.1 更新提案\n\n## 目标\n精简发布，修复安全审查问题，仅上传核心文件。\n\n## 变更\n- 收紧触发词列表，移除过于宽泛的 pattern\n- 批量处理改为仅复制媒体文件（不再复制整个目录树）\n- 剔除非核心文件（.env, Temp, .bak, .json 等）\n- 仅上传核心文件：doubao_vision_recognize.py, SKILL.md, skill-card.md"},{"path":"skill-card.md","content":"## Description:\n\nMultimodal visual recognition via Doubao-Seed and Zhipu GLM for image or video analysis, image or video to code, batch processing, follow-up questions, provider fallback, local media caching, persistent history, and optional IAM usage sync.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[etmnb](https://clawhub.ai/user/etmnb)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and external users can use this skill to send selected images or videos to configured Doubao or GLM vision models and receive scene descriptions, extracted text, structured Markdown analysis, or generated code. It also supports batch directory processing and follow-up questions over prior recognition context.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Selected images, videos, screenshots, documents, and prompts may contain private or sensitive information and are sent to third-party cloud vision APIs.\n\nMitigation: Review media before use, avoid sending regulated or confidential content unless approved, and configure only providers whose data handling terms are acceptable for the deployment.\n\nRisk: API keys and optional IAM credentials are stored in plaintext config.json.\n\nMitigation: Use least-privileged provider keys, avoid shared or synced folders, rotate exposed keys, and remove unused credentials from config.json.\n\nRisk: Local files such as Temp/, vision_history.json, and .last_response can retain media context or model responses after recognition.\n\nMitigation: Set the shortest practical retention period, manually delete Temp/, vision_history.json, and .last_response after sensitive sessions, and do not rely on automatic cleanup for GLM follow-up context.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/etmnb/skills/bytedance-visual-recognition)\n- [Volcengine Doubao visual understanding documentation](https://www.volcengine.com/docs/82379/1569618)\n- [Zhipu Open Platform](https://open.bigmodel.cn)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, Code, Shell commands, Configuration, Guidance]\n\n**Output Format:** [CLI text and Markdown-style analysis, with generated code when code output is requested.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Supports image files up to 15 MB, video files up to 50 MB, batch directory processing, follow-up context, local cache files, and persistent history.]\n\n## Skill Version(s):\n\n5.0.1 (source: server release metadata; artifact script and skill tags agree)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Multimodal visual recognition via Doubao-Seed (6 models) + Zhipu GLM (2 free models). Image/video to text/code, auto-fallback, batch directory processing, follow-up conversations, local media caching (Temp/), persistent history (vision_history.json), auto IAM console sync. First-run privacy notice, plaintext config.json, cross-platform. Skill: bytedance-visual-recognition Owner: etmnb Summary: Multimodal visual recognition via Doubao-Seed (6 models) + Zhipu GLM (2 free models). Image/video to text/code, auto-fallback, batch directory processing, follow-up conversations, local media caching (Temp/), persistent history (vision_history.json), auto IAM console sync. First-run privacy notice, plaintext config.json, cross-platform. Tags: latest:5.0.1 Vers","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1209,"uniquenessScore":47,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-10T01:38:38.452Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-10T01:38:38.452Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T06:43:46.834Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}