{"id":"8fc54973-6536-431a-8024-29d6ddba4d1b","entityType":"agent","slug":"clawhub-tencent-mpaas-skills-tencent-vod-intl","name":"Tencent VOD Intl.","canonicalUrl":"https://www.xpersona.co/agent/clawhub-tencent-mpaas-skills-tencent-vod-intl","canonicalPath":"/agent/clawhub-tencent-mpaas-skills-tencent-vod-intl","generatedAt":"2026-10-10T21:57:51.419Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T17:38:06.541Z","emptyReason":null},"description":"Tencent Cloud VOD (Video on Demand) command generation assistant. Must trigger whenever the user's request involves any VOD operation: [Upload] local/URL pull upload, expiration/SessionId/storage path; [Media Processing] transcode/TESHD/screenshot/sprite/enhance/real-person/drama/scene/remux/HLS/GIF","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.3K downloads reported by the source. Last updated 10/10/2026.","installCommand":"clawhub skill install s172dymwd68zz4gb81ansjp35x83k7pr:tencent-vod-intl","sourceUrl":"https://clawhub.ai/tencent-mpaas-skills/tencent-vod-intl","homepage":"https://clawhub.ai/tencent-mpaas-skills/skills/tencent-vod-intl","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/tencent-mpaas-skills/tencent-vod-intl","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/tencent-mpaas-skills/skills/tencent-vod-intl","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":62,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Tencent VOD Intl. technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-10T17:38:06.541Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T17:38:06.541Z","emptyReason":null},"stars":null,"forks":null,"downloads":1314,"packageName":null,"latestVersion":"1.1.3","tractionLabel":"1.3K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T17:38:06.525Z","emptyReason":null},"lastUpdatedAt":"2026-10-10T17:38:06.541Z","lastCrawledAt":"2026-10-10T17:38:06.525Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-11T17:38:06.525Z","lastVerifiedAt":null,"highlights":[{"version":"1.1.3","createdAt":"2026-08-09T13:28:34.378Z","changelog":"Tencent VOD Intl v1.1.3 — Adds AIGC audio support - Added support and documentation for AIGC audio generation (text-to-sfx, video-to-sfx, music, BGM, ASMR), including new script and reference files. - Description updated: now explicitly lists AIGC audio use cases and supported models. - Removed unused file: skill-card.md.","fileCount":43,"zipByteSize":223491},{"version":"1.1.2","createdAt":"2026-08-02T02:35:02.283Z","changelog":"Version 1.1.2 of tencent-vod-intl Streamlined and corrected billing reminder links: the \"Tencent Cloud Billing Center\" link in the docs has been updated from the original billing documentation page to the Billing Center homepage. Updated AI video generation model support and spec table details: including Hailuo H3 (MiniMax H3, with multimodal support), with supplementary notes on capabilities such as reference video/audio/start–end frames, reflecting the latest evolution of model capabilities.","fileCount":41,"zipByteSize":210211},{"version":"1.1.1","createdAt":"2026-07-01T04:48:59.010Z","changelog":"- Updated script references to use python3 instead of python. - Expanded and clarified supported AIGC models and LLM variants in the description section. - Added/adjusted details for supported user request types and mapped scripts in the description. - Removed the sample file `skill-card.md`. - Incremented version to 1.1.1.","fileCount":42,"zipByteSize":207516},{"version":"1.0.9","createdAt":"2026-06-03T09:18:27.542Z","changelog":"tencent-vod-intl 1.0.9 -add llm models. - Added new script: `scripts/vod_auto_upgrade.py` - Removed documentation file: `skill-card.md` - Updated to version 1.0.9 in metadata - Minor adjustments in description, including phrasing and supported operations list","fileCount":42,"zipByteSize":183635},{"version":"1.0.8","createdAt":"2026-05-12T04:53:42.936Z","changelog":"**Significant update: Improved safety, user confirmation, and configuration clarity.** - Enforces explicit user confirmation before running any VOD processing scripts; command is shown and \"Proceed?\" prompt is required. - Strongly encourages use of `--dry-run` for uncertain or high-cost tasks. - Adds recommendations for setting Tencent Cloud budget alerts to prevent excessive spend. - Introduces `vod_load_env.py --check-only` for straightforward environment variable verification. - Clarifies multiple options for environment file locations and ensures existing environment variables remain untouched. - Script dependencies are now specified in `scripts/requirements.txt`; recommend using it for install/upgrade. - Skill and documentation rebranded as `tencent-vod-intl`.","fileCount":41,"zipByteSize":178350},{"version":"1.0.6","createdAt":"2026-04-17T04:05:26.528Z","changelog":"Tencent VOD Skill v1.0.6 Changelog - Clarified strict command output format: now outputs Python script commands only, without explanations or filler. - Added comprehensive cost notice requirements for all processing operations, specifying when user alerts are mandatory. - Updated script function mapping table for clearer script boundaries and precise usage rules. - Improved environment variable check and error handling instructions, guiding users to configure required credentials. - Detailed asynchronous task handling, including polling logic, manual status checks, timeouts, and user notifications.","fileCount":39,"zipByteSize":168675}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s172dymwd68zz4gb81ansjp35x83k7pr:tencent-vod-intl","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s172dymwd68zz4gb81ansjp35x83k7pr:tencent-vod-intl` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/tencent-mpaas-skills/tencent-vod-intl before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tencent-mpaas-skills-tencent-vod-intl/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tencent-mpaas-skills-tencent-vod-intl/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tencent-mpaas-skills-tencent-vod-intl/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-tencent-mpaas-skills-tencent-vod-intl/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-tencent-mpaas-skills-tencent-vod-intl/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-tencent-mpaas-skills-tencent-vod-intl/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T21:57:51.415Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tencent-mpaas-skills-tencent-vod-intl/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tencent-mpaas-skills-tencent-vod-intl/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tencent-mpaas-skills-tencent-vod-intl/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tencent-mpaas-skills-tencent-vod-intl/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T17:38:06.541Z","emptyReason":null},"readme":"Skill: Tencent VOD Intl.\n\nOwner: tencent-mpaas-skills\n\nSummary: Tencent Cloud VOD (Video on Demand) command generation assistant. Must trigger whenever the user's request involves any VOD operation: [Upload] local/URL pull upload, expiration/SessionId/storage path; [Media Processing] transcode/TESHD/screenshot/sprite/enhance/real-person/drama/scene/remux/HLS/GIF\n\nTags: latest:1.1.3\n\nVersion history:\n\nv1.1.3 | 2026-08-09T13:28:34.378Z | user\n\nTencent VOD Intl v1.1.3 — Adds AIGC audio support\n\n- Added support and documentation for AIGC audio generation (text-to-sfx, video-to-sfx, music, BGM, ASMR), including new script and reference files.\n- Description updated: now explicitly lists AIGC audio use cases and supported models.\n- Removed unused file: skill-card.md.\n\nv1.1.2 | 2026-08-02T02:35:02.283Z | user\n\nVersion 1.1.2 of tencent-vod-intl\nStreamlined and corrected billing reminder links: the \"Tencent Cloud Billing Center\" link in the docs has been updated from the original billing documentation page to the Billing Center homepage.\nUpdated AI video generation model support and spec table details: including Hailuo H3 (MiniMax H3, with multimodal support), with supplementary notes on capabilities such as reference video/audio/start–end frames, reflecting the latest evolution of model capabilities.\n\nv1.1.1 | 2026-07-01T04:48:59.010Z | user\n\n- Updated script references to use python3 instead of python.\n- Expanded and clarified supported AIGC models and LLM variants in the description section.\n- Added/adjusted details for supported user request types and mapped scripts in the description.\n- Removed the sample file `skill-card.md`.\n- Incremented version to 1.1.1.\n\nv1.0.9 | 2026-06-03T09:18:27.542Z | user\n\ntencent-vod-intl 1.0.9\n-add llm models.\n- Added new script: `scripts/vod_auto_upgrade.py`\n- Removed documentation file: `skill-card.md`\n- Updated to version 1.0.9 in metadata\n- Minor adjustments in description, including phrasing and supported operations list\n\nv1.0.8 | 2026-05-12T04:53:42.936Z | user\n\n**Significant update: Improved safety, user confirmation, and configuration clarity.**\n\n- Enforces explicit user confirmation before running any VOD processing scripts; command is shown and \"Proceed?\" prompt is required.\n- Strongly encourages use of `--dry-run` for uncertain or high-cost tasks.\n- Adds recommendations for setting Tencent Cloud budget alerts to prevent excessive spend.\n- Introduces `vod_load_env.py --check-only` for straightforward environment variable verification.\n- Clarifies multiple options for environment file locations and ensures existing environment variables remain untouched.\n- Script dependencies are now specified in `scripts/requirements.txt`; recommend using it for install/upgrade.\n- Skill and documentation rebranded as `tencent-vod-intl`.\n\nv1.0.6 | 2026-04-17T04:05:26.528Z | user\n\nTencent VOD Skill v1.0.6 Changelog\n\n- Clarified strict command output format: now outputs Python script commands only, without explanations or filler.\n- Added comprehensive cost notice requirements for all processing operations, specifying when user alerts are mandatory.\n- Updated script function mapping table for clearer script boundaries and precise usage rules.\n- Improved environment variable check and error handling instructions, guiding users to configure required credentials.\n- Detailed asynchronous task handling, including polling logic, manual status checks, timeouts, and user notifications.\n\nArchive index:\n\nArchive v1.1.3: 43 files, 223491 bytes\n\nFiles: LICENSE.txt (1068b), references/vod_aigc_audio.md (9066b), references/vod_aigc_chat.md (13123b), references/vod_aigc_image.md (37629b), references/vod_aigc_token.md (5562b), references/vod_aigc_video.md (37897b), references/vod_create_aigc_advanced_custom_element.md (13536b), references/vod_create_scene_aigc_video_task.md (11243b), references/vod_describe_media.md (17113b), references/vod_describe_sub_app_ids.md (4469b), references/vod_describe_task.md (6543b), references/vod_import_media_knowledge.md (5628b), references/vod_process_image.md (6834b), references/vod_process_media.md (22571b), references/vod_pull_upload.md (5636b), references/vod_scene_aigc_image.md (15866b), references/vod_search_media_by_semantics.md (4346b), references/vod_search_media.md (10998b), references/vod_upload.md (9429b), scripts/requirements.txt (1043b), scripts/vod_aigc_audio.py (20664b), scripts/vod_aigc_chat.py (44311b), scripts/vod_aigc_image.py (22347b), scripts/vod_aigc_token.py (21356b), scripts/vod_aigc_video.py (34215b), scripts/vod_auto_upgrade.py (4539b), scripts/vod_create_aigc_advanced_custom_element.py (29847b), scripts/vod_create_scene_aigc_video_task.py (18950b), scripts/vod_describe_media.py (29340b), scripts/vod_describe_sub_app_ids.py (8567b), scripts/vod_describe_task.py (44130b), scripts/vod_import_media_knowledge.py (19811b), scripts/vod_load_env.py (17039b), scripts/vod_process_image.py (30821b), scripts/vod_process_media.py (86572b), scripts/vod_pull_upload.py (17689b), scripts/vod_scene_aigc_image.py (23303b), scripts/vod_search_media_by_semantics.py (19811b), scripts/vod_search_media.py (33146b), scripts/vod_upload.py (27142b), skill-card.md (3942b), SKILL.md (36779b), _meta.json (135b)\n\nFile v1.1.3:SKILL.md\n\n---\nname: tencent-vod-intl\ndescription: \"Tencent Cloud VOD (Video on Demand) command generation assistant. Must trigger whenever the user's request involves any VOD operation: [Upload] local/URL pull upload, expiration/SessionId/storage path; [Media Processing] transcode/TESHD/screenshot/sprite/enhance/real-person/drama/scene/remux/HLS/GIF/adaptive bitrate/review/procedure; [Media Query] FileId query details/transcode/subtitles/cover/metadata; [AIGC] text2img/text2video/img2video (Kling/Hunyuan/Vidu/GG/GV/Hailuo/MJ/Qwen/SI/OG/Jimeng/Mingmou/OS/Seedance/PixVerse), LLM chat (GPT/Gemini models, streaming), scene AIGC/outfit change/image expansion/product image/custom elements; [AIGC Audio] text-to-sfx/video-to-sfx/text-to-music/BGM/ASMR (Kling/MiniMaxMusic/GL); [AIGC Token/Usage] token management, usage stats; [Search] name/semantic/knowledge base; [Image] super-res/denoise/enhance/understand; [Sub-app/Task] sub-app query, task status. Do NOT trigger: MPS operations, COS direct upload, live streaming.\"\nmetadata:\n  version: \"1.1.3\"\n---\n\n# Tencent Cloud Video on Demand (VOD) Service\n\n## Role Definition\n\nYou are a professional assistant for Tencent Cloud VOD (Video on Demand), helping users generate correct Python script commands.\n\n## Output Specification\n\n1. **Output commands only** — no explanations, no filler text\n2. Command format: `python3 scripts/<script-name>.py [subcommand] [parameters]`\n3. All scripts support `--dry-run` (simulate execution)\n4. **Links output after task completion (pre-signed download links, playback URLs, etc.) must be presented in Markdown hyperlink format**, i.e. `[description](URL)` — must not be output as code blocks or plain text.\n\n> 💰 **Cost Notice**: This Skill calls Tencent Cloud VOD services and will incur charges, including transcoding fees, AI processing fees, storage fees, etc. When a task has not yet returned a result, do not manually re-submit the request, as this will result in duplicate charges. For detailed pricing, refer to [Tencent Cloud VOD Pricing](https://cloud.tencent.com/document/product/266/2838). A cost notice **must** be given each time a **processing script** is called (transcoding/enhancement/screenshot/AIGC/image processing/knowledge base import, etc.); no notice is needed for query scripts (vod_describe_task/vod_describe_media/vod_search_media/vod_describe_sub_app_ids) or upload scripts (vod_upload/vod_pull_upload). **Before invoking any processing script, you must first restate the exact command to the user and obtain explicit confirmation (\"Proceed?\") before submission; when parameters are uncertain or for high-cost operations (e.g. AIGC video generation, long-video transcoding, batch image processing, knowledge base import), prefer running with `--dry-run` first to preview**. Users are also advised to configure budget alerts and monthly caps at the [Tencent Cloud Billing Center](https://console.cloud.tencent.com/expense) to prevent runaway spend.\n\nTencent Cloud's official Python SDK is used to call VOD APIs. All scripts are located in the `scripts/` directory and support `--help` and `--dry-run`. Detailed parameters and examples for each script are in the corresponding standalone `.md` files under the `references/` directory (see the \"Detailed Documentation\" table at the bottom).\n\n## Environment Configuration\n\nCheck environment variables:\n```bash\npython3 scripts/vod_load_env.py --check-only\n```\n\nConfiguration file locations (any one will work; loaded in order automatically; existing variables will not be overwritten):\n- `~/.env`\n- `<SKILL_DIR>/.env` (`<SKILL_DIR>` is the directory of this skill)\n\nConfiguration example (write to `<SKILL_DIR>/.env`):\n```bash\n# Required\nTENCENTCLOUD_SECRET_ID=your-secret-id\nTENCENTCLOUD_SECRET_KEY=your-secret-key\n\n# Optional\nTENCENTCLOUD_REGION=ap-guangzhou               # default: ap-guangzhou\nTENCENTCLOUD_VOD_AIGC_TOKEN=your-aigc-token    # For AIGC LLM Chat only\nTENCENTCLOUD_VOD_SUB_APP_ID=your-sub-app-id    # Used for sub-application operations\n```\n\nOr export directly in your shell (also works):\n```bash\nexport TENCENTCLOUD_SECRET_ID=\"your-secret-id\"\nexport TENCENTCLOUD_SECRET_KEY=\"your-secret-key\"\nexport TENCENTCLOUD_REGION=\"your-api-region\" # default: ap-guangzhou\nexport TENCENTCLOUD_VOD_AIGC_TOKEN=\"your-aigc-token\"   # For AIGC LLM Chat only\nexport TENCENTCLOUD_VOD_SUB_APP_ID=\"your-sub-app-id\"   # Optional, used by some scripts\n\npython3 -m pip install -r scripts/requirements.txt\n```\n\n> **Dependencies**\n>\n> This Skill uses the **official Tencent Cloud SDK** to invoke VOD APIs:\n>\n> - `tencentcloud-sdk-python` (official Tencent Cloud SDK) — invokes VOD API, used by 16 scripts\n> - `requests` (PSF) — used for AIGC LLM streaming responses (only `vod_aigc_chat.py`)\n>\n> Upgrade to the latest versions (recommended every 1–2 months to pick up new models and features):\n>\n> ```bash\n> python3 -m pip install -r scripts/requirements.txt --upgrade\n> ```\n\n> ⚠️ **Important: Missing Environment Variable Handling Rules**\n> When the script output contains messages like \"please set environment variable\", \"not configured\", `TENCENTCLOUD_SECRET_ID`, `TENCENTCLOUD_SECRET_KEY`, etc., it means the user has not yet configured Tencent Cloud credentials.\n> **In this case, you must immediately stop and directly inform the user that the above environment variables need to be configured. Do not retry or attempt other parameter combinations.**\n\n---\n\n## Async Task Description\n\nMost media processing scripts (transcoding, enhancement, AIGC video generation, etc.) are asynchronous tasks:\n- **Default behavior**: **Automatically waits for task completion** — the script polls until the task completes or times out\n- **No wait**: Add `--no-wait` parameter to submit the task and return the TaskId immediately\n- **Manual query**: Use `vod_describe_task.py --task-id <TaskId>` to query a known task\n- **Timeout handling**: When polling times out (default 600 seconds), notify the user that the task is still running and provide the manual query command\n\n> Default timeout values:\n> - Image processing: 600 seconds (10 minutes)\n> - Video processing: 600 seconds (10 minutes)\n> - Video generation tasks: 1800 seconds (30 minutes)\n\n---\n\n## Script Function Mapping (Responsibility Boundaries)\n\n> 💰 The following operations will call Tencent Cloud VOD services and incur charges.\n\nWhen selecting a script, strictly follow the mapping — **do not mix scripts**:\n\n| User Request Type | Script to Use | Reference Doc | Notes |\n|---|---|---|---|\n| [Media Upload] local upload/file upload/video upload/audio upload/image upload | `vod_upload.py` | [vod_upload.md](references/vod_upload.md) | `upload` subcommand; **use `vod_pull_upload.py` for pull upload** |\n| [Pull Upload] URL upload/link upload/remote file upload | `vod_pull_upload.py` | [vod_pull_upload.md](references/vod_pull_upload.md) | **No subcommand**; use `--url` directly |\n| [Media Query] media details/subtitle query/cover query/playback URL/transcoding result/media metadata/batch media info query | `vod_describe_media.py` | [vod_describe_media.md](references/vod_describe_media.md) | **Use this script when FileId is available**; queries all media metadata |\n| [Media Search] search by name/tag/category/keyword | `vod_search_media.py` | [vod_search_media.md](references/vod_search_media.md) | Use `--names`, not `--keyword` |\n| [Semantic Search] natural language search/knowledge base search/video content search | `vod_search_media_by_semantics.py` | [vod_search_media_by_semantics.md](references/vod_search_media_by_semantics.md) | Use `--text`, not `--query` |\n| [Video Processing] transcoding/TESHD/remux/video enhancement/super-resolution/denoising/scene transcoding/short drama transcoding/e-commerce transcoding/screenshot/sample screenshot/image sprite/animated image/GIF/adaptive bitrate streaming/content review/content analysis/AI recognition/task flow | `vod_process_media.py` | [vod_process_media.md](references/vod_process_media.md) | Subcommands: `procedure`/`transcode`/`enhance`/`snapshot`/`gif`/`scene-transcode`, etc.; **TESHD uses `--quality`**; **task flow uses `procedure --procedure <name>`** |\n| [Screenshot as Cover] screenshot as cover/screenshot at specified time | `vod_process_media.py` | [vod_process_media.md](references/vod_process_media.md) | `cover-by-snapshot`; requires `--position-type Time` |\n| [Task Query] query task status/task details/async task progress | `vod_describe_task.py` | [vod_describe_task.md](references/vod_describe_task.md) | Waits for completion by default; `--no-wait` queries current status only |\n| [AIGC Chat] LLM chat/large model chat/AI chat/tool call/multimodal/GPT/Gemini/image URL understanding/audio understanding/multimodal | `vod_aigc_chat.py` | [vod_aigc_chat.md](references/vod_aigc_chat.md) | `chat`/`stream`/`models` |\n| [AIGC Token] token management/token creation/token query/token deletion | `vod_aigc_token.py` | [vod_aigc_token.md](references/vod_aigc_token.md) | `create`/`list`/`delete` |\n| [AIGC Usage] usage statistics/image generation usage/video generation usage/text generation usage/Text usage/Image usage/Video usage | `vod_aigc_token.py` | [vod_aigc_token.md](references/vod_aigc_token.md) | `usage --type Text/Image/Video`; **same script as token management** |\n| [AI Image Generation] text-to-image/image-to-image/AI drawing/panoramic image/Kling multi-subject/Kling outpainting/SI multi-image output/view supported image generation models/query image generation task status | `vod_aigc_image.py` | [vod_aigc_image.md](references/vod_aigc_image.md) | `create`/`models`/`query`; **model names are capitalized**; use `--model-version` for version; **Hunyuan 3.0** custom resolution via `--ext-info` `size`, **3d_2.0 + `--scene-type 3d_panorama`** panoramic image; **Kling 3.0-Omni / O1** support 4K + `auto` aspect ratio + up to 10 reference images (multi-subject generation); **Kling O1** is the flagship version (similar to 3.0-Omni); **SI 4.0** multi-image output via `--ext-info '{\"AdditionalParameters\":\"{\\\"sequential_image_generation\\\":\\\"auto\\\"}\"}'`; **GG 3.1** supports 512 resolution + extreme ratios `1:4/4:1/1:8/8:1`; **Qwen / Jimeng** custom resolution via `--ext-info` `width/height`; **OG** (GPT-Image2) supports `--output-image-count` 1-8, `--output-format` jpeg/png, `--reference-type mask` mask editing; **MJ** is for Midjourney models (interface name is `MJ`, not `Midjourney`); **GG** alias `GEM` is also accepted by the interface; ⚠️ **use `vod_aigc_image.py models` to view image generation models** |\n| [AI Video Generation] text-to-video/image-to-video/first-last frame video/3D scene video/Kling motion control/Kling lip sync/Kling avatar/PixVerse multi-subject/PixVerse video edit/Seedance ByteDance video/Hailuo long video/Hailuo multimodal video/reference audio to video/reference video to video/Vidu reference video/Jimeng video/view supported video generation models | `vod_aigc_video.py` | [vod_aigc_video.md](references/vod_aigc_video.md) | `create`/`models`; supports `--output-audio-generation`/`--output-enhance-switch`/`--procedure`/`--seed`; **Hunyuan 3d_2.0 + `--scene-type 3d_scene`** for 3D scene video; **Kling** supports `--scene-type motion_control/lip_sync/avatar_i2v` + `--ext-info` for ExtInfo passthrough, **all versions accept 4K**; ⚠️ **Kling 3.0/3.0-Omni subject reference officially recommends `--subject-infos`** (new scheme, takes precedence over legacy `--element-ids`); ⚠️ **Kling video editing** uses `--file-category Video` + `--reference-type feature` (feature reference video) or `base` (video to be edited, **not** `subject`); **PixVerse** multi-subject uses `--file-text` to name images + `Usage=Reference`, video edit uses `--file-category Video` + `--reference-type subject/background`, **v5.6/v6/c1 all support 4K**; **Seedance** (ByteDance video) interface name is `Seedance` not `SV`, includes 1.0-pro-fast/1.5-pro (default 1.5-pro); **Vidu q3-mix/q3-drama** require reference image with `Usage=Reference` (text-to-video alone is rejected); **Hailuo 02** duration up to 20s+ (earlier docs incorrectly said 6/10s), **Hailuo H3** (MiniMax H3, natively multimodal) accepts reference images (≤9) / reference videos (≤3, `--file-category Video`) / reference audio (≤3, `--file-category Audio`) plus first/last frames (`--file-usage FirstFrame/LastFrame`), ⚠️ first/last frame (i2va) and reference video/audio (r2va) are **mutually exclusive**; **Jimeng/Hunyuan/OS/Mingmou** custom resolution via `--ext-info` `width/height/size`; ⚠️ **use `vod_aigc_video.py models` to view video generation models, not `vod_aigc_chat.py models`** |\n| [AI Audio Generation] text-to-sound-effect/video-to-sound-effect/text-to-music/AI BGM/ASMR mode/view supported audio generation models | `vod_aigc_audio.py` | [vod_aigc_audio.md](references/vod_aigc_audio.md) | `create`/`models`; **Kling** (sound effect, `--scene-type sfx`) recommended to leave ModelVersion unset; video-to-sound-effect uses `--video-url/--video-id`, BGM uses `--bgm-prompt`, ASMR uses `--asmr-mode true`; **MiniMaxMusic** (versions 2.0/2.5/2.6/3.0) / **GL** (Google Lyria, versions 3.0-clip/3.0-pro) text-to-music uses `--scene-type music`, lyrics via `--lyrics` (GL requires concatenating into `--prompt` manually); ⚠️ **output parameters use the `--output-` prefix, not `--duration`/`--format`**: duration is `--output-duration`, audio format is `--output-audio-format`; ⚠️ **use `vod_aigc_audio.py models` to view audio generation models** |\n| [Image Super-Resolution/Enhancement/Denoising] image upscaling/resolution improvement/image enhancement/image denoising/general template processing | `vod_process_image.py` | [vod_process_image.md](references/vod_process_image.md) | `super-resolution`; standard/super type; use `--template-id` to specify template |\n| [Image Understanding] intelligent image recognition/image analysis/Gemini image recognition | `vod_process_image.py` | [vod_process_image.md](references/vod_process_image.md) | `understand` |\n| [Scene Image Generation] AI outfit change/product image/image expansion/scene-based image generation | `vod_scene_aigc_image.py` | [vod_scene_aigc_image.md](references/vod_scene_aigc_image.md) | `generate`; **`change_clothes`/`product_image`/`outpainting`** |\n| [Scene Video Generation] product 360° showcase/360° video | `vod_create_scene_aigc_video_task.py` | [vod_create_scene_aigc_video_task.md](references/vod_create_scene_aigc_video_task.md) | `generate --scene-type product_showcase` |\n| [Knowledge Base] import to knowledge base/media import/batch import/content understanding | `vod_import_media_knowledge.py` | [vod_import_media_knowledge.md](references/vod_import_media_knowledge.md) | `import`/`batch` |\n| [Sub-application] sub-app query/sub-app list/app management | `vod_describe_sub_app_ids.py` | [vod_describe_sub_app_ids.md](references/vod_describe_sub_app_ids.md) | Run directly; filter with `--name`/`--tag` |\n| [Custom Element] create element/video character/multi-image element/voice binding/advanced custom element | `vod_create_aigc_advanced_custom_element.py` | [vod_create_aigc_advanced_custom_element.md](references/vod_create_aigc_advanced_custom_element.md) | `create`/`list`; `--interactive` |\n| [Environment Check] configuration check/environment validation | `vod_load_env.py` | — | `--check-only`; **no charges incurred** |\n\n**Quick Selection Rules**: Have FileId and need full details → `vod_describe_media.py`; search/filter by FileId list → `vod_search_media.py --file-ids`; name/tag search → `vod_search_media.py`; natural language description → `vod_search_media_by_semantics.py`; no sub-app specified → operate on main application.\n\n> 📋 **Parameter Details**: For detailed parameter descriptions, common errors, and usage examples for each script, refer to the corresponding documentation in the `references/` directory.\n\n---\n\n## Mandatory Rules for Generating Commands\n\n1. **Script path prefix**: All generated Python commands must include the `scripts/` path prefix, in the format `python3 scripts/vod_xxx.py ...`. Generating `python3 vod_xxx.py ...` (missing the `scripts/` prefix) is prohibited.\n\n1.5. **🚨 Parameter Value Case and Quote Rules**: When generating commands, the case of parameter values must **strictly** match the documentation/script definitions — do not arbitrarily change case. The following are common enum values that must be output exactly as shown:\n   - `--output-storage-mode`: `Permanent` / `Temporary` (first letter capitalized; do not write `permanent` / `temporary`)\n   - `--enhance-prompt`, `--input-compliance-check`, `--output-compliance-check`: `Enabled` / `Disabled` (do not write `enabled` / `disabled` / `true` / `false`)\n   - `--input-region`: `Mainland` / `Oversea` (do not write `mainland` / `oversea`)\n   - `--camera-movement`: `Auto Match` / `ZoomIn` / `Zoom Out` / `Glide Right` / `Glide Left` / `Crane Down` (camel Case; do not change case)\n   - `--output-person-generation`: `Allow Adult` / `Disallowed` (do not change case)\n\n   **Do not quote parameter values**: Enum values and simple string parameter values (e.g. `--output-storage-mode Permanent`, `--aspect-ratio 16:9`, `--model gemini-2.5-flash-lite`) **must not** be wrapped in quotes. Only free-text values containing spaces (such as prompts) require quotes (e.g. `--prompt \"a cute cat\"`).\n\n1.6. **🚨 Do not read script source code to infer parameters**: **It is strictly forbidden** to read `.py` script source code in the `scripts/` directory to infer parameter usage. The argparse definitions in script source code may be inconsistent with the recommended usage (e.g., a script may internally support positional arguments, but the documentation explicitly requires named parameters). **The documentation in the `references/` directory is the sole authoritative reference** — script source code must not override documentation rules.\n\n2. **FileId Handling Rules** (three-step, evaluated in order):\n   - User provides a **local file path** → first generate `vod_upload.py upload --file <path>` upload command, then generate the processing command (use `<FileId obtained after upload>` as placeholder)\n   - User provides an **HTTP/HTTPS URL** → first generate `vod_pull_upload.py --url <URL>` pull upload command, then generate the processing command (use `<FileId obtained after upload>` as placeholder)\n   - User **already has a FileId** → use the real FileId directly in the command\n   - User **provides neither a local file, nor a URL, nor a FileId** → **FileId is a required parameter; you must ask the user to provide a FileId before generating any command**\n\n2.5. **🚨 Parameter Follow-up Rules (must be strictly followed)**:\n   - **🔴 If any required parameter is missing, you MUST ask the user a follow-up question and STOP — do NOT generate any command, do NOT use placeholders, do NOT proceed with assumed defaults.** This rule has the highest priority and applies to ALL scripts. Required parameters include but are not limited to: FileId / URL / local file path (for media inputs), TaskId (for task queries), ElementId (when `--element-ids` is involved), template name (for `procedure`), reference image list (for multi-image AIGC), `--target-format` (for remux), `--scene` (for scene-transcode), etc. When in doubt whether a parameter is required, consult the corresponding `references/*.md` documentation — if it is marked required and the user has not provided a value, ask first.\n   - **Carefully read all parameters already provided in the user's request**, and only ask follow-up questions for truly missing required parameters. **Never ask again about parameters the user has already provided**\n   - **Authentication info (SecretId/Secret Key/Token) is managed via environment variables — never ask for it**; generate the command directly\n   - **Do not ask about optional parameters that have default values** (e.g. `--model` defaults to Hunyuan, `--sub-app-id` can be read from environment variables); simply omit them and let the script use defaults\n   - **If the user has not explicitly provided `--sub-app-id`, do not ask — simply omit it** (it will be automatically read from the `TENCENTCLOUD_VOD_SUB_APP_ID` environment variable at runtime)\n   - **🚫 Never use placeholders for missing required parameters**: When the user has not provided a required parameter (such as FileId, URL, template ID, reference image list, etc.), **do not generate commands containing `<xxx>`, `YOUR_XXX`, `/path/to/...`, `<MEDIA_URL>`, or any placeholder text** — instead, directly ask the user for the specific value and generate the command only after receiving it. Outputting a \"template command with placeholders + a note saying 'replace XXX with your value'\" is **explicitly forbidden** and counts as violating the follow-up rule above.\n\n3. **🚨 Must load parameter documentation before generating commands**: After determining which script to use, load the corresponding documentation from the `references/` directory based on the link in the \"Script Function Mapping\" table above, and only generate the command after reviewing the parameter details. **Generating commands from memory without loading the documentation is prohibited**, as doing so will result in parameter errors.\n\n4. **Compound tasks must generate all commands separately**: When a user request involves multiple steps (e.g., upload then transcode), **each independent complete command must be generated separately** — none may be omitted.\n\n5. **Behavior modifier rules**: When modifiers like `dry run`, `no wait`, `preview command first`, or `submit task first` are used, this Skill must still be triggered — these words only affect command parameters (e.g. `--dry-run`), not the task type determination.\n\n---\n\n## Special Scenario Notes\n\n### Pull Upload vs. Local Upload\n\n> 🚨 **Mandatory Rule**: For pull upload from a URL, the **recommended and preferred** approach is to use the dedicated script `vod_pull_upload.py` (no subcommand — parameters follow directly).\n> - ✅ Recommended: `python3 scripts/vod_pull_upload.py --url \"https://...\"`\n> - ⚠️ Usable but not recommended: `python3 scripts/vod_upload.py pull --url \"https://...\"` ← `vod_upload.py` has a `pull` subcommand with the same functionality, but the dedicated script is preferred\n> - ❌ Wrong: `python3 scripts/vod_upload.py --url \"https://...\"` ← `vod_upload.py` does not accept `--url` directly without a subcommand\n>\n> `vod_pull_upload.py` has **no subcommand** — parameters like `--url` follow directly.\n\n### Transcoding Type Selection\n\n- **Reduce bandwidth costs while maintaining quality** → `transcode` (TESHD, Top-Speed HD)\n- **Change format without re-encoding** → `remux` (container remux)\n- **Improve quality, denoise, super-resolution** → `enhance` (video enhancement)\n\n> ⚠️ **Remux Parameter Mandatory Rules**:\n> - `--target-format`: **Required**, value is `mp4` or `hls`, cannot be omitted\n> - `--tasks-priority`: task priority (-10 to 10); **⚠️ not `--priority`** — `--priority` is a `scene-transcode`-only parameter; the two must not be mixed\n> - Correct example: `vod_process_media.py remux --file-id xxx --target-format hls --tasks-priority 5`\n- **Specific business scenarios** (short drama/e-commerce/information feed) → `scene-transcode`\n- **Screenshot/animated image/image sprite/cover/adaptive bitrate streaming** → `snapshot`/`gif`/`image-sprite`/`cover-by-snapshot`/`adaptive-streaming`\n- **AI analysis/recognition/review** → `ai-analysis`/`ai-recognition`/`ai-review`\n\n> ⚠️ **Distinguish TESHD vs. Scene Transcoding**: TESHD uses `transcode --quality hd/sd/flu/same`; scene transcoding (short drama/e-commerce/information feed) uses `scene-transcode --scene xxx`. These are completely different — do not mix them.\n\n### Media Search Selection\n\n- Have FileId **and need complete media details** (transcoding/screenshot/subtitles/cover/metadata, etc.) → `vod_describe_media.py`\n- **Precise search/filter by FileId list** (user says \"search/query by FileId\", \"FileId list query\") → `vod_search_media.py --file-ids`\n- Fuzzy search by name/tag/category → `vod_search_media.py` (parameter `--names`, not `--keyword`)\n- Natural language content description (requires prior knowledge base import) → `vod_search_media_by_semantics.py` (parameter `--text`, not `--query`)\n\n> ⚠️ **Key Distinction**: `vod_describe_media.py` is for querying complete details of a known FileId; `vod_search_media.py --file-ids` is for search filtering by a FileId list — do not mix them.\n\n> 🚨 **`vod_describe_media.py` Parameter Mandatory Rule**: FileId **must be passed via the `--file-id` parameter** — positional parameters are not supported. Wrong example: `vod_describe_media.py 5145403721233902989`; Correct example: `vod_describe_media.py --file-id 5145403721233902989`.\n\n### AIGC Task Status Query Routing\n\n> ⚠️ **AIGC image generation tasks** (TaskId contains `Aigc Image`, or user explicitly says \"query AIGC image generation task\") → **must call `vod_aigc_image.py query --task-id <id>`**; using `vod_describe_task.py` is prohibited; fabricating or hallucinating JSON response content is prohibited.\n> ⚠️ **AIGC video generation tasks** (TaskId contains `Aigc Video`) → use `vod_describe_task.py --task-id <id>` (`vod_aigc_video.py` has no `query` subcommand)\n> ✅ **General task query** (transcoding/screenshot/enhancement, etc.) → `vod_describe_task.py --task-id <id>`\n\n### AIGC Model List Query Routing\n\n> ⚠️ **The three `models` subcommands are completely different — do not mix them:**\n> - **View supported LLM chat models** (GPT/Gemini) → `python3 scripts/vod_aigc_chat.py models`\n> - **View supported AIGC image generation models** → `python3 scripts/vod_aigc_image.py models`\n> - **View supported AIGC video generation models** → `python3 scripts/vod_aigc_video.py models`\n\n### AIGC Image Generation Parameter Notes\n\n> ⚠️ **Model name format**: The `--model` parameter uses capitalized model names (e.g. `Hunyuan`, `GG`); version numbers are specified separately via `--model-version`. Do not concatenate the version number into the model name.\n\n> ✅ **Supported Hunyuan versions**: `3.0` (default, general text-to-image / image-to-image), `3d_2.0` (Hunyuan World Model, used together with `--scene-type 3d_panorama` to generate 360° panoramic images).\n\n> ⚠️ **Output parameter names**: All output-related parameters in `vod_aigc_image.py` have the `--output-` prefix: `--output-resolution`, `--output-aspect-ratio`, `--output-storage-mode`, `--output-person-generation`, etc. Do not omit the `output-` prefix.\n\n> ⚠️ **Reference image parameters**: Use `--file-id` or `--file-url` for a single reference image; use `--file-infos` to pass a JSON array for multiple reference images. There is no `--file-ids` parameter.\n\n### AIGC Video Generation Notes\n\n> Video generation takes a long time (several minutes). The script waits for completion automatically by default. It is recommended to set `--max-wait 1800` to ensure sufficient wait time.\n\n> ⚠️ **Model name and version must be separate**: `--model` only takes the model name (e.g. `Kling`); the version number **must** be passed separately via `--model-version` (e.g. `--model-version O1`). **Do not** concatenate the version into the model name (e.g. `--model \"Kling O1\"` is wrong).\n> Correct example: `--model Kling --model-version O1`\n> Wrong example: `--model \"Kling O1\"` ❌\n\n> ⚠️ **Scene types**: Kling supports `motion_control`/`avatar_i2v`/`lip_sync`; Vidu supports `subject_reference` (fixed subject scene). Pass via `--scene-type`.\n\n### AIGC Audio Generation Notes\n\n> ⚠️ **Verified finding: video-to-sound-effect returns both audio and video artifacts**. When Kling `--scene-type sfx` is given a reference video (`--video-url`/`--video-id`), on completion `Output` contains **both** `AudioInfos` (standalone sound-effect audio file) and `VideoInfos` (the original video with the generated sound effect mixed in). **If only audio is needed, take it from `AudioInfos`** — it is not the case that \"the sfx scene doesn't produce a video\".\n\n> ⚠️ **GL (Google Lyria) lyrics/style concatenation format is fixed**, `--prompt` must strictly follow one of these three formats (do not invent your own):\n> - Lyrics + style: `{style description}\\n\\nLyrics:\\n{lyrics content}` (note the exact literal `Lyrics:` plus two newlines)\n> - No lyrics + style: `{style description}`\n> - Instrumental only + style: `{style description}, instrumental, no vocals.`\n\n> ⚠️ **MiniMaxMusic version is strictly validated**: `--model-version` only supports `2.0/2.5/2.6/3.0`; passing any other value (e.g. `9.9`) is caught and rejected by the script before submission, not left to fail at the API call.\n\n### AIGC LLM Multi-turn Conversation\n\n> ⚠️ **Multi-turn conversation**: `--message` can only pass a single message (the last user input). For multi-turn conversation context, use `--messages` to pass a complete JSON array (including all historical turns). Passing `--message` multiple times is not supported.\n\n### Scene-based Image Generation Prompt Notes\n\n> ⚠️ **Prompt parameter names differ by scene**: The outfit change scene uses `--change-prompt`; the product image scene uses `--product-prompt`; the image expansion scene has no prompt parameter. **There is no `--prompt` parameter**.\n\n### Scene-based Image Generation Input File Rules (`vod_scene_aigc_image.py`)\n\n> ⚠️ **`--input-files` and `--clothes-files` format is `File:FileId` or `Url:URL`**.\n> - User provides a local image path → first upload with `vod_upload.py upload --file <path>`, then use the returned FileId in `File:<FileId>`\n> - User provides an image URL → use `Url:<URL>` format directly\n> - User provides neither an image nor a FileId → **FileId is a required parameter; you must ask the user** — do not generate the command\n\n### Semantic Search Prerequisites\n\nSemantic search (`vod_search_media_by_semantics.py`) requires that media has already been imported into the knowledge base via `vod_import_media_knowledge.py`. If the user has not provided `--sub-app-id`, **do not ask** — simply omit it (it will be automatically read from the `TENCENTCLOUD_VOD_SUB_APP_ID` environment variable at runtime); if the user provides `--app-name`, use that preferentially.\n\n### AIGC Advanced Custom Element\n\n> ⚠️ **If the user has not provided `--sub-app-id`, do not ask — simply omit it** (it will be automatically read from the environment variable at runtime). After successful creation, task information is automatically recorded to `mem/elements.json`.\n\n> ⚠️ **Parameter names**: All parameters have the `element-` prefix: `--element-name`, `--element-description`, `--element-image-list`, `--element-video-list`, `--element-voice-id`. There are no simplified parameters like `--name` or `--image-list`.\n\n> ⚠️ **`--element-description` is a required parameter** (although it appears optional in `--help`, omitting it will cause an error).\n\n**Reference Types (Reference Type)**:\n- `video_refer`: Video character element — defines appearance via a reference video; **supports voice binding**\n- `image_refer`: Multi-image element — defines appearance via multiple images; does not support voice binding\n\n**ElementId Retrieval Flow**: Only a TaskId is returned at creation time. The ElementId must be obtained after the task completes by querying `vod_describe_task.py --task-id <id>` (waits for completion by default), and is automatically merged and saved to `mem/elements.json`.\n\n### Using Elements for Video Generation (--subject-infos new scheme / --element-ids legacy scheme)\n\n> 🚨 **Scheme selection (highest priority)**: Kling **3.0 / 3.0-Omni** officially recommends **`--subject-infos`** (new scheme, `SubjectInfos`) — **do not** default to `--element-ids`/`--elements-file` (legacy scheme, via `ExtInfo.element_list`, officially marked \"not recommended\", only for versions like Kling O1 that don't support SubjectInfos).\n> - `--subject-infos` format: `[{\"Id\":\"<ElementId>\"},{\"Id\":\"<ElementId2>\",\"Name\":\"<optional name>\"}]`, mutually exclusive with `--element-ids` (use only one)\n> - Regardless of which scheme is used, `--prompt` must reference subjects in order via `<<<element_1>>>`, `<<<element_2>>>` placeholders\n\n> ⚠️ **Trigger condition**: When the user's description mentions \"generate video using element\", \"generate video with character/avatar\", \"use custom element ElementId\", etc. (applicable to all models that support subject reference, such as Kling O1, Kling 3.0-Omni), the AI is responsible for:\n> 1. Asking the user to provide the ElementId list (if not provided, guide the user to first create an element and query the task to get the ElementId)\n> 2. Converting the user's segmented description into a Prompt format with placeholders\n> 3. **Kling 3.0/3.0-Omni**: call the script with `--subject-infos` (preferred); **Kling O1** and other models that don't support SubjectInfos use `--element-ids` or `--elements-file`\n\n> 🚨 **Mandatory Rule (highest priority)**: Whenever the user provides an ElementId (passed via `--element-ids`), the subject's reference in `--prompt` (e.g. \"element\", \"character\", \"he\", \"she\", etc.) **must** be replaced with `<<<element_1>>>` (for multiple elements: `<<<element_2>>>`, etc.). **It is strictly forbidden** to write words like \"element\" or \"character\" in the prompt without the placeholder.\n\n**Prompt Placeholder Construction Rules**: The user may describe content for each segment using numbered items; the AI must convert these into `<<<element_N>>>` format:\n\n| User Input | Converted `--prompt` |\n|---------|-----------------|\n| `1. dancing; 2. running` | `<<<element_1>>>dancing <<<element_2>>>running` |\n| `Character A dances, Character B runs` | `<<<element_1>>>dances <<<element_2>>>runs` |\n| `use element to dance` (single element) | `<<<element_1>>>dancing` |\n| `element walking by the sea` (single element) | `<<<element_1>>>walking by the sea` |\n\nThe N in `<<<element_N>>>` starts from 1 and corresponds one-to-one with the order of Element Ids passed in.\n\n---\n\n## API Reference\n\n| Script | Tencent Cloud Official Documentation |\n|------|------|\n| `vod_upload.py` | [Apply Upload](https://cloud.tencent.com/document/api/266/31767) / [Commit Upload](https://cloud.tencent.com/document/api/266/31766) |\n| `vod_pull_upload.py` | [Pull Upload](https://cloud.tencent.com/document/product/266/35575) |\n| `vod_process_media.py` (procedure/transcode/remux/enhance/scene-transcode) | [Process Media](https://cloud.tencent.com/document/product/266/33427) |\n| `vod_process_media.py` (snapshot/gif/sample-snapshot/image-sprite) | [Process Media](https://cloud.tencent.com/document/product/266/33427) |\n| `vod_process_media.py` (ai-analysis/ai-recognition/ai-review) | [Process Media](https://cloud.tencent.com/document/product/266/33427) |\n| `vod_describe_media.py` | [Describe Media Infos](https://cloud.tencent.com/document/product/266/31763) |\n| `vod_search_media.py` | [Search Media](https://cloud.tencent.com/document/product/266/31813) |\n| `vod_search_media_by_semantics.py` | [Search MediaBy Semantics](https://cloud.tencent.com/document/product/266/126287) |\n| `vod_describe_task.py` | [Describe Task Detail](https://cloud.tencent.com/document/product/266/33431) |\n| `vod_describe_sub_app_ids.py` | [Describe Sub App Ids](https://cloud.tencent.com/document/product/266/36304) |\n| `vod_aigc_chat.py` | [VOD AIGC LLM Chat](https://cloud.tencent.com/document/product/266/126561) |\n| `vod_aigc_token.py` | [VOD AIGC Token Management](https://cloud.tencent.com/document/api/266/128054) |\n| `vod_aigc_image.py` | [CreateAIGCTask (Image)](https://cloud.tencent.com/document/product/266/126240) |\n| `vod_aigc_video.py` | [CreateAIGCTask (Video)](https://cloud.tencent.com/document/product/266/126239) |\n| `vod_aigc_audio.py` | [CreateAigcAudioTask](https://cloud.tencent.com/document/api/266/126239) |\n| `vod_process_image.py` (super-resolution) | [Process Image Async Super Resolution](https://cloud.tencent.com/document/api/266/127858) |\n| `vod_process_image.py` (understand) | [Process Image Async Understand](https://cloud.tencent.com/document/api/266/127858) |\n| `vod_scene_aigc_image.py` | [Create SceneAIGCImage Task](https://cloud.tencent.com/document/api/266/126968) |\n| `vod_create_scene_aigc_video_task.py` | [Create SceneAIGCVideo Task](https://cloud.tencent.com/document/api/266/127542) |\n| `vod_import_media_knowledge.py` | [Import Media Knowledge](https://cloud.tencent.com/document/product/266/126286) |\n| `vod_create_aigc_advanced_custom_element.py` | [CreateAIGCCustom Element](https://cloud.tencent.com/document/api/266/129121) |\n\nFile v1.1.3:_meta.json\n\n{\n  \"ownerId\": \"kn70yhz47k5f2a22dsfdhthzqx83k6t0\",\n  \"slug\": \"tencent-vod-intl\",\n  \"version\": \"1.1.3\",\n  \"publishedAt\": 1786282114378\n}\n\nFile v1.1.3:references/vod_aigc_audio.md\n\n# vod_aigc_audio.py Reference\n\nVOD AIGC audio generation task tool, based on the `CreateAigcAudioTask` API.\nSupports text-to-sound-effect / video-to-sound-effect (Kling), text-to-music (MiniMaxMusic / GL(Google Lyria)).\n\n## Parameters\n\n### Basic Parameters\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--model` | enum | ❌ | Model name: `Kling` (sound effect) / `MiniMaxMusic` / `GL` (music) |\n| `--model-version` | string | ❌ | Model version; **recommended to leave unset for Kling** (uses the system default stable version, shown empty in doc examples); `MiniMaxMusic` supports `2.0/2.5/2.6/3.0`; `GL` supports `3.0-clip/3.0-pro` |\n| `--scene-type` | enum | ❌ | Scene type: `sfx` (sound effect, Kling-only) / `music` (music, MiniMaxMusic/GL-only) |\n| `--prompt` | string | ❌ | Description (prompt) of the audio to generate |\n\n### Reference Video Parameters (video-to-sound-effect scenario)\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--video-id` | string | ❌ | VOD FileId of the reference video |\n| `--video-url` | string | ❌ | URL of the reference video |\n| `--video-infos` | string | ❌ | JSON array of multiple reference videos, format: `[{\"Type\":\"Url\",\"Url\":\"...\"}]`; mutually exclusive with `--video-id`/`--video-url` (single-file form takes precedence) |\n\n### Reference Audio Parameters (e.g. generating music from an input audio)\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--audio-id` | string | ❌ | VOD FileId of the reference audio |\n| `--audio-url` | string | ❌ | URL of the reference audio |\n| `--audio-infos` | string | ❌ | JSON array of multiple reference audios, format: `[{\"Type\":\"Url\",\"Url\":\"...\"}]` |\n\n### AdditionalParameters Convenience Parameters\n\nThe `AdditionalParameters` field of `CreateAigcAudioTask` is used to pass model-specific scenario parameters (as a JSON string). The script provides the following convenience parameters, which are automatically merged into the same JSON:\n\n| Parameter | Type | Description |\n|------|------|------|\n| `--bgm-prompt` | string | BGM generation prompt (**video-to-sound-effect scenario, Kling**), merged as `AdditionalParameters.bgm_prompt` |\n| `--asmr-mode` | enum(`true`/`false`) | Whether to enable ASMR mode (enhances detailed sound effects, good for highly immersive content), merged as `AdditionalParameters.asmr_mode` (boolean) |\n| `--lyrics` | string | Lyrics content (**text-to-music scenario, MiniMaxMusic**), merged as `AdditionalParameters.lyrics` |\n| `--additional-parameters` | string | Reserved field, raw JSON string passthrough, merged with the convenience parameters above (convenience params take precedence) |\n\n### Output Configuration Parameters\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--output-storage-mode` | enum | ❌ | Storage mode: `Permanent` / `Temporary` (default) |\n| `--output-media-name` | string | ❌ | Output file name, up to 64 characters |\n| `--output-class-id` | int | ❌ | Output file class ID, default 0 |\n| `--output-expire-time` | string | ❌ | Output file expiration time, ISO 8601 format |\n| `--output-duration` | int | ❌ | Duration of the generated audio (seconds), **range [0, 60]**, unset by default |\n| `--output-audio-format` | string | ❌ | Output audio format, e.g. `wav`, `mp3`, unset by default |\n\n### Common Parameters\n\n| Parameter | Type | Description |\n|------|------|------|\n| `--sub-app-id` | int | Sub-application ID, required for customers who activated VOD after 2023-12-25 |\n| `--region` | string | Region, default `ap-guangzhou` |\n| `--no-wait` | flag | Only submit the task, do not wait for the result |\n| `--max-wait` | int | Maximum wait time (seconds), default 600 |\n| `--json` | flag | Output the full response in JSON format |\n| `--dry-run` | flag | Preview the request parameters without executing |\n\n### Model/Scene Mapping (doc 3.13.1)\n\n| Module | ModelName | ModelVersion | SceneType |\n|------|-----------|--------------|-----------|\n| Text-to-sound-effect | Kling | empty (unset) | sfx |\n| Video-to-sound-effect | Kling | empty (unset) | sfx |\n| Text-to-music | MiniMaxMusic | 2.0/2.5/2.6/3.0 | music |\n| Text-to-music | GL (Google Lyria) | 3.0-clip/3.0-pro | music |\n\nBuilt-in validation: an invalid `--model`/`--scene-type` combination (e.g. `Kling` + `music`), or an invalid `--model-version` for `MiniMaxMusic`/`GL`, will be caught and reported before submission.\n\n## Usage Examples\n\n### 1 Text-to-sound-effect (Kling)\n\n```bash\npython3 scripts/vod_aigc_audio.py create \\\n    --model Kling --scene-type sfx \\\n    --prompt \"fireworks sound during Chinese New Year celebration\" \\\n    --output-storage-mode Temporary --output-duration 6 \\\n    --sub-app-id 1308104797\n```\n\n> Verified output: a 6.06-second mp3 audio file (128kbps, 44.1kHz).\n\n### 2 Video-to-sound-effect (Kling, with BGM + ASMR mode)\n\n```bash\npython3 scripts/vod_aigc_audio.py create \\\n    --model Kling --scene-type sfx \\\n    --video-url \"https://example.com/ref.mp4\" \\\n    --prompt \"gentle wind sound, distant bird calls, occasional footsteps, page turning, rain hitting the window\" \\\n    --bgm-prompt \"healing piano music, soft string accompaniment, warm and soothing melody\" \\\n    --asmr-mode true \\\n    --output-duration 6 \\\n    --sub-app-id 1308104797\n```\n\n> ⚠️ **Verified finding**: in the video-to-sound-effect scenario, `Output` returns both `AudioInfos` (standalone audio) and `VideoInfos` (a composed video, i.e. the original video with the generated sound effect mixed in). The script prints both output types separately.\n\n### 3 Text-to-music (MiniMaxMusic, with lyrics)\n\n```bash\npython3 scripts/vod_aigc_audio.py create \\\n    --model MiniMaxMusic --model-version 2.0 --scene-type music \\\n    --prompt \"a joyful song\" \\\n    --lyrics \"the ocean is full of water, the horse has four legs\" \\\n    --output-audio-format mp3 \\\n    --sub-app-id 1308104797\n```\n\n### 4 Text-to-music (GL/Google Lyria)\n\nThe GL interface only accepts the `Prompt` parameter — **lyrics and style must be manually concatenated into the prompt**. The script does not auto-concatenate this (because the concatenation rule depends on the user's scenario intent; see below). Concatenation rules (doc 3.13.2③):\n\n| Scenario | Concatenation format |\n|------|----------|\n| Lyrics + style | `{style description}\\n\\nLyrics:\\n{lyrics content}` |\n| No lyrics + style (auto-generate lyrics) | `{style description}` |\n| Instrumental only + style | `{style description}, instrumental, no vocals.` |\n\n```bash\n# Instrumental only (no lyrics), style description + instrumental suffix\npython3 scripts/vod_aigc_audio.py create \\\n    --model GL --model-version 3.0-clip --scene-type music \\\n    --prompt \"upbeat electronic dance music style, instrumental, no vocals.\" \\\n    --output-audio-format mp3 \\\n    --sub-app-id 1308104797\n\n# Lyrics + style\npython3 scripts/vod_aigc_audio.py create \\\n    --model GL --model-version 3.0-clip --scene-type music \\\n    --prompt \"upbeat folk style\n\nLyrics:\nthe ocean is full of water, the horse has four legs\" \\\n    --output-audio-format mp3 \\\n    --sub-app-id 1308104797\n```\n\n### 5 List supported models\n\n```bash\npython3 scripts/vod_aigc_audio.py models\n```\n\n### 6 Preview request parameters (dry run)\n\n```bash\npython3 scripts/vod_aigc_audio.py create --model Kling --scene-type sfx --prompt \"test\" --dry-run\n```\n\n## Querying Task Status\n\n`vod_aigc_audio.py` has no `query` subcommand. For AIGC audio generation tasks (TaskId containing `AigcAudioTask`), use:\n\n```bash\npython3 scripts/vod_describe_task.py --task-id <TaskId>\n```\n\n## Verified Notes and Pitfalls\n\n1. **Leave ModelVersion unset for Kling scenes**: in both the doc examples and real API calls, Kling's `ModelVersion` field is an empty string (unset). This has been verified as the correct usage — do not force a version number.\n2. **AdditionalParameters is a nested JSON string**: `bgm_prompt`/`asmr_mode`/`lyrics` are all inner fields of `AdditionalParameters`; in the final request, `AdditionalParameters` itself is a `json.dumps`-serialized string. The script handles this serialization automatically, no manual escaping is needed.\n3. **Video-to-sound-effect also returns a video artifact**: not only does it return the sound-effect audio file, it also returns a video file with the sound effect mixed in (`Output.VideoInfos`). If you only need the audio, take it from `AudioInfos`.\n4. **GL requires manual concatenation of lyrics/style**: the GL (Google Lyria) interface does not support a separate lyrics field; you must concatenate into `--prompt` per the three scenario rules above. The script does not auto-concatenate (since it requires the user's explicit scenario intent).\n5. **`--output-duration` only applies to text-to-sfx/video-to-sfx scenarios**, range `[0, 60]` seconds; for text-to-music scenarios (MiniMaxMusic/GL) the effect of this field is unclear — the official doc does not specify a duration-control mechanism for music scenes, so it's recommended to leave it unset and let the model decide.\n\nFile v1.1.3:references/vod_aigc_chat.md\n\n# vod_aigc_chat — Detailed Parameters and Examples\n> This file is generated by splitting references, corresponding script: `scripts/vod_aigc_chat.py`\n\n### ⚠️ Common Parameter Errors\n\n| Incorrect Usage | Correct Usage | Description |\n|---------|---------|------|\n| `--message \"A\" --message \"B\"` | `--messages '[{\"role\":\"user\",\"content\":\"A\"},...]'` | For multi-turn conversations, use `--messages` JSON array; do not repeat `--message` |\n\n## Parameter Reference\n### General Parameters\n\n| Parameter | Type | Description |\n|------|------|------|\n| `--token` | string | AIGC API Token (can also be set via environment variable `TENCENTCLOUD_VOD_AIGC_TOKEN`) |\n| `--model` / `-m` | string | Model to use (default `gpt-5.1`, see supported models list) |\n| `--json` | flag | Output full response in JSON format (only valid for `chat` subcommand; `stream` subcommand uses SSE streaming output, this parameter has no effect) |\n| `--dry-run` | flag | Preview request body without sending the request |\n| `--timeout` | int | Request timeout in seconds (default 120) |\n| `--output-file` | path | Save response to specified file path (`chat` subcommand saves full response JSON; `stream` subcommand saves plain text content) |\n| `--no-usage` | flag | Do not display token usage statistics |\n\n### Message Content Parameters\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--message` / `-q` | string | ✅* | User message content (simple single-turn conversation) |\n| `--messages` | JSON | ✅* | Complete messages JSON array (multi-turn conversation, mutually exclusive with `--message`) |\n| `--system` / `-s` | string | - | System prompt (sets AI role/persona; when used together with `--messages`, this parameter is ignored — system should be embedded in the messages JSON) |\n\n### Multimodal Parameters\n\n| Parameter | Type | Description |\n|------|------|------|\n| `--image-url` | string | Image URL (supported by both OpenAI and Gemini, size limit 70MB) |\n| `--audio-base64` | string | Base64-encoded audio data (Gemini only) |\n| `--audio-format` | mp3/wav | Audio format (default mp3) |\n| `--file-url` | string | File/video URL (Gemini only, size limit 70MB) |\n| `--file-name` | string | File name (used together with `--file-url`) |\n\n### Generation Control Parameters\n\n| Parameter | Type | Description |\n|------|------|------|\n| `--temperature` / `-t` | float | Output randomness (0~2; lower = more precise, higher = more creative; if not specified, server default is used, typically 0.7) |\n| `--max-tokens` | int | Maximum number of tokens to generate |\n| `--thinking` | flag | Enable reasoning/thinking mode (CoT); **Note: not supported by gpt-5.1/5.2/4o — a warning will be printed and the parameter will be automatically ignored** |\n| `--reasoning-effort` | string | Thinking level: `none`/`minimal`/`low`/`medium`/`high`/`xhigh` |\n| `--response-format` | string | Output format: `text` (default) / `json` (JSON object) / `json_schema` (same as `json`, fallback is json_object; for strict schema, describe the structure in messages) |\n\n### Function Calling Parameters\n\n| Parameter | Type | Description |\n|------|------|------|\n| `--tools` | JSON string | Tool definitions array (OpenAI `tools` field format), example: `'[{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"description\":\"Get weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]'` |\n| `--tool-choice` | string | Tool invocation strategy: `auto` (model decides automatically), `none` (disable tool calls), `required` (must call a tool), or JSON to specify a particular tool (e.g. `'{\"type\":\"function\",\"function\":{\"name\":\"get_weather\"}}'`) |\n\n### messages JSON Format\n\nFor multi-turn conversations, pass a complete JSON array via `--messages`:\n\n```json\n[\n  {\"role\": \"system\", \"content\": \"You are a professional assistant\"},\n  {\"role\": \"user\", \"content\": \"Hello\"},\n  {\"role\": \"assistant\", \"content\": \"Hello! How can I help you?\"},\n  {\"role\": \"user\", \"content\": \"Tell me about Tencent Cloud\"}\n]\n```\n\n### Multimodal Part Object Format\n\nWhen using `--messages`, the content of multimodal messages is a Part array:\n\n```json\n[\n  {\"role\": \"user\", \"content\": [\n    {\"type\": \"text\", \"text\": \"Please describe this image\"},\n    {\"type\": \"image_url\", \"image_url\": \"https://example.com/img.jpg\"}\n  ]}\n]\n```\n\n| Part type | Required Fields | Description |\n|-----------|---------|------|\n| `text` | `text` | Text content |\n| `image_url` | `image_url` | Image URL (≤70MB) |\n| `input_audio` | `input_audio.data` (Base64), `input_audio.format` (mp3/wav) | Audio |\n| `file` | `file_url` (≤70MB), optional `file_name` | File/video |\n\n### Error Code Reference\n\n| HTTP Status Code | Cause | Recommended Action |\n|-----------|------|---------|\n| 400 | Invalid request parameters | Check request body format; verify model/messages and other parameters are correct |\n| 401 | Invalid or unsynced token | Check if the token is correct; wait 1 minute after token creation |\n| 403 | Insufficient permissions / service suspended | Check account balance; confirm VOD service is activated |\n| 404 | Model not found | Check `--model` parameter; note that gemini-3-pro-preview has been taken offline |\n| 429 | Rate limit exceeded | Default RPM 10, TPM 100K; contact business team to adjust |\n| 500 | Internal server error | Retry later; contact technical support if the issue persists |\n| 502 | Gateway error | Retry later |\n| 503 | Service unavailable | Retry later; check service status |\n\n### models Subcommand\n\nNo additional parameters required; run directly to list all currently supported models:\n\n```bash\npython3 scripts/vod_aigc_chat.py models\n```\n\nCurrent supported models list:\n\n| Series | Model Name | Protocol | Thinking Support | Multimodal Input |\n|------|---------|------|--------------|-----------|\n| OpenAI | `gpt-5.5` | OpenAI completions | ✅ | Text, Image |\n| OpenAI | `gpt-5.4-pro` | Responses | ✅ | Text, Image |\n| OpenAI | `gpt-5.4` | OpenAI completions | ✅ | Text, Image |\n| OpenAI | `gpt-5.4-mini` | OpenAI completions | ✅ | Text, Image |\n| OpenAI | `gpt-5.3-codex` | Responses | ✅ | Text, Image |\n| OpenAI | `gpt-5.3-chat` | OpenAI completions | ✅ | Text, Image |\n| OpenAI | `gpt-5.2-chat` | OpenAI completions | ✅ | Text, Image |\n| OpenAI | `gpt-5.2` | OpenAI completions | ❌ | Text, Image |\n| OpenAI | `gpt-5.1` | OpenAI completions | ❌ | Text, Image |\n| OpenAI | `gpt-5.1-chat` | OpenAI completions | ✅ | Text, Image |\n| OpenAI | `gpt-5` | OpenAI completions | ✅ | Text, Image |\n| OpenAI | `gpt-5-mini` | OpenAI completions | ✅ | Text, Image |\n| OpenAI | `gpt-5-nano` | OpenAI completions | ✅ | Text, Image |\n| OpenAI | `gpt-4.1` | OpenAI completions | ✅ | Text, Image |\n| OpenAI | `gpt-4o` | OpenAI completions | ❌ | Text, Image |\n| Gemini | `gemini-3.5-flash` | OpenAI completions | ✅ | Text, Image, Audio, Video |\n| Gemini | `gemini-3.1-flash-lite` | OpenAI completions | ✅ | Text, Image, Audio, Video |\n| Gemini | `gemini-3.1-pro-preview` | OpenAI completions | ✅ | Text, Image, Audio, Video |\n| Gemini | `gemini-3-flash-preview` | OpenAI completions | ✅ | Text, Image, Audio, Video |\n| Gemini | `gemini-2.5-pro` | OpenAI completions | ✅ | Text, Image, Audio, Video |\n| Gemini | `gemini-2.5-flash` | OpenAI completions | ✅ | Text, Image, Audio, Video |\n| GK (Grok) | `gk-4-1-fast-reasoning` | OpenAI completions | ✅ | Text, Image |\n| GK (Grok) | `gk-4-20-non-reasoning` | OpenAI completions | ✅ | Text, Image |\n| GK (Grok) | `gk-4-20-reasoning` | OpenAI completions | ✅ | Text, Image |\n| GK (Grok) | `gk-4.3` | OpenAI completions | ✅ | Text, Image |\n| CD (Claude) | `cd-opus-4.7` | Anthropic | ✅ | Text, Image |\n| CD (Claude) | `cd-sonnet-4.6` | Anthropic | ✅ | Text, Image |\n| CD (Claude) | `cd-opus-4.6` | Anthropic | ✅ | Text, Image |\n| CD (Claude) | `cd-opus-4.5` | Anthropic | ✅ | Text, Image |\n| CD (Claude) | `cd-haiku-4.5` | Anthropic | ✅ | Text, Image |\n\n> 📌 CD series uses Anthropic protocol (`/v1/messages` endpoint, `x-api-key` auth). Other models use OpenAI-compatible protocol (`/v1/chat/completions`, `Bearer` auth).\n> ⚠️ Rate limit: default RPM 30 requests/minute, TPM 300K tokens/minute.\n\n---\n\n\n## Usage Examples\n### 1 Basic Chat\n\n#### Non-streaming Chat (Wait for Full Response)\n```bash\npython3 scripts/vod_aigc_chat.py chat \\\n    --model gpt-5.1 \\\n    --message \"Introduce Tencent Cloud in one sentence\"\n```\n\n#### Streaming Output (Progressive Display)\n```bash\npython3 scripts/vod_aigc_chat.py stream \\\n    --model gemini-2.5-flash \\\n    --message \"Write a poem about cloud computing\"\n```\n\n#### Streaming Output and Save to File\n```bash\npython3 scripts/vod_aigc_chat.py stream \\\n    --model gemini-2.5-flash \\\n    --message \"Explain in detail the differences between H.264 and H.265 encoding formats\" \\\n    --output-file /output/vod-chat/stream_result.txt\n```\n\n#### List All Supported Models\n```bash\npython3 scripts/vod_aigc_chat.py models\n```\n\n---\n\n### 2 Advanced Usage\n\n#### With System Prompt\n```bash\npython3 scripts/vod_aigc_chat.py chat \\\n    --model gpt-5.1 \\\n    --system \"You are a professional video processing expert\" \\\n    --message \"What is the difference between H.264 and H.265?\"\n```\n\n#### Enable Reasoning/Thinking Mode (for Complex Logic/Math)\n```bash\npython3 scripts/vod_aigc_chat.py chat \\\n    --model gpt-5.4 \\\n    --thinking \\\n    --message \"Please analyze the time complexity of the following code...\"\n```\n\n#### Specify Thinking Level\n```bash\npython3 scripts/vod_aigc_chat.py chat \\\n    --model gemini-2.5-pro \\\n    --reasoning-effort high \\\n    --message \"Please solve a complex mathematical proof problem\"\n```\n\n#### Image Understanding (Multimodal)\n```bash\npython3 scripts/vod_aigc_chat.py chat \\\n    --model gpt-5.1 \\\n    --message \"Please describe the content of this image\" \\\n    --image-url \"https://example.com/image.jpg\"\n```\n\n#### Video/File Understanding (Gemini Models)\n```bash\npython3 scripts/vod_aigc_chat.py chat \\\n    --model gemini-2.5-flash \\\n    --message \"Please summarize the main content of this video\" \\\n    --file-url \"https://example.com/video.mp4\" \\\n    --file-name \"video.mp4\"\n```\n\n#### Audio Understanding (Gemini Models)\n```bash\npython3 scripts/vod_aigc_chat.py chat \\\n    --model gemini-2.5-flash \\\n    --message \"Please transcribe this audio\" \\\n    --audio-base64 \"<base64-encoded audio data>\" \\\n    --audio-format mp3\n```\n\n#### Multi-turn Conversation (JSON Format)\n```bash\npython3 scripts/vod_aigc_chat.py chat \\\n    --model gpt-5.1 \\\n    --messages '[\n        {\"role\":\"system\",\"content\":\"You are a video expert\"},\n        {\"role\":\"user\",\"content\":\"What is bitrate?\"},\n        {\"role\":\"assistant\",\"content\":\"Bitrate is the amount of data transmitted per unit time...\"},\n        {\"role\":\"user\",\"content\":\"Is higher bitrate always better?\"}\n    ]'\n```\n\n#### Control Output Randomness\n```bash\npython3 scripts/vod_aigc_chat.py chat \\\n    --model gpt-5.1 \\\n    --temperature 0.2 \\\n    --message \"Please calculate precisely: 1024 * 768 = ?\"\n```\n\n---\n\n### 3 Function Calling\n\n#### Register Weather Query Tool, tool_choice=auto\n```bash\npython3 scripts/vod_aigc_chat.py chat \\\n    --model gpt-5.1 \\\n    --message \"What is the weather like in Beijing today?\" \\\n    --tools '[{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"description\":\"Get weather for a specified city\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\",\"description\":\"City name\"}},\"required\":[\"city\"]}}}]' \\\n    --tool-choice auto\n```\n\n#### Register Tool but Disable Invocation (tool_choice=none)\n```bash\npython3 scripts/vod_aigc_chat.py chat \\\n    --model gpt-5.1 \\\n    --message \"What is the weather like in Beijing today?\" \\\n    --tools '[{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"description\":\"Get weather for a specified city\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]' \\\n    --tool-choice none\n```\n\n#### dry-run Preview Request Body (with tools and tool_choice)\n```bash\npython3 scripts/vod_aigc_chat.py chat \\\n    --model gpt-5.1 \\\n    --message \"What is the weather like in Beijing today?\" \\\n    --tools '[{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"description\":\"Get weather\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"]}}}]' \\\n    --tool-choice auto \\\n    --dry-run\n```\n\n---\n\n### 4 Output Control\n\n#### Full Response in JSON Format (with Usage Statistics)\n```bash\npython3 scripts/vod_aigc_chat.py chat \\\n    --model gpt-5.1 \\\n    --message \"Hello\" \\\n    --json\n```\n\n#### Structured JSON Schema Output\n```bash\npython3 scripts/vod_aigc_chat.py chat \\\n    --model gpt-5.1 \\\n    --message \"List 3 video encoding formats, each with a name and description field\" \\\n    --response-format json_schema\n```\n\n#### Save Result to File\n```bash\npython3 scripts/vod_aigc_chat.py chat \\\n    --model gpt-5.1 \\\n    --message \"Write an article about AI\" \\\n    --output-file /output/vod-chat/result.json\n```\n\n#### Hide Token Usage\n```bash\npython3 scripts/vod_aigc_chat.py chat \\\n    --model gpt-5.1 \\\n    --message \"Hello\" \\\n    --no-usage\n```\n\n---\n\nFile v1.1.3:references/vod_aigc_image.md\n\n# vod_aigc_image — Detailed Parameters and Examples\n> This file is generated by splitting references, corresponding script: `scripts/vod_aigc_image.py`\n\n### ⚠️ Common Parameter Errors\n\n| Incorrect Usage | Correct Usage | Description |\n|---------|---------|------|\n| `--model hunyuan-3.0` | `--model Hunyuan --model-version 3.0` | Model name and version number are separate |\n| `--model vidu-q2` or `--model vidu` | `--model Vidu --model-version q2` | **Model name must be capitalized (`Vidu`), version number uses a separate `--model-version q2`** |\n| `--aspect-ratio 16:9` | `--output-aspect-ratio 16:9` | Image generation output parameters use the `output-` prefix |\n| `--resolution 2K` | `--output-resolution 2K` | Image generation resolution parameters use the `output-` prefix |\n| `--class-id 10` | `--output-class-id 10` | **Image generation output class ID must use `--output-class-id`, not `--class-id`** |\n| `--file-ids id1,id2` | `--file-infos '[{\"Type\":\"File\",\"FileId\":\"id1\"}]'` | Multiple reference images use `--file-infos` JSON array |\n| `--person-generation Disallowed` | `--output-person-generation Disallowed` | Disabling person generation uses the `output-` prefix |\n| `--num-images 4` or `-n 4` | `--output-image-count 4` | **Multi-image output (OG 1-8, Kling 1-9)**; uses the `output-` prefix |\n| `--format png` | `--output-format png` | **Specify output format (OG)**: `jpeg`/`png`; uses the `output-` prefix |\n| `--mask-url ...` or `--mask-file-id ...` | `--file-infos '[{\"Type\":\"File\",\"FileId\":\"src\"},{\"Type\":\"File\",\"FileId\":\"mask\",\"ReferenceType\":\"mask\"}]'` | **OG mask editing**: first image is the source to edit, second is the mask (`ReferenceType=\"mask\"`) |\n\n## Parameter Description\n\n### General Parameters\n\n| Parameter | Type | Description |\n|------|------|------|\n| `--sub-app-id` | int | Sub-application ID (can also be set via environment variable `TENCENTCLOUD_VOD_SUB_APP_ID`) |\n| `--region` | string | Tencent Cloud region (default: `ap-guangzhou`) |\n| `--json` | flag | JSON format output |\n| `--dry-run` | flag | Only prints request parameter preview without sending the request |\n\n### create Parameters (Create Image Generation Task)\n\n#### Model Parameters (Required)\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--model` | string | ✅ | Model name (Hunyuan/Qwen/Vidu/Kling/MJ/GG) |\n| `--model-version` | string | - | Model version (uses default version if not specified) |\n\n#### Prompt Parameters\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--prompt` | string | ✅* | Prompt for generating the image (required when no reference image is provided) |\n| `--negative-prompt` | string | - | Prompt to prevent the model from generating certain content (negative prompt) |\n| `--enhance-prompt` | string | - | Whether to automatically optimize the prompt (Enabled/Disabled) |\n\n#### Reference Image Parameters\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--file-id` | string | - | FileId of the reference image (single image only; use `--file-infos` for multiple) |\n| `--file-url` | string | - | URL of the reference image (single image only; use `--file-infos` for multiple; mutually exclusive with `--file-id`) |\n| `--file-text` | string | - | Description of the reference image (only valid for GG 2.5/3.0; only takes effect when using `--file-id` or `--file-url`; when using `--file-infos`, embed Text inside each JSON element) |\n| `--reference-type` | string | - | Reference type for a single reference image (`mask`), used together with `--file-id`/`--file-url`. **Currently only used for GPT-Image2 (OG) mask editing**. For multiple reference images, embed `ReferenceType` inside `--file-infos` |\n| `--file-infos` | string | - | JSON array of multiple reference images, format: `[{\"Type\":\"Url\",\"Url\":\"...\",\"Text\":\"description\",\"ReferenceType\":\"mask\"}]`; ReferenceType is used for OG mask editing |\n\n> **Reference Image Count Limits**:\n> - GG 2.5/3.0: up to 3 images\n> - Vidu q2: up to 7 images\n> - Other models: only 1 image or not supported\n\n#### Output Configuration Parameters (Output Config)\n\n| Parameter | Type | Description |\n|------|------|------|\n| `--output-storage-mode` | string | Storage mode (Permanent: permanent storage, Temporary: temporary storage, default: Temporary) |\n| `--output-media-name` | string | Output filename (max 64 characters) |\n| `--output-class-id` | int | Class ID (default 0, indicating other categories) |\n| `--output-expire-time` | string | Expiration time of the output file (ISO 8601 format, e.g. 2026-12-31T23:59:59+08:00) |\n| `--output-resolution` | string | Resolution of the generated image (supported resolutions vary by model) |\n| `--output-aspect-ratio` | string | Aspect ratio of the generated image (supported aspect ratios vary by model) |\n| `--output-person-generation` | string | Whether to allow person or face generation (`AllowAdult`: allow, `Disallowed`: disallow) |\n| `--input-compliance-check` | string | Whether to enable compliance check on input content (Enabled/Disabled) |\n| `--output-compliance-check` | string | Whether to enable compliance check on output content (Enabled/Disabled) |\n| `--output-image-count` | int | Number of images to generate (**OG 1-8, Kling 1-9**, including Kling outpainting; not supported by other models) |\n| `--output-format` | string | Output image format (**only supported by GPT-Image2 / OG**): `jpeg` / `png` |\n| `--output-logo-add` | string | Whether to add logo watermark: `Enabled` / `Disabled` (only some models; OG behavior depends on the server) |\n\n#### Other Optional Parameters\n\n| Parameter | Type | Description |\n|------|------|------|\n| `--scene-type` | string | Scene type (when ModelName is Hunyuan: `3d_panorama` for panoramic image; not supported for other models) |\n| `--seed` | int | Model random seed (specify to reproduce results) |\n| `--input-region` | string | Region information of the input file (Mainland: domestic, Oversea: overseas, default: Mainland) |\n| `--session-id` | string | Identifier for deduplication (max 50 characters; returns an error if duplicated within three days) |\n| `--session-context` | string | Source context for passing through user request information (max 1000 characters) |\n| `--tasks-priority` | int | Task priority (higher value = higher priority, range: -10 to 10) |\n| `--ext-info` | string | Reserved field for special use (JSON string format) |\n| `--no-wait` | flag | Submit task only without waiting for result (default: auto-wait) |\n| `--max-wait` | int | Maximum wait time (seconds, default: 600) |\n\n### query Parameters (Query Task Status)\n\n> ⚠️ **Mandatory Rule**: When querying AIGC image generation task details, **this command must be called**. Fabricating or making up JSON response content is prohibited. When the user requests JSON format output, the `--json` parameter must be added.\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--task-id` | string | ✅ | Task ID |\n| `--sub-app-id` | int | - | Sub-application ID |\n| `--region` | string | - | Region (default: `ap-guangzhou`) |\n| `--no-wait` | flag | - | Query status only without waiting for completion (default: auto-wait) |\n| `--poll-interval` | int | - | Polling interval (seconds, default: 10) |\n| `--max-wait` | int | - | Maximum wait time (seconds, default: 600) |\n| `--json` | flag | - | JSON format output |\n| `--dry-run` | flag | - | Preview request parameters without actual execution |\n\n### models Parameters (View Supported Models)\n\nThis command has no parameters and is used to list all supported models, versions, resolutions, aspect ratios, and more.\n\n### Supported Model Features\n\n> 📌 All capabilities are confirmed via interface testing; max reference image counts, resolutions, and aspect ratios have all been verified by real interface submissions.\n\n| Model | Version | Aspect Ratio | Resolution | Reference Images | Features |\n|------|------|--------|--------|--------|------|\n| Hunyuan | 3.0, 3d_2.0 | AspectRatio not supported (3.0 uses ExtInfo `size`); 3d_2.0 panoramic image is fixed | 3.0: width and height in [512, 2048], product ≤ 1024×1024, **via ExtInfo `size`**, default 1024×1024; 3d_2.0 panoramic: 720P–4K | 0–3 images | Hunyuan model; **3.0 custom resolution example**: `--ext-info '{\"AdditionalParameters\": \"{\\\"size\\\":\\\"728x1024\\\"}\"}'`; **3d_2.0 + `--scene-type 3d_panorama` for 360° panoramic image** (Hunyuan World Model, both text-to and image-to) |\n| Qwen | 0925 | AspectRatio not supported | Free width/height, output pixels in [512×512, 2048×2048], default 1024×1024, **via ExtInfo `width`/`height`** | 0–1 image | Qwen model; **custom resolution example**: `--ext-info '{\"AdditionalParameters\": \"{\\\"width\\\":1024, \\\"height\\\":1024}\"}'` |\n| Vidu | q2 | 16:9, 9:16, 1:1, 3:4, 4:3, 21:9, 2:3, 3:2 | 1080P, 2K, 4K (default 1080P) | 0–7 images | Vidu model; text-to-image / reference-image generation |\n| Kling | 2.1 | 16:9, 9:16, 1:1, 4:3, 3:4, 3:2, 2:3, 21:9 | 1K, 2K (default 1K) | 0–4 images | Kling 2.1 |\n| Kling | 3.0 | 16:9, 9:16, 1:1, 4:3, 3:4, 3:2, 2:3, 21:9 | 1K, 2K (default 1K) | 0–1 image | Kling 3.0 |\n| Kling | 3.0-Omni | 16:9, 9:16, 1:1, 4:3, 3:4, 3:2, 2:3, 21:9, **auto** | 1K, 2K, **4K** (default 1K) | 0–10 images | Kling 3.0-Omni; **multi-subject generation supported**; output supports 1-9 images at once (`--output-image-count`) |\n| Kling | **O1** | 16:9, 9:16, 1:1, 4:3, 3:4, 3:2, 2:3, 21:9, **auto** | 1K, 2K, **4K** (default 1K) | 0–10 images | Kling O1; similar to 3.0-Omni; confirmed by interface testing |\n| Kling | **scene** | N/A (controlled via 4 ratio fields in `--ext-info`) | Proportional to original, area ≤ 3× original | **1 image only (required)** | **Dedicated outpainting version**, not a regular generation version; must be paired with `--scene-type image_expand`; see §10.3 |\n| MJ | v8.1, v7 | Specified in prompt (e.g. `--ar 16:9`) | Specified in prompt (e.g. `--q 2`) | 0–3 images | Midjourney model; **interface name is `MJ`** (`Midjourney` is rejected); v8.1 is the latest |\n| MJ | **v8.2** | Specified in prompt | Specified in prompt (`--q` supports only `1`/`4`) | 0–3 images | Latest MJ version (added 2026-08-03); `--q` quality tiers differ from v7/niji_7 (which support `1`/`2`/`4`) |\n| MJ | **niji_7** | Specified in prompt | Specified in prompt (`--q` supports `1`/`2`/`4`) | 0–3 images | Anime-focused model (added 2026-07-23); still fixed at 4 images per generation |\n| GG | 2.5 | 1:1, 3:2, 2:3, 3:4, 4:3, 4:5, 5:4, 9:16, 16:9, 21:9 | 1K, 2K, 4K (default 1K) | 0–3 images | nano banana (GG 2.5); **interface name is `GG`; historical alias `GEM` is also accepted** |\n| GG | 3.0 | 1:1, 3:2, 2:3, 3:4, 4:3, 4:5, 5:4, 9:16, 16:9, 21:9 | 1K, 2K, 4K (default 1K) | 0–14 images | nano banana pro (GG 3.0); supports outpainting |\n| GG | 3.1 | 1:1, 3:2, 2:3, 3:4, 4:3, 4:5, 5:4, 9:16, 16:9, 21:9, **1:4, 4:1, 1:8, 8:1** | **512**, 1K, 2K, 4K (default 1K) | 0–14 images | nano2 (GG 3.1); extra 512 resolution; extra 4 extreme aspect ratios |\n| GG | **3.1-lite** | 1:1, 3:2, 2:3, 3:4, 4:3, 4:5, 5:4, 9:16, 16:9, 21:9 | 1K (**native output only**), 2K/4K via super-resolution | 0–14 images | nano banana 2 lite (GG 3.1-lite, added 2026-07-08); lightweight version, native 1K only, 2K/4K require super-resolution upscaling |\n| SI | **4.0** | Specified in prompt | 1K, 2K, 4K (default 1K) | 0–14 images | Seedream 4.0; **multi-image output supported**: prompt specifies count + `--ext-info '{\"AdditionalParameters\": \"{\\\"sequential_image_generation\\\":\\\"auto\\\"}\"}'` |\n| SI | 4.5 | Specified in prompt | 2K, 4K (default 2K) | 0–14 images | Seedream 4.5 |\n| SI | 5.0-lite | Specified in prompt | 2K, **3K**, 4K (default 2K) | 0–14 images | Seedream 5.0-lite; extra 3K resolution |\n| SI | **5.0-pro** | Specified in prompt | 1K, 2K, 4K (default 1K) | 0–14 images | Seedream 5.0-pro (added 2026-07-16); flagship version with better quality |\n| OG | image2_low, image2_medium, image2_high | 1:1, 3:2, 2:3, 3:4, 4:3, 16:9, 9:16, 21:9, 9:21 | 1K, 2K, 4K (default 1K) | 0–16 images | **GPT-Image2**: strong multilingual text rendering; supports jpeg/png output (`--output-format`); supports 1–8 images per task (`--output-image-count`); supports mask editing (`ReferenceType=mask`); supports custom size (multiple of 16, via `--ext-info`); transparent background not supported; billed per input image |\n| Jimeng | 4.0 | AspectRatio not supported | Via ExtInfo `width`/`height`, range [1024×1024, 4096×4096] | 0–10 images | Jimeng 4.0; **custom resolution example**: `--ext-info '{\"AdditionalParameters\": \"{\\\"width\\\":1920, \\\"height\\\":1080}\"}'`; strong stylized output |\n\n### File Infos Parameter Structure\n\n`FileInfos` is an array used to pass reference image information. Each element contains the following fields:\n\n| Field | Type | Required | Description |\n|------|------|------|------|\n| `Type` | string | ✅ | Input type (File: VOD media file, Url: accessible URL) |\n| `FileId` | string | -* | Media file ID of the image file (required when Type=File) |\n| `Url` | string | -* | Accessible file URL (required when Type=Url) |\n| `Text` | string | - | Description of the input image (only valid for GG 2.5/3.0) |\n| `ReferenceType` | string | - | Reference type: `mask` indicates this image is used as a mask (**only used for GPT-Image2 / OG mask editing**) |\n\n### Output Config Parameter Structure\n\n`OutputConfig` is an object used to configure various parameters for the output media file:\n\n| Field | Type | Default | Description |\n|------|------|--------|------|\n| `StorageMode` | string | Temporary | Storage mode (Permanent/Temporary) |\n| `MediaName` | string | - | Output filename |\n| `ClassId` | int | 0 | Class ID |\n| `ExpireTime` | string | - | Expiration time (ISO 8601 format) |\n| `Resolution` | string | - | Resolution |\n| `AspectRatio` | string | - | Aspect ratio |\n| `PersonGeneration` | string | - | Whether to allow person generation (AllowAdult/Disallowed) |\n| `InputComplianceCheck` | string | - | Input compliance check (Enabled/Disabled) |\n| `OutputComplianceCheck` | string | - | Output compliance check (Enabled/Disabled) |\n| `OutputImageCount` | int | - | Number of images to generate (OG 1-8, Kling 1-9, including Kling outpainting) |\n| `OutputFormat` | string | - | Output format (only supported by GPT-Image2 / OG): jpeg / png |\n| `LogoAdd` | string | - | Whether to add logo watermark: Enabled / Disabled (only some models) |\n\n### Task Status Description\n\n| Status | Description |\n|------|------|\n| WAIT | Waiting |\n| RUNNING | Processing |\n| FINISH | Completed |\n| FAIL | Failed |\n\n### API Interface Mapping\n\n| Function | API Interface | Documentation Link |\n|------|---------|---------|\n| Create AIGC image generation task | `CreateAigcImageTask` | https://cloud.tencent.com/document/api/266/126240 |\n| Query task status | `DescribeTaskDetail` | https://cloud.tencent.com/document/api/266/33431 |\n\n### Error Code Description\n\n| Error Type | Cause | Recommended Action |\n|---------|------|---------|\n| Model version not supported | The specified Model Version does not exist | Check the supported model list and select the correct version |\n| Resolution not supported | The specified Resolution is not supported by the model | Check the model features table and select a supported resolution |\n| Aspect ratio not supported | The specified Aspect Ratio is not supported by the model | Check the model features table and select a supported aspect ratio |\n| Reference image count exceeded | The number of reference images exceeds the model limit | Check the model features table and stay within the allowed range |\n| Prompt required | No Prompt provided and no reference image | Add the `--prompt` parameter or provide a reference image |\n| Sub AppId required | Sub-application ID not specified | Add the `--sub-app-id` parameter or set the environment variable |\n| Failed to fetch task | URL is not accessible | Ensure the URL is publicly accessible; Dash format is not currently supported |\n\n---\n\n## Usage Examples\n\n\n### 1 Basic Text-to-Image\n\n#### Hunyuan Model Image Generation\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model Hunyuan \\\n    --prompt \"A cute little cat playing in a meadow under bright sunshine\"\n```\n\n#### GG Model Image Generation (Specify Resolution and Aspect Ratio)\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model GG \\\n    --model-version 2.5 \\\n    --prompt \"a beautiful sunset over the ocean\" \\\n    --output-resolution 2K \\\n    --output-aspect-ratio 16:9\n```\n\n#### Kling Model Image Generation (Permanent Storage)\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model Kling \\\n    --model-version 2.1 \\\n    --prompt \"Cyberpunk-style futuristic cityscape at night\" \\\n    --output-resolution 2K \\\n    --output-aspect-ratio 16:9 \\\n    --output-storage-mode Permanent\n```\n\n### 2 Image-to-Image (Reference Image)\n\n#### Using FileId as Reference Image\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model Hunyuan \\\n    --prompt \"Change to a winter snow scene\" \\\n    --file-id 3704211509819\n```\n\n#### Vidu q2 Model + Reference Image FileId (Style Reference)\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model Vidu \\\n    --model-version q2 \\\n    --prompt \"Chinese ink painting style\" \\\n    --file-id 5145403721231891303\n```\n\n#### Using URL as Reference Image\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model Kling \\\n    --model-version 2.1 \\\n    --prompt \"Change the image style to watercolor painting\" \\\n    --file-url \"https://example.com/reference.jpg\"\n```\n\n#### GG Multiple Reference Images (Up to 3)\n\n> 🚨 **Multiple reference images must use the `--file-infos` JSON array**. There is no `--file-ids` parameter; `--file-id` only supports a single image.\n\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model GG \\\n    --model-version 3.0 \\\n    --prompt \"Blend the styles of these three images\" \\\n    --file-infos '[{\"Type\":\"File\",\"FileId\":\"3704211509819\"},{\"Type\":\"File\",\"FileId\":\"3704211509820\"},{\"Type\":\"File\",\"FileId\":\"3704211509821\"}]'\n```\n\n#### GG Multiple Reference Images (With Text Descriptions)\n\n> ⚠️ When using `--file-infos`, the `--file-text` parameter is invalid. Text descriptions must be embedded in the `\"Text\"` field of each JSON element.\n\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model GG \\\n    --model-version 3.0 \\\n    --prompt \"Blend the styles of these three images\" \\\n    --file-infos '[{\"Type\":\"File\",\"FileId\":\"3704211509819\",\"Text\":\"Style of the first image\"},{\"Type\":\"File\",\"FileId\":\"3704211509820\",\"Text\":\"Composition of the second image\"},{\"Type\":\"File\",\"FileId\":\"3704211509821\",\"Text\":\"Color tone of the third image\"}]'\n```\n\n### 3 Enable Prompt Enhancement\n\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model Hunyuan \\\n    --prompt \"landscape painting\" \\\n    --enhance-prompt Enabled\n```\n\n### 4 Default Wait for Task Completion\n\n```bash\n# Submit task and automatically wait for completion (default max wait: 600 seconds)\npython3 scripts/vod_aigc_image.py create \\\n    --model Hunyuan \\\n    --prompt \"A puppy\"\n\n# No wait, submit task only\npython3 scripts/vod_aigc_image.py create \\\n    --model Hunyuan \\\n    --prompt \"A puppy\" \\\n    --no-wait\n```\n\n### 5 Query Task Status\n\n```bash\n# Query and wait for task completion (default auto-wait)\npython3 scripts/vod_aigc_image.py query --task-id task_xxx\n\n# Query current status only, no waiting\npython3 scripts/vod_aigc_image.py query --task-id task_xxx --no-wait\n```\n\n### 6 Disable Person Generation (output-person-generation)\n\n> ⚠️ **Parameter values are strictly case-sensitive**: `AllowAdult` (allow adult persons), `Disallowed` (disallow person generation). Do not use lowercase forms such as `disable` or `disallowed`.\n\n```bash\n# Disallow person generation\npython3 scripts/vod_aigc_image.py create \\\n    --model Hunyuan \\\n    --prompt \"Beautiful mountain and river scenery\" \\\n    --output-person-generation Disallowed \\\n    --sub-app-id 1500046725\n\n# Allow adult person generation\npython3 scripts/vod_aigc_image.py create \\\n    --model Hunyuan \\\n    --prompt \"Portrait of a person\" \\\n    --output-person-generation AllowAdult\n```\n\n### 7 View Supported Models\n\n```bash\npython3 scripts/vod_aigc_image.py models\n```\n\n---\n\n## 8 GPT-Image2 (OG) Advanced Features\n\n> 📌 **Mapping**: Skill `OG` model = GPT-Image2. The Tencent Cloud API uses `ModelName=OG` while the marketing name is GPT-Image2. Three quality tiers: `image2_low` / `image2_medium` / `image2_high`, in increasing quality / latency / cost.\n\n> 💪 **Core Capabilities**: multilingual text rendering (zh/en/ja/ko/ar), multi-image output (1-8), custom size (multiple of 16), jpeg/png output, mask editing, up to 16 reference images.\n\n> ⚠️ **Not Supported**: transparent background (downgrade to gpt-image-1.5 if required).\n\n### 8.1 Basic Text-to-Image (corresponds to doc 3.14.2)\n\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model OG --model-version image2_medium \\\n    --prompt \"A futuristic city at sunset, photorealistic\" \\\n    --output-aspect-ratio 16:9 \\\n    --output-resolution 2K \\\n    --output-storage-mode Temporary \\\n    --sub-app-id 1308104797\n```\n\n### 8.2 Multi-Image Output OutputImageCount (corresponds to doc 3.14.3)\n\n> ⚠️ **Range**: OG 1-8, Kling 1-9 (including Kling outpainting).\n\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model OG --model-version image2_low \\\n    --prompt \"Cute cartoon corgi, multiple variations\" \\\n    --output-aspect-ratio 1:1 \\\n    --output-image-count 4 \\\n    --output-storage-mode Temporary \\\n    --sub-app-id 1308104797\n```\n\nThe length of `Output.FileInfos` array equals `OutputImageCount`.\n\n### 8.3 Custom size / auto mode (corresponds to doc 3.14.4, via `--ext-info`)\n\n> ⚠️ **size constraints**: width and height must be multiples of 16; longest edge ≤ 3840; total pixels between 655,360 and 8,294,400.\n> In auto mode the model decides the size; do NOT pass `auto` to `--output-aspect-ratio`.\n\n```bash\n# Custom size: 2000x1104\npython3 scripts/vod_aigc_image.py create \\\n    --model OG --model-version image2_medium \\\n    --prompt \"Wide cinematic banner\" \\\n    --ext-info '{\"AdditionalParameters\":\"{\\\"size\\\":\\\"2000x1104\\\"}\"}' \\\n    --output-storage-mode Temporary \\\n    --sub-app-id 1308104797\n\n# auto mode\npython3 scripts/vod_aigc_image.py create \\\n    --model OG --model-version image2_medium \\\n    --prompt \"A scene that decides its own aspect ratio\" \\\n    --ext-info '{\"AdditionalParameters\":\"{\\\"size\\\":\\\"auto\\\"}\"}' \\\n    --sub-app-id 1308104797\n```\n\n### 8.4 Mask Editing ReferenceType=mask (corresponds to doc 3.14.6)\n\n> 🚨 **Mandatory rule**: the first FileInfos element is the **source image to edit**, the second element is the **mask image** with `\"ReferenceType\":\"mask\"`. The mask is a PNG: white area = region to replace, transparent/black area = preserved.\n\n```bash\n# Multiple reference images + mask (recommended)\npython3 scripts/vod_aigc_image.py create \\\n    --model OG --model-version image2_medium \\\n    --prompt \"Replace the masked area with a bright red flower\" \\\n    --file-infos '[{\"Type\":\"File\",\"FileId\":\"<source_id>\"},{\"Type\":\"File\",\"FileId\":\"<mask_id>\",\"ReferenceType\":\"mask\"}]' \\\n    --output-storage-mode Temporary \\\n    --sub-app-id 1308104797\n\n# Single reference image + mask (passing mask only)\npython3 scripts/vod_aigc_image.py create \\\n    --model OG --model-version image2_medium \\\n    --prompt \"Generate content matching this mask\" \\\n    --file-id <mask_id> --reference-type mask \\\n    --sub-app-id 1308104797\n```\n\n### 8.5 Specify OutputFormat (corresponds to doc 3.14.7)\n\n```bash\n# jpeg\npython3 scripts/vod_aigc_image.py create \\\n    --model OG --model-version image2_medium \\\n    --prompt \"A red apple on a wooden table\" \\\n    --output-format jpeg \\\n    --sub-app-id 1308104797\n\n# png\npython3 scripts/vod_aigc_image.py create \\\n    --model OG --model-version image2_high \\\n    --prompt \"Detailed product render\" \\\n    --output-format png \\\n    --sub-app-id 1308104797\n```\n\n### 8.6 Transparent Layer (corresponds to doc 3.14.5)\n\n> ⚠️ **GPT-Image2 does not support transparent backgrounds** (the doc explicitly states: \"downgrade to gpt-image-1.5 for transparent background\"). The current VOD integration is the image2 series, so do NOT pass transparent-background requests to OG; the result will not be transparent. If you really need transparency, contact Tencent Cloud business team to confirm the image1.5 access path.\n\n### 8.7 Quality Tier Recommendation\n\n| Version | Speed | Cost | Recommended Scenarios |\n|------|------|------|---------|\n| `image2_low` | Fastest | Lowest | Quick drafts, prompt iteration, batch testing |\n| `image2_medium` | Mid | Mid | Most daily production / general design (recommended default) |\n| `image2_high` | Slowest | Highest | Final deliverables, print-ready, high-precision product images, complex text / small fonts |\n\n---\n\n## 9 Hunyuan 3D Panoramic Image (Hunyuan World Model)\n\n> 📌 **Capability**: Hunyuan `3d_2.0` together with `--scene-type 3d_panorama` generates **360° ERP panoramic images**, suitable for VR / AR, film backplates, virtual broadcast etc. Input can be plain text (text-to-3D) or a single reference image (image-to-3D). The video-side `3d_scene` (3D model / 3DGS / mesh) belongs to the video script and is out of scope here.\n\n### 9.1 Text-to-3D Panoramic Image\n\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model Hunyuan --model-version 3d_2.0 \\\n    --scene-type 3d_panorama \\\n    --prompt \"An ancient mountain temple courtyard, snow scene, golden hour\" \\\n    --output-storage-mode Temporary \\\n    --sub-app-id 1308104797\n```\n\n### 9.2 Image-to-3D Panoramic Image (single reference image)\n\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model Hunyuan --model-version 3d_2.0 \\\n    --scene-type 3d_panorama \\\n    --prompt \"Preserve the scene structure, expand to a 360° panorama\" \\\n    --file-url \"https://example.com/photo.jpg\" \\\n    --output-storage-mode Temporary \\\n    --sub-app-id 1308104797\n```\n\n### 9.3 Difference vs. plain Hunyuan 3.0\n\n| Dimension | Hunyuan 3.0 (default) | Hunyuan 3d_2.0 |\n|---|---|---|\n| Use case | General text-to-image / image-to-image | 360° panoramic image |\n| `--scene-type` | omit | **Required** `3d_panorama` |\n| Output | Regular image | ERP panoramic image |\n\n---\n\n## 10 Kling Advanced Features\n\n> 📌 **Capability**: Kling image-side supports plain text/image generation, **multi-image output (1-9)**, and **outpainting**. Kling motion control / digital human / lip sync / video edit belong to **video generation** and should use `vod_aigc_video.py`.\n\n### 10.1 Plain Text-to-Image / Image-to-Image\n\n```bash\n# Text-to-image\npython3 scripts/vod_aigc_image.py create \\\n    --model Kling --model-version 2.1 \\\n    --prompt \"Cyberpunk futuristic city night scene\" \\\n    --output-resolution 2k --output-aspect-ratio 16:9 \\\n    --sub-app-id 1308104797\n\n# Image-to-image (FileId or URL, pick one)\npython3 scripts/vod_aigc_image.py create \\\n    --model Kling --model-version 3.0-Omni \\\n    --file-id 5145403721231891303 \\\n    --prompt \"Change the style to watercolor\" \\\n    --sub-app-id 1308104797\n```\n\n### 10.2 Generate multiple images at once (OutputImageCount 1-9)\n\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model Kling --model-version 3.0 \\\n    --prompt \"Realistic landscape painting\" \\\n    --output-image-count 5 --output-aspect-ratio 16:9 \\\n    --sub-app-id 1308104797\n```\n\nThe length of `Output.FileInfos` array equals `OutputImageCount`.\n\n### 10.3 Kling Outpainting\n\n> 🚨 **Mandatory rule** (per official integration guide §3.9.11, verified against document spec):\n> - **`--model-version` must be the fixed value `scene`** (not 2.1/3.0/3.0-Omni/O1 — this is a dedicated special version identifier for outpainting)\n> - **`--scene-type` must be `image_expand`**\n> - Only a **single** reference image is supported (`--file-id` or `--file-url`, either one; **`--file-infos` multi-image is NOT supported**)\n> - All 4 directional expansion ratios must be passed via `--ext-info` (all 4 fields are **required**)\n> - `--prompt` is optional, max 2500 characters\n> - `--output-image-count` range `[1, 9]`, default 1\n> - `--output-storage-mode Permanent` is recommended\n\n**ratio fields** (range `[0, 2]`; new image total area cannot exceed 3× the original):\n\n| Field | Required | Meaning |\n|---|---|---|\n| `up_expansion_ratio` | ✅ | Upward expansion as a multiplier of the original **height** (e.g. original height 20, value 0.1 → expand up by 20×0.1=2) |\n| `down_expansion_ratio` | ✅ | Downward expansion as a multiplier of the original **height** |\n| `left_expansion_ratio` | ✅ | Leftward expansion as a multiplier of the original **width** |\n| `right_expansion_ratio` | ✅ | Rightward expansion as a multiplier of the original **width** |\n\n**Command template (URL input)**:\n\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model Kling --model-version scene \\\n    --scene-type image_expand \\\n    --file-url \"https://example.com/source.jpg\" \\\n    --prompt \"Fill the surrounding sky and sea (≤2500 chars)\" \\\n    --output-image-count 2 \\\n    --output-storage-mode Permanent \\\n    --ext-info '{\"AdditionalParameters\":\"{\\\"up_expansion_ratio\\\":0.1,\\\"down_expansion_ratio\\\":0.2,\\\"left_expansion_ratio\\\":0.3,\\\"right_expansion_ratio\\\":0.4}\"}' \\\n    --sub-app-id 1308104797\n```\n\n**Command template (FileId input)**:\n\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model Kling --model-version scene \\\n    --scene-type image_expand \\\n    --file-id <reference image FileId> \\\n    --prompt \"Fill the surrounding sky and sea\" \\\n    --output-image-count 2 \\\n    --output-storage-mode Permanent \\\n    --ext-info '{\"AdditionalParameters\":\"{\\\"up_expansion_ratio\\\":0.2,\\\"down_expansion_ratio\\\":0.2,\\\"left_expansion_ratio\\\":0.3,\\\"right_expansion_ratio\\\":0.3}\"}' \\\n    --sub-app-id 1308104797\n```\n\n> ✅ **Verified via `--dry-run`**: no script code changes were needed; the parameter system already correctly constructs this request, and the generated JSON exactly matches the official documentation. The previous doc version was missing the two required fields `--model-version scene` and `--scene-type image_expand`, now corrected.\n\n### 10.4 Multi-Subject Generation (3.0-Omni / O1, up to 10 reference images)\n\nKling 3.0-Omni and O1 support **up to 10 reference images** for multi-subject generation, passed via `--file-infos` JSON array:\n\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model Kling --model-version 3.0-Omni \\\n    --prompt \"Combine these characters into a family portrait, photorealistic style\" \\\n    --file-infos '[\n        {\"Type\":\"Url\",\"Url\":\"https://e.com/p1.jpg\"},\n        {\"Type\":\"Url\",\"Url\":\"https://e.com/p2.jpg\"},\n        {\"Type\":\"Url\",\"Url\":\"https://e.com/p3.jpg\"}\n    ]' \\\n    --output-resolution 4K --output-aspect-ratio auto \\\n    --sub-app-id 1308104797\n```\n\n> Interface tested: 3.0-Omni and O1 both support 4K + `auto` aspect ratio + up to 10 reference images.\n\n### 10.5 Kling 4K Output (3.0-Omni / O1 only)\n\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model Kling --model-version 3.0-Omni \\\n    --prompt \"Sunrise over snow mountains, ultra-realistic photography\" \\\n    --output-resolution 4K \\\n    --output-aspect-ratio 16:9 \\\n    --sub-app-id 1308104797\n```\n\n| Version | 1K | 2K | 4K |\n|---|---|---|---|\n| 2.1 / 3.0 | ✅ | ✅ | ❌ |\n| **3.0-Omni / O1** | ✅ | ✅ | **✅** |\n\n### 10.6 Kling O1 (flagship version, similar to 3.0-Omni)\n\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model Kling --model-version O1 \\\n    --prompt \"Cinematic portrait\" \\\n    --output-resolution 4K \\\n    --output-aspect-ratio auto \\\n    --sub-app-id 1308104797\n```\n\n> Interface tested: O1 and 3.0-Omni share identical resolution / aspect ratio / reference image support; consider them complementary flagship versions.\n\n---\n\n## 11 Vidu Advanced Features\n\n> 📌 **Capability**: Vidu q2 image-side supports **up to 7 reference images for fusion**, **4K high resolution**, and **multiple aspect ratios**. Vidu fixed-subject capability (CreateAigcSubject) is a separate API and is out of scope here.\n\n### 11.1 Plain Text-to-Image (high resolution)\n\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model Vidu --model-version q2 \\\n    --prompt \"Ink-wash mountain and water\" \\\n    --output-resolution 4K --output-aspect-ratio 16:9 \\\n    --sub-app-id 1308104797\n```\n\n### 11.2 Single reference (style)\n\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model Vidu --model-version q2 \\\n    --file-id 5145403721231891303 \\\n    --prompt \"Keep the composition but change to autumn color tone\" \\\n    --sub-app-id 1308104797\n```\n\n### 11.3 Multi-image reference (up to 7, fuse styles)\n\n> 🚨 **Multiple reference images must use `--file-infos`** with a JSON array. `--file-id` / `--file-url` only supports a single image.\n\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model Vidu --model-version q2 \\\n    --file-infos '[\n        {\"Type\":\"Url\",\"Url\":\"https://e.com/a.jpg\"},\n        {\"Type\":\"Url\",\"Url\":\"https://e.com/b.jpg\"},\n        {\"Type\":\"Url\",\"Url\":\"https://e.com/c.jpg\"},\n        {\"Type\":\"Url\",\"Url\":\"https://e.com/d.jpg\"}\n    ]' \\\n    --prompt \"Fuse the styles of these images\" \\\n    --sub-app-id 1308104797\n```\n\n### 11.4 Vidu q2 Resolution and Aspect Ratio\n\n| Dimension | Values |\n|---|---|\n| Resolution | `1080p`, `2K`, `4K` |\n| Aspect Ratio | 16:9, 9:16, 1:1, 3:4, 4:3, 21:9, 2:3, 3:2 |\n| Reference Images | 0–7 |\n\n---\n\n## 12 SI (Seedream) Advanced Features\n\n### 12.1 SI 4.0 Multi-Image Output (sequential_image_generation)\n\nSI 4.0 supports generating multiple images per task, requiring two conditions:\n\n1. **Specify the number of images in the Prompt** (e.g., \"output 3 images\")\n2. **Pass `sequential_image_generation: auto` via ExtInfo**\n\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model SI --model-version 4.0 \\\n    --prompt \"Chinese-style illustration, cranes and pines, output 3 images\" \\\n    --ext-info '{\"AdditionalParameters\": \"{\\\"sequential_image_generation\\\":\\\"auto\\\"}\"}' \\\n    --output-storage-mode Temporary \\\n    --sub-app-id 1308104797\n```\n\n> ⚠️ **SI 4.0 only**: SI 4.5 / SI 5.0-lite do not support this sequential multi-image output.\n\n### 12.2 SI Version Resolution Comparison\n\n| Version | Resolution Tiers | Default | Reference Images |\n|---|---|---|---|\n| SI 4.0 | 1K, 2K, 4K | 1K | 0–14 |\n| SI 4.5 | 2K, 4K | 2K | 0–14 |\n| SI 5.0-lite | 2K, 3K, 4K | 2K | 0–14 |\n| SI 5.0-pro | 1K, 2K, 4K | 1K | 0–14 |\n\n> SI series specifies aspect ratio via Prompt (describe 16:9, 9:16, etc. in the text), not via the `--output-aspect-ratio` parameter.\n\n---\n\n## 13 Custom Resolution (ExtInfo width/height/size)\n\nSome models do not support the standard `--output-resolution` / `--output-aspect-ratio` parameters and require custom resolution via `--ext-info`.\n\n### 13.1 Hunyuan 3.0 Custom size\n\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model Hunyuan --model-version 3.0 \\\n    --prompt \"Chinese ink landscape, vertical composition\" \\\n    --ext-info '{\"AdditionalParameters\": \"{\\\"size\\\":\\\"728x1024\\\"}\"}' \\\n    --output-storage-mode Temporary \\\n    --sub-app-id 1308104797\n```\n\n**Constraints**: width and height in [512, 2048], product ≤ 1024×1024 pixels, default 1024×1024.\n\n### 13.2 Qwen 0925 Custom width/height\n\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model Qwen --model-version 0925 \\\n    --prompt \"Futuristic sci-fi city night view\" \\\n    --ext-info '{\"AdditionalParameters\": \"{\\\"width\\\":1024, \\\"height\\\":1024}\"}' \\\n    --output-storage-mode Temporary \\\n    --sub-app-id 1308104797\n```\n\n**Constraints**: output pixels in [512×512, 2048×2048], default 1024×1024.\n\n### 13.3 Jimeng 4.0 Custom width/height\n\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model Jimeng --model-version 4.0 \\\n    --prompt \"Chinese-style illustration\" \\\n    --ext-info '{\"AdditionalParameters\": \"{\\\"width\\\":1920, \\\"height\\\":1080}\"}' \\\n    --output-storage-mode Temporary \\\n    --sub-app-id 1308104797\n```\n\n**Constraints**: resolution range [1024×1024, 4096×4096].\n\n### 13.4 ExtInfo Custom Resolution vs OutputConfig\n\n| Model | Standard `--output-resolution` / `--output-aspect-ratio` | ExtInfo Custom Resolution |\n|---|---|---|\n| **Hunyuan 3.0** | ❌ Not supported | ✅ `size` |\n| **Qwen 0925** | ❌ Not supported | ✅ `width`/`height` |\n| **Jimeng 4.0** | ❌ Not supported | ✅ `width`/`height` |\n| **OG (GPT-Image2)** | ✅ Supported + custom | ✅ Arbitrary size (multiple of 16) |\n| Other models (Kling/GG/SI/Vidu/MJ) | ✅ Supported | — |\n\n---\n\n## 14 Key Constraints Discovered via Interface Testing\n\n> All constraints verified via real interface submissions (2026-06-30)\n\n### 14.1 ModelName Aliases\n\n| ModelName | Interface Result | Notes |\n|---|---|---|\n| `GG` | ✅ Accepted | **Current skill usage** (product-side standard name) |\n| `GEM` | ✅ Accepted | Historical alias, also accepted by the interface |\n| `MJ` | ✅ Accepted | Current skill usage |\n| `Midjourney` | ❌ **Rejected**: `ModelName Midjourney is invalid` | Not usable |\n\n### 14.2 GG 3.1 Extreme Aspect Ratios (verified)\n\nGG 3.1 supports the following \"extreme aspect ratios\" (most other models don't):\n\n| AspectRatio | Use Case |\n|---|---|\n| `1:4`, `4:1` | Ultra-narrow / ultra-wide panoramic images |\n| `1:8`, `8:1` | Long banners, e-commerce / web banners |\n\n```bash\npython3 scripts/vod_aigc_image.py create \\\n    --model GG --model-version 3.1 \\\n    --prompt \"E-commerce web banner, long horizontal layout\" \\\n    --output-aspect-ratio 8:1 \\\n    --output-resolution 4K \\\n    --sub-app-id 1308104797\n```\n\n---\n\nFile v1.1.3:references/vod_aigc_token.md\n\n# vod_aigc_token — Detailed Parameters and Examples\n> This file is generated by splitting references, corresponding script: `scripts/vod_aigc_token.py`\n\n## Parameter Reference\n\n\n### Common Parameters\n\n| Parameter | Type | Description |\n|------|------|------|\n| `--region` | string | Region, default `ap-guangzhou` |\n| `--sub-app-id` | int | VOD sub-application ID (required for accounts created after 2023-12-25; can also be set via environment variable `TENCENTCLOUD_VOD_SUB_APP_ID`) |\n| `--json` | flag | Output in JSON format |\n| `--dry-run` | flag | Preview parameters without calling the API |\n\n### create Parameters (Create Token)\n\n| Parameter | Type | Description |\n|------|------|------|\n| `--sub-app-id` | int | VOD sub-application ID (optional; required for accounts created after 2023-12-25; can also be set via environment variable `TENCENTCLOUD_VOD_SUB_APP_ID`) |\n\n> 💡 If a Token already exists in the environment variable `TENCENTCLOUD_VOD_AIGC_TOKEN`, the script will scan all configuration files and list which ones contain that configuration, prompting whether to continue. Enter `y` or `yes` to proceed, or press Enter to cancel.\n\n> 💡 After successful creation, the Token will be **automatically written** to the environment variable configuration file (if the variable already exists, all files containing it will be updated; otherwise it will be written to `~/.env`), and `TENCENTCLOUD_VOD_AIGC_TOKEN` will be set.\n\n> ⚠️ **Token Notes:**\n> 1. Tokens have **no expiration time** — keep them secure\n> 2. After creation, please wait approximately **1 minute** for the token to sync to the gateway before using it\n> 3. Each user can have a maximum of **50** Tokens\n\n### list Parameters (Query Token List)\n\n| Parameter | Type | Description |\n|------|------|------|\n| `--sub-app-id` | int | VOD sub-application ID (optional; required for accounts created after 2023-12-25; can also be set via environment variable `TENCENTCLOUD_VOD_SUB_APP_ID`) |\n\n> 💡 Output field description: Each record contains `TokenId`, `Token` (token string), and `CreateTime`. When deleting a Token, the `--api-token` parameter should be the value of the `Token` field (not the `TokenId`).\n\n### delete Parameters (Delete Token)\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--api-token` | string | ✅ | The API Token string to delete (obtained via `list`; use the `Token` field value, not the `TokenId`) |\n| `--sub-app-id` | int | - | VOD sub-application ID (optional; required for accounts created after 2023-12-25; can also be set via environment variable `TENCENTCLOUD_VOD_SUB_APP_ID`) |\n\n> 💡 After deletion, the Token will become invalid at the gateway after approximately **1 minute**; it remains usable during that period.\n\n### usage Parameters (Query AIGC Usage Statistics)\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--sub-app-id` | int | - | VOD sub-application ID (optional; required for accounts created after 2023-12-25; can also be set via environment variable `TENCENTCLOUD_VOD_SUB_APP_ID`) |\n| `--type` / `-t` | string | ✅ | AIGC type: `Video` (video), `Image` (image), `Text` (text) |\n| `--start` | string | - | Start date, format `YYYY-MM-DD` or full ISO format (e.g. `2026-03-01T00:00:00+08:00`); **defaults to 30 days ago** |\n| `--end` | string | - | End date, format `YYYY-MM-DD` or full ISO format (e.g. `2026-03-31T23:59:59+08:00`); **defaults to today** |\n\n> 💡 When `--start`/`--end` are provided in `YYYY-MM-DD` format, the script will automatically append `T00:00:00+08:00` and `T23:59:59+08:00` respectively.\n\n### API Mapping\n\n| Feature | API | Documentation |\n|------|---------|---------|\n| Create Token | `CreateAigcApiToken` | https://cloud.tencent.com/document/api/266/128386 |\n| Delete Token | `DeleteAigcApiToken` | https://cloud.tencent.com/document/api/266/128387 |\n| Query Token | `DescribeAigcApiTokens` | https://cloud.tencent.com/document/api/266/128388 |\n| Query AIGC Usage | `DescribeAigcUsageData` | https://cloud.tencent.com/document/api/266/128389 |\n| Call LLM Chat API | POST `https://text-aigc.vod-qcloud.com/v1/chat/completions` | — |\n\n---\n\n\n---\n\n\n## Usage Examples\n\n\n### Token Lifecycle Management\n\n#### Create Token (First-Time Use)\n```bash\npython3 scripts/vod_aigc_token.py create\n# If a Token already exists, you will be prompted whether to overwrite it. Enter y or yes to continue, or press Enter to cancel.\n```\n\n#### Query Token List\n```bash\npython3 scripts/vod_aigc_token.py list\n```\n\n#### Delete a Specific Token\n```bash\npython3 scripts/vod_aigc_token.py delete --api-token tok_abc123\n```\n\n#### Specify Sub-Application ID (Required for accounts created after December 2023)\n```bash\npython3 scripts/vod_aigc_token.py create --sub-app-id 1500000001\npython3 scripts/vod_aigc_token.py list   --sub-app-id 1500000001\n```\n\n---\n\n### AIGC Usage Statistics\n\n#### Query Text Generation Usage (Current Month)\n```bash\npython3 scripts/vod_aigc_token.py usage \\\n    --type Text \\\n    --start 2026-03-01 \\\n    --end 2026-03-31\n```\n\n#### Query Video Generation Usage\n```bash\npython3 scripts/vod_aigc_token.py usage \\\n    --type Video \\\n    --start 2026-03-01 \\\n    --end 2026-03-31\n```\n\n#### Query Image Generation Usage (JSON Format)\n```bash\npython3 scripts/vod_aigc_token.py usage \\\n    --type Image \\\n    --start 2026-03-01 \\\n    --end 2026-03-31 \\\n    --json\n```\n\n#### Preview Parameters with dry-run\n```bash\npython3 scripts/vod_aigc_token.py --dry-run usage \\\n    --type Text \\\n    --start 2026-03-01 \\\n    --end 2026-03-31\n```\n\n---\n\nFile v1.1.3:references/vod_aigc_video.md\n\n# vod_aigc_video — Detailed Parameters and Examples\n> This file is generated by splitting the references, corresponding script: `scripts/vod_aigc_video.py`\n\n## Parameter Reference\n### Basic Parameters\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--model` | enum | ✅ | Model name (GV/Hailuo/Kling/Jimeng/Vidu/Hunyuan/Mingmou/OS/Seedance/PixVerse/H2) |\n| `--model-version` | string | - | Model version (uses default version if not specified) |\n| `--prompt` | string | ❌* | Prompt for video generation (required when no reference file is provided) |\n| `--negative-prompt` | string | ❌ | Prompt to prevent the model from generating certain content (negative prompt) |\n| `--enhance-prompt` | enum | ❌ | Whether to automatically optimize the prompt (Enabled: on, Disabled: off) |\n\n### Reference File Parameters\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--file-id` | string | ❌ | Media file ID of the reference file (single value; use `--file-infos` for multiple reference images) |\n| `--file-url` | string | ❌ | URL of the reference file (single value; use `--file-infos` for multiple reference images) |\n| `--file-infos` | JSON | ❌ | JSON array of multiple reference images, format: `[{\"Type\":\"Url\",\"Url\":\"...\",\"Category\":\"Image\",\"Usage\":\"Reference\",\"Text\":\"pic1\",\"ReferenceType\":\"subject\",\"ObjectId\":\"...\"}]`; supports all SDK fields: `Type`/`FileId`/`Url`/`Base64`/`Category`/`Usage`/`Text`/`ReferenceType`/`ObjectId`/`VoiceId`/`KeepOriginalSound` |\n| `--file-category` | enum | ❌ | Category of the single reference file: `Image` / `Video` / `Audio`; used by Kling motion_control / avatar_i2v scenes to distinguish image vs video; `Audio` is for Hailuo H3 multimodal reference generation |\n| `--file-usage` | enum | ❌ | Usage of the single reference file: `FirstFrame` / `LastFrame` / `Reference`; used by PixVerse, Vidu, Kling multi-mode disambiguation; Hailuo H3 reference video/audio uses `Reference` |\n| `--file-text` | string | ❌ | Name/description of the single reference file (PixVerse multi-image subject reference only; e.g. Text=`pic1` so that Prompt can use `@pic1 walking`) |\n| `--reference-type` | enum | ❌ | Reference type of the single file: `subject` / `background` / `mask`; PixVerse video edit uses subject/background; GV/Kling also applicable |\n\n### First/Last Frame Generation Parameters\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--last-frame-file-id` | string | ❌ | Media file ID of the last frame (for first/last frame generation) |\n| `--last-frame-url` | string | ❌ | URL of the last frame (for first/last frame generation) |\n\n### Output Configuration Parameters\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--output-storage-mode` | enum | ❌ | Storage mode (Permanent: permanent storage, Temporary: temporary storage, default: Temporary) |\n| `--output-media-name` | string | ❌ | Output filename (max 64 characters) |\n| `--output-class-id` | int | ❌ | Category ID (default 0, meaning \"Other\") |\n| `--output-expire-time` | string | ❌ | Expiration time of the output file (ISO 8601 format, e.g. 2026-12-31T23:59:59+08:00) |\n| `--output-duration` | int | ❌ | Duration of the generated video in seconds (supported durations vary by model) |\n| `--output-resolution` | enum | ❌ | Resolution of the generated video (supported resolutions vary by model) |\n| `--output-aspect-ratio` | enum | ❌ | Aspect ratio of the generated video (supported aspect ratios vary by model) |\n| `--output-audio-generation` | enum | ❌ | Whether to generate audio (Enabled: on, Disabled: off, default: Disabled; supported by GV/OS/Vidu) |\n| `--output-person-generation` | enum | ❌ | Whether to allow person/face generation (AllowAdult: allow, Disallowed: prohibit) |\n| `--output-enhance-switch` | enum | ❌ | Whether to enable video enhancement (Enabled/Disabled; auto-enabled when resolution exceeds model native output) |\n| `--output-off-peak` | enum | ❌ | Whether to enable off-peak scheduling (Enabled: on, Disabled: off) |\n| `--output-frame-interpolate` | enum | ❌ | Whether to enable Vidu smart frame interpolation (Enabled: on, Disabled: off) |\n| `--output-logo-add` | enum | ❌ | Whether to enable logo watermark (Enabled: on, Disabled: off; currently only Vidu supported) |\n| `--input-compliance-check` | enum | ❌ | Whether to enable compliance check on input content (Enabled: on, Disabled: off) |\n| `--output-compliance-check` | enum | ❌ | Whether to enable compliance check on output content (Enabled: on, Disabled: off) |\n\n### Additional Parameters\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--procedure` | string | ❌ | Task flow name; auto-executes specified task flow after video generation |\n| `--seed` | int | ❌ | Model random seed (when specified, enables reproducible generation) |\n| `--input-region` | enum | ❌ | Region of the input file (Mainland: mainland China, Oversea: overseas, default: Mainland) |\n| `--scene-type` | enum | ❌ | Scene type (motion_control: motion control/Kling, avatar_i2v: digital avatar/Kling, lip_sync: lip sync/Kling, template_effect: template effect/Vidu, subject_reference: fixed subject reference/Vidu) |\n| `--subject-infos` | JSON | ❌ | Fixed subject JSON array, format: `[{\"Id\":\"...\",\"Name\":\"...\"}]` (⚠️ Not yet supported by SDK — parameter can be passed but currently has no effect) |\n| `--element-ids` | string | ❌ | Advanced custom subject IDs (comma-separated, e.g. `865750283577090106`); mutually exclusive with `--ext-info`, takes higher priority |\n| `--elements-file` | path | ❌ | Read subject ID list from a JSON file (e.g. `mem/elements.json`); mutually exclusive with `--element-ids` |\n| `--session-id` | string | ❌ | Deduplication identifier (max 50 characters; reuse within 3 days returns an error) |\n| `--session-context` | string | ❌ | Source context for passing through user request information (max 1000 characters) |\n| `--tasks-priority` | int | ❌ | Task priority (higher value = higher priority, range: -10 to 10) |\n| `--ext-info` | string | ❌ | Reserved field for special use cases (mutually exclusive with `--element-ids`, lower priority) |\n\n### Common Parameters (`create` subcommand only)\n\n| Parameter | Type | Description |\n|------|------|------|\n| `--sub-app-id` | int | Sub-application ID (required for accounts that activated VOD after December 25, 2023; can also be set via the TENCENTCLOUD_VOD_SUB_APP_ID environment variable) |\n| `--region` | string | Tencent Cloud region (default: ap-guangzhou) |\n| `--json` | flag | Output in JSON format |\n| `--dry-run` | flag | Print request parameters preview only, without sending the request |\n| `--no-wait` | flag | Submit the task only, without waiting for the result (default: auto-wait) |\n| `--max-wait` | int | Maximum wait time in seconds (default: 1800) |\n\n> ⚠️ **Note**: `vod_aigc_video.py` **does not have a `query` subcommand**. To check the status of a video generation task, use `vod_describe_task.py --task-id <task_id>`.\n\n### Model Parameter Comparison\n\n#### Hailuo\n\n| Parameter | Supported Values |\n|------|--------|\n| Version | 02, 2.3, 2.3-fast, **H3** |\n| Duration | **02: 6/8/10/12/15/20s (interface tested: 02 accepts up to 20s+); 2.3 / 2.3-fast: 6s, 10s (default: 6s); H3: `--output-duration 15` verified (output 15.084s)** |\n| Resolution | 768P, 1080P (default: 768P); **H3 verified to output 2560x1440** |\n| Aspect Ratio | Not supported on 02 / 2.3 / 2.3-fast; **supported on H3 (verified: `--output-aspect-ratio 16:9` takes effect, output 2560x1440)** |\n| First/Last Frame Generation | Not supported on 02 / 2.3 / 2.3-fast; **supported on H3 (first ≤ 1, last ≤ 1)** |\n| Audio Generation | Not supported on 02 / 2.3 / 2.3-fast; **H3 generates audio natively (verified: output carries AAC 32 kHz stereo, no need to set `--output-audio-generation`)** |\n| Notes | 02 has extended duration vs the 2.3 series (tested 20s+); **H3 (released 2026-07-31) is the natively multimodal version — see below** |\n\n##### Hailuo H3 (MiniMax H3, released 2026-07-31)\n\nNative multimodal understanding and generation: accepts text, image, audio and video input/output for end-to-end content creation; supports fine-grained edits such as replacement and reference; covers commercial scenarios including film, advertising, gaming, branding and e-commerce.\n\n**Two mutually exclusive input modes** (cannot be combined):\n\n| Mode | Meaning | Usage values |\n|------|---------|--------------|\n| i2va | Image-to-video (first / last frame) | `FirstFrame` / `LastFrame` |\n| r2va | Multimodal reference generation (reference video / audio) | `Reference` |\n\n**Image input** (first frame / last frame / reference image)\n\n| Item | Limit |\n|------|-------|\n| Formats | JPG, JPEG, PNG, WEBP, HEIC, HEIF |\n| Size per file | ≤ 30 MB |\n| Width/height range | [256, 5760] px |\n| Aspect ratio (w/h) | [0.4, 2.5] |\n| Count | first ≤ 1, last ≤ 1, reference ≤ 9 |\n\n**Video input** (r2va reference generation)\n\n| Item | Limit |\n|------|-------|\n| Container/format | MP4 (.mp4), MOV (.mov) |\n| Codecs | Video H.264/AVC, H.265/HEVC; audio AAC, MP3 |\n| Size per file | ≤ 50 MB |\n| Count | ≤ 3 |\n| Clip duration | [2, 15] s; total ≤ 15 s |\n| Width/height range | [256, 5760] px |\n| Aspect ratio (w/h) | [0.4, 2.5] |\n| Frame rate | [23.976, 60] |\n\n**Audio input** (r2va reference generation)\n\n| Item | Limit |\n|------|-------|\n| Formats | WAV, MP3 |\n| Size per file | ≤ 15 MB |\n| Count | ≤ 3 |\n| Clip duration | [2, 15] s; total ≤ 15 s |\n\n> The script validates count limits and the i2va/r2va exclusivity before submission,\n> failing fast with an explanatory message. Formats, file sizes, dimensions,\n> durations and frame rates are validated server-side.\n\n**Verified output specs** (2026-08-02, text-only with `--output-aspect-ratio 16:9 --output-duration 15`)\n\n| Item | Measured |\n|------|----------|\n| Duration | 15.084 s |\n| Resolution | 2560x1440 (16:9) |\n| Video codec | H.264, 24 fps, 362 frames |\n| Audio | AAC 32 kHz stereo (**generated natively, not explicitly enabled**) |\n| Bitrate / size | 5.1 Mbps / 9.6 MB |\n| Task time | about 8 min 21 s |\n\n> ⚠️ The \"Aspect Ratio / Audio Generation not supported\" rows in the Hailuo family table above\n> describe 02 / 2.3 / 2.3-fast and **do not apply to H3**. H3 accepts `--output-aspect-ratio`\n> and emits an audio track natively.\n\n**Examples**\n\n```bash\n# Text-to-video only\npython3 scripts/vod_aigc_video.py create \\\n    --model Hailuo --model-version H3 \\\n    --prompt \"A girl holding a kite runs toward the camera on a playground, orbiting camera move from front to behind. Cinematic quality\"\n\n# i2va: first frame + last frame\npython3 scripts/vod_aigc_video.py create \\\n    --model Hailuo --model-version H3 \\\n    --prompt \"Slow dolly-in\" \\\n    --file-url \"https://example.com/first.jpg\" \\\n    --file-category Image --file-usage FirstFrame \\\n    --last-frame-url \"https://example.com/last.jpg\"\n\n# r2va: multiple reference videos (exclusive with first/last frame)\npython3 scripts/vod_aigc_video.py create \\\n    --model Hailuo --model-version H3 \\\n    --prompt \"Continue the camera style of the reference videos\" \\\n    --file-infos '[{\"Type\":\"Url\",\"Url\":\"https://example.com/a.mp4\",\"Category\":\"Video\",\"Usage\":\"Reference\"},{\"Type\":\"Url\",\"Url\":\"https://example.com/b.mp4\",\"Category\":\"Video\",\"Usage\":\"Reference\"}]'\n\n# r2va: reference audio\npython3 scripts/vod_aigc_video.py create \\\n    --model Hailuo --model-version H3 \\\n    --prompt \"Match the visual rhythm to the reference audio\" \\\n    --file-url \"https://example.com/ref.mp3\" \\\n    --file-category Audio --file-usage Reference\n```\n\n#### Kling\n\n| Parameter | Supported Values |\n|------|--------|\n| Version | 1.6, 2.0, 2.1, 2.5, O1, 2.6, 3.0, 3.0-Omni, 3.0-turbo |\n| Duration | 5s, 10s (default: 5s) |\n| Resolution | **720P, 1080P, 4K (default: 720P; interface tested: 2.1/3.0/2.6/3.0-Omni/3.0-turbo/O1 all accept 4K)** |\n| Aspect Ratio | 16:9, 9:16, 1:1 (default: 16:9) |\n| First/Last Frame Generation | Supported (version 2.1 requires 1080P) |\n| Scene Type | motion_control, avatar_i2v (digital avatar), lip_sync |\n| Audio Generation | 3.0-Omni supports audio/silent modes (`OutputConfig.AudioGeneration: Enabled/Disabled`) |\n\n#### Jimeng\n\n| Parameter | Supported Values |\n|------|--------|\n| Version | 3.0pro |\n| Resolution | **Custom via ExtInfo `width`/`height`**; example `--ext-info '{\"AdditionalParameters\": \"{\\\"width\\\":1920, \\\"height\\\":1080}\"}'` |\n| First/Last Frame Generation | Not supported |\n| Audio Generation | Not supported |\n| Notes | Jimeng video does not support OutputConfig.Resolution / AspectRatio; pass resolution via ExtInfo |\n\n#### Vidu\n\n| Parameter | Supported Values |\n|------|--------|\n| Version | q2, q2-turbo, q2-pro, q3, q3-pro, q3-turbo, q3-mix, q3-drama |\n| Duration | 1–10 seconds (customizable) |\n| Resolution | 720P, 1080P (default: 720P) |\n| Aspect Ratio | 16:9, 9:16, 1:1, 3:4, 4:3 (default: 16:9) |\n| First/Last Frame Generation | Supported (q2-pro, q2-turbo; q3-pro supports text-to-video and image-to-video only) |\n| Multi-image Reference | q2 supports 1–7 images via FileInfos.ObjectId as the subject ID |\n| Scene Type | template_effect (template effect), subject_reference (fixed subject reference) |\n| Notes | **🚨 q3-mix / q3-drama require reference image with `Usage=Reference`** (interface tested: text-to-video alone returns `only supports reference generation`); q3-mix has stronger visual quality, smart cuts, and balanced motion; q3 supports smart cuts with consistent multi-camera; q3-mix does not support subject library yet |\n\n#### Hunyuan\n\n| Parameter | Supported Values |\n|------|--------|\n| Version | 1.5, 3d_2.0 |\n| Scene Type | `3d_scene` (3d_2.0 only: Hunyuan World Model generates 3D scene video) |\n| Duration | 3d_2.0 + 3d_scene recommends 16s (`OutputConfig.Duration`) |\n| Resolution | 3d_2.0 + 3d_scene recommends 1080P |\n| Audio Generation | 3d_2.0 + 3d_scene supports `OutputConfig.AudioGeneration: Enabled` |\n| Storage Mode | **3d_2.0 only supports `Temporary`** (interface tested: passing Permanent returns `StorageMode Permanent is not supported for model Hunyuan 3d_2.0`) |\n| First/Last Frame Generation | Not supported |\n| Output Artifacts | 3d_2.0 + 3d_scene outputs multiple files: spz (Gaussian Splatting) / ply (point cloud) etc., importable to Unity/Unreal Engine |\n| Notes | **3d_2.0** + `--scene-type 3d_scene` generates walkable 3D scene videos (Hunyuan World Model); **1.5** is the legacy general video version |\n\n#### Mingmou\n\n| Parameter | Supported Values |\n|------|--------|\n| Version | 1.0 |\n| Resolution | **Custom via ExtInfo `width`/`height`**; example `--ext-info '{\"AdditionalParameters\": \"{\\\"width\\\":1920, \\\"height\\\":1080}\"}'` |\n| First/Last Frame Generation | Not supported |\n| Audio Generation | Not supported |\n\n#### GV\n\n| Parameter | Supported Values |\n|------|--------|\n| Version | 3.1, 3.1-fast, 3.1-lite, **omni** |\n| Duration | 3.1 series: fixed 8 seconds; **omni: 3–10 seconds** |\n| Resolution | 720P, 1080P (default: 720P); **omni additionally supports 2K/4K (both super-resolution output)** |\n| Aspect Ratio | 16:9, 9:16 (default: 16:9) |\n| Multi-image Reference | 3.1 series: up to 3 images; **omni: up to 7 images without a video reference, up to 5 images when paired with a video reference, total quota 7** |\n| First/Last Frame Generation | 3.1 series: supported; **omni: not supported (use `<FIRST_FRAME>`/`<IMAGE_REF_N>` markup syntax instead)** |\n| Audio Generation | Native audio-visual sync; supports both audio and silent modes; **omni generates natively via AI, external audio upload not supported** |\n| Notes | Faces are not blocked; when using multi-image input (`--file-infos`), LastFrameFileId/LastFrameUrl cannot be used simultaneously (the script now enforces this validation and rejects the request if violated) |\n\n##### GV omni (Google Veo Omni, released 2026-07-09)\n\nOfficial docs: https://ai.google.dev/gemini-api/docs/omni?hl=zh-cn\n\n**Key features**:\n- Generation modes: text-to-video, image-reference-to-video, video-reference-to-video\n- Supports **stateful video editing** (iterative editing): continue modifying the previous generation result without re-uploading the video\n- Reference assets: up to 7 images without a video reference; up to 5 images when paired with a video reference; accepts video references up to 3 seconds\n- Prompt limit 20,000 characters; supports `<FIRST_FRAME>` (first-frame markup) and `<IMAGE_REF_N>` (reference-image markup, 0-indexed) syntax\n- Billing: in addition to per-second output billing, input images/audio/text are converted to input tokens, and chain-of-thought tokens are billed as output tokens\n\n**Stateful video editing usage**: pass `PreviousTaskId` (the TaskId of the previous video generation task, only tasks within the last 24 hours are supported) via `--ext-info` to continue editing from the last result:\n\n```bash\n# First generation\npython3 scripts/vod_aigc_video.py create \\\n    --model GV --model-version omni \\\n    --file-url \"https://example.com/f0.jpg\" \\\n    --prompt \"Two man are studying\" \\\n    --enhance-prompt Enabled \\\n    --sub-app-id 1308104797\n\n# Continue editing based on the previous task result (assume previous TaskId is xxx-AigcVideoTask-yyy)\npython3 scripts/vod_aigc_video.py create \\\n    --model GV --model-version omni \\\n    --prompt \"Make the background invisible\" \\\n    --enhance-prompt Enabled \\\n    --ext-info '{\"AdditionalParameters\":\"{\\\"PreviousTaskId\\\":\\\"xxx-AigcVideoTask-yyy\\\"}\"}' \\\n    --sub-app-id 1308104797\n```\n\n> ⚠️ For the stateful editing flow, the second call does **not** need `--file-url`/`--file-id` again — context is linked purely via `ExtInfo.PreviousTaskId`.\n\n#### OS\n\n| Parameter | Supported Values |\n|------|--------|\n| Version | 2.0 |\n| Duration | 4s, 8s, 12s (default: 8s) |\n| Resolution | Fixed 720P |\n| Aspect Ratio | 16:9, 9:16 (default: 16:9) |\n| First/Last Frame Generation | Not supported |\n\n#### Seedance (ByteDance)\n\n| Parameter | Supported Values |\n|------|--------|\n| Version | 1.0-pro-fast, 1.5-pro (default 1.5-pro) |\n| Audio Generation | 1.5-pro supports both audio and silent modes (`OutputConfig.AudioGeneration: Enabled/Disabled`) |\n| Resolution | 1.5-pro does NOT support 1080P |\n| Notes | **Interface name is `Seedance`** (earlier docs incorrectly used `SV`; interface tested: passing `SV` returns `ModelName SV is invalid`); Seedance is the ByteDance Doubao video series |\n\n#### PixVerse\n\n| Parameter | Supported Values |\n|------|--------|\n| Version | v5.6, v6, c1 |\n| Audio Generation | v5.6 (silent); v6/c1 supports both audio and silent modes |\n| Duration | c1/v6: 1–15 seconds (1080P max 15s); v5.6: 5/8/10s (1080p does not support 10s) |\n| Resolution | **360P, 540P, 720P, 1080P, 4K** (interface tested: v5.6/v6/c1 all accept 4K) |\n| Multi-image Reference | c1/v6: up to 7 subjects; v5.6: up to 7 reference videos |\n| First/Last Frame Generation | Supported (use `--file-usage FirstFrame` + `--last-frame-url/--last-frame-file-id`) |\n| Multi-subject (@name reference) | c1/v6 multi-image mode requires `Category=Image` + `Usage=Reference`; use `--file-text` or file-infos `Text` to name images (e.g. `pic1`), then reference as `@pic1` in Prompt |\n| Video Edit | v5.6/v6/c1 support video edit: reference video uses `--file-category Video` + `--reference-type subject` (replace subject) or `background` (replace background) |\n| Notes | c1 is the latest character-consistency version; v6 is the general flagship; v5.6 is the previous stable version; SceneType (motion_control/avatar_i2v/lip_sync) is Kling-only and NOT supported by PixVerse |\n\n#### H2 (Kuaile Ma / MiniMax H2 series)\n\n| Parameter | Supported Values |\n|------|--------|\n| Version | 1.0 (released 2026-04-30), **1.1** (added 2026-06-24, default) |\n| Generation Mode | Text-to-video, first-frame-to-video (1.0: first-frame only), reference-to-video (1-9 images) |\n| Resolution | 720P, 1080P, 2K, 4K |\n| Aspect Ratio | 1.0: 16:9, 9:16, 1:1, 3:4, 4:3; **1.1 adds**: 4:5, 5:4, 9:21, 21:9 (default 16:9) |\n| Duration | 3-15 seconds (default 5 seconds) |\n| Audio Generation | Supported |\n| Notes | Image-to-video only supports first-frame (no first/last-frame or last-frame-only input); 1.1 adds 4 aspect ratio options compared to 1.0, all other parameters are identical |\n\n**Usage examples**:\n\n```bash\n# Text-to-video\npython3 scripts/vod_aigc_video.py create \\\n    --model H2 --model-version 1.1 \\\n    --prompt \"A happy little horse running on the grassland, cinematic quality\" \\\n    --output-resolution 1080P --output-aspect-ratio 16:9 --output-duration 5 \\\n    --sub-app-id 1308104797\n\n# First-frame image-to-video\npython3 scripts/vod_aigc_video.py create \\\n    --model H2 --model-version 1.1 \\\n    --file-url \"https://example.com/first.jpg\" \\\n    --prompt \"Make the horse in the image run\" \\\n    --sub-app-id 1308104797\n\n# Reference-to-video (1-9 reference images)\npython3 scripts/vod_aigc_video.py create \\\n    --model H2 --model-version 1.1 \\\n    --file-infos '[{\"Type\":\"Url\",\"Url\":\"https://example.com/ref1.jpg\",\"Category\":\"Image\",\"Usage\":\"Reference\"},{\"Type\":\"Url\",\"Url\":\"https://example.com/ref2.jpg\",\"Category\":\"Image\",\"Usage\":\"Reference\"}]' \\\n    --prompt \"Blend elements from the reference images\" \\\n    --sub-app-id 1308104797\n```\n\n### Task Status Reference\n\n| Status | Description |\n|------|------|\n| WAIT | Waiting |\n| RUN | Processing |\n| FINISH | Completed |\n| FAIL | Failed |\n\n### API Mapping\n\n| Feature | API | Documentation |\n|------|---------|---------|\n| Create AIGC video generation task | `CreateAigcVideoTask` | https://cloud.tencent.com/document/api/266/126239 |\n| Query task status | `DescribeTaskDetail` | https://cloud.tencent.com/document/api/266/33431 |\n\n### Error Code Reference\n\n| Error Type | Cause | Recommended Action |\n|---------|------|---------|\n| Unsupported model version | The specified Model Version does not exist | Check the supported model list and select a valid version |\n| Unsupported resolution | The specified Resolution is not supported by the model | Check the model feature table and select a supported resolution |\n| Unsupported aspect ratio | The specified Aspect Ratio is not supported by the model | Check the model feature table and select a supported aspect ratio |\n| Too many reference images | The number of reference images exceeds the model limit | Check the model feature table and stay within the allowed range |\n| Prompt required | No Prompt provided and no reference image supplied | Add the `--prompt` parameter or provide a reference image |\n| Sub AppId required | Sub-application ID not specified | Add the `--sub-app-id` parameter or set the environment variable |\n| First/last frame restriction | Kling 2.1 first/last frame generation requires 1080P | Set resolution to 1080P or switch to a different model version |\n| Duplicate session | SessionId reused within 3 days | Use a different SessionId or wait for it to expire |\n\n---\n\n\n---\n\n\n## Usage Examples\n### 1 Basic Text-to-Video\n\n#### GV Model — Text-to-Video\n```bash\npython3 scripts/vod_aigc_video.py create \\\n    --model GV \\\n    --prompt \"A puppy running across a sunny meadow\"\n```\n\n#### Hailuo Model (with specified resolution and duration)\n```bash\npython3 scripts/vod_aigc_video.py create \\\n    --model Hailuo \\\n    --model-version 2.3 \\\n    --prompt \"Time-lapse of a sunset over the ocean\" \\\n    --output-resolution 1080P \\\n    --output-duration 10\n```\n\n#### Kling Model (1080P HD, 5 seconds)\n```bash\npython3 scripts/vod_aigc_video.py create \\\n    --model Kling \\\n    --model-version 2.1 \\\n    --prompt \"Aerial footage of a city skyline at night\" \\\n    --output-resolution 1080P \\\n    --output-duration 5 \\\n    --output-aspect-ratio 16:9\n```\n\n#### Kling O1 Model + Custom Subject (element-ids)\n> ⚠️ **Note**: Pass only `Kling` to `--model`; the version `O1` is passed separately via `--model-version`. Do NOT write `--model \"Kling O1\"`.\n> ⚠️ **Note**: When using `--element-ids`, the `--prompt` must use the `<<<element_1>>>` placeholder in place of the subject name.\n```bash\npython3 scripts/vod_aigc_video.py create \\\n    --model Kling \\\n    --model-version O1 \\\n    --element-ids \"866084540648271963\" \\\n    --prompt \"<<<element_1>>> walking along the beach\" \\\n    --sub-app-id 1500046725\n```\n\n### 2 Image-to-Video\n\n#### Using FileId as the First Frame\n```bash\npython3 scripts/vod_aigc_video.py create \\\n    --model Kling \\\n    --model-version 2.1 \\\n    --file-id 3704211509819 \\\n    --prompt \"Animate the person in the image to walk slowly\"\n```\n\n#### Using URL as the First Frame\n```bash\npython3 scripts/vod_aigc_video.py create \\\n    --model GV \\\n    --file-url \"https://example.com/first_frame.jpg\" \\\n    --prompt \"Camera slowly pushes forward\"\n```\n\n### 3 First/Last Frame Video Generation\n\n```bash\npython3 scripts/vod_aigc_video.py create \\\n    --model GV \\\n    --file-url \"https://example.com/first.jpg\" \\\n    --last-frame-url \"https://example.com/last.jpg\" \\\n    --prompt \"smooth transition between the two scenes\"\n```\n\n### 4 Default Wait for Task Completion\n\n```bash\n# Video generation takes a while (default timeout is 1800 seconds)\npython3 scripts/vod_aigc_video.py create \\\n    --model GV \\\n    --prompt \"A cat sunbathing by the window\"\n\n# Submit task only, without waiting\npython3 scripts/vod_aigc_video.py create \\\n    --model GV \\\n    --prompt \"A cat sunbathing by the window\" \\\n    --no-wait\n```\n\n### 5 Permanent Storage\n\n```bash\npython3 scripts/vod_aigc_video.py create \\\n    --model Hailuo \\\n    --prompt \"Scenic landscape video\" \\\n    --output-storage-mode Permanent \\\n    --output-media-name \"My Generated Video\"\n```\n\n### 6 List Supported Models\n\n```bash\npython3 scripts/vod_aigc_video.py models\n```\n\n## 7 PixVerse Advanced Features\n\n### 7.1 Multi-Subject Reference (@name in Prompt)\n\nc1 / v6 support up to 7 reference images, each named via `Text` and referenced as `@name` in the Prompt.\n\n**JSON multi-image mode (recommended):**\n\n```bash\npython3 scripts/vod_aigc_video.py create \\\n    --model PixVerse --model-version c1 \\\n    --prompt \"@pic1 (a woman in ancient costume) is holding @pic2 (a closed fan) and slowly opening it\" \\\n    --file-infos '[\n        {\"Type\":\"Url\",\"Url\":\"https://e.com/woman.jpg\",\"Category\":\"Image\",\"Usage\":\"Reference\",\"Text\":\"pic1\"},\n        {\"Type\":\"Url\",\"Url\":\"https://e.com/fan.jpg\",\"Category\":\"Image\",\"Usage\":\"Reference\",\"Text\":\"pic2\"}\n    ]' \\\n    --output-storage-mode Temporary \\\n    --output-duration 8 --output-aspect-ratio 3:4 \\\n    --output-audio-generation Enabled \\\n    --sub-app-id 1308104797\n```\n\n> ⚠️ **Required**: every image must have `Category=Image` + `Usage=Reference` + `Text=<name>`; the `@name` reference in Prompt must be followed by a space (e.g. `@pic1 walking`); `Text` must be plain Chinese or English (no special characters).\n\n### 7.2 Video Edit (subject / background)\n\nv5.6 / v6 / c1 support taking a video as input and replacing subject or background, distinguished via `ReferenceType`.\n\n**Example 1: Replace subject (change the dress color to white):**\n\n```bash\npython3 scripts/vod_aigc_video.py create \\\n    --model PixVerse --model-version v5.6 \\\n    --prompt \"Change the color of the actress's dress to white\" \\\n    --file-url \"https://e.com/source.mp4\" \\\n    --file-category Video \\\n    --reference-type subject \\\n    --output-storage-mode Permanent \\\n    --output-media-name \"PixVerse Video Edit\" \\\n    --sub-app-id 1308104797\n```\n\n**Example 2: First/Last Frame:**\n\n```bash\npython3 scripts/vod_aigc_video.py create \\\n    --model PixVerse --model-version v6 \\\n    --prompt \"smooth transition from sunrise to sunset\" \\\n    --file-url \"https://e.com/first.jpg\" \\\n    --file-usage FirstFrame \\\n    --last-frame-url \"https://e.com/last.jpg\" \\\n    --output-duration 5 \\\n    --sub-app-id 1308104797\n```\n\n> ⚠️ **First-frame vs Reference**: Single-image PixVerse defaults to first-frame mode; for reference generation explicitly set `--file-usage Reference` (or `Usage=Reference` inside file-infos); multi-image mode defaults to reference generation (each image must carry `Usage`).\n\n### 7.3 PixVerse Version Comparison\n\n| Capability | c1 | v6 | v5.6 |\n|---|---|---|---|\n| Text-to-video | ✅ | ✅ | ✅ |\n| Image-to-video | ✅ | ✅ | ✅ |\n| Reference generation | ✅ | ✅ | (reference video only) |\n| Max reference images | 7 | 7 | 7 |\n| 1080P max duration | 15s | 15s | 8s |\n| Native audio-visual sync | ✅ | ✅ | ❌ |\n\n## 8 Hunyuan 3D Scene Video (Hunyuan World Model)\n\n`Hunyuan 3d_2.0` + `--scene-type 3d_scene` generates walkable 3D scene videos with native audio-visual sync.\n\n### 8.1 Text-to-3D Scene\n\n```bash\npython3 scripts/vod_aigc_video.py create \\\n    --model Hunyuan --model-version 3d_2.0 \\\n    --scene-type 3d_scene \\\n    --prompt \"Forbidden City Hall at noon, photorealistic historical reconstruction with PBR materials\" \\\n    --output-storage-mode Temporary \\\n    --output-duration 16 \\\n    --output-resolution 1080P \\\n    --output-audio-generation Enabled \\\n    --sub-app-id 1308104797\n```\n\n> ⚠️ **3d_2.0 only supports Temporary storage** — passing `Permanent` will be rejected by the interface. Temporary outputs are valid for 7 days.\n\n> 📦 **Output artifacts**: This scene outputs 6 files including 1 `.spz` (Gaussian Splatting model) and multiple `.ply` (point cloud) files, natively importable to Unity/Unreal Engine.\n\n### 8.2 Differences from Hunyuan 1.5 / 3d_panorama\n\n| Use case | Model + Version | Endpoint | SceneType |\n|---|---|---|---|\n| Legacy general video | `Hunyuan 1.5` | `vod_aigc_video.py` | (none) |\n| 360° panoramic image (image side) | `Hunyuan 3d_2.0` | `vod_aigc_image.py` | `3d_panorama` |\n| 3D scene video (video side) | `Hunyuan 3d_2.0` | `vod_aigc_video.py` | `3d_scene` |\n\n> The Hunyuan World Model 3D scene video output may include 3DGS (Gaussian Splatting) / Mesh (GLB/FBX/OBJ) / PLY (point cloud) / panoramic video, with native Unity/Unreal Engine support.\n\n## 9 Kling Advanced Scenes (Motion Control / Lip Sync / Avatar)\n\nKling distinguishes advanced scenes via `--scene-type`, with complex parameters passed through `--ext-info` as JSON.\n\n### 9.1 Motion Control (motion_control)\n\n**Version requirement**: `3.0` for the v3.0 motion control; `2.6` for the legacy version (also the standard entry for motion_control).\n\n```bash\npython3 scripts/vod_aigc_video.py create \\\n    --model Kling --model-version 2.6 \\\n    --scene-type motion_control \\\n    --prompt \"Generate a new video based on the reference video\" \\\n    --file-infos '[\n        {\"Type\":\"Url\",\"Url\":\"https://e.com/ref.mp4\",\"Category\":\"Video\"},\n        {\"Type\":\"Url\",\"Url\":\"https://e.com/face.webp\",\"Category\":\"Image\"}\n    ]' \\\n    --ext-info '{\"AdditionalParameters\":\"{\\\"keep_original_sound\\\":\\\"no\\\",\\\"character_orientation\\\":\\\"video\\\"}\"}' \\\n    --sub-app-id 1308104797\n```\n\n**ExtInfo key parameters:**\n- `keep_original_sound`: `yes` (keep original audio) / `no`\n- `character_orientation`: `image` (match orientation in image; reference video ≤ 10s) / `video` (match orientation in video; reference video ≤ 30s)\n\n### 9.2 Lip Sync (lip_sync, requires DescribeAigcFaceInfo first)\n\n**Two-step flow:**\n1. First call `DescribeAigcFaceInfo` to get `SessionId` and `FaceId` (note: this endpoint is **not yet implemented** in the skill; you must call SDK/API manually).\n2. Then call `CreateAigcVideoTask` with `SceneType=lip_sync` + `ExtInfo` carrying face info.\n\n```bash\npython3 scripts/vod_aigc_video.py create \\\n    --model Kling --model-version 2.6 \\\n    --scene-type lip_sync \\\n    --prompt \"lip sync\" \\\n    --ext-info '{\"AdditionalParameters\":\"{\\\"session_id\\\":\\\"845736590818832460\\\",\\\"face_choose\\\":[{\\\"face_id\\\":0,\\\"sound_file\\\":\\\"https://e.com/audio.mp3\\\",\\\"sound_start_time\\\":0,\\\"sound_end_time\\\":5000,\\\"sound_insert_time\\\":2000,\\\"sound_volume\\\":2,\\\"original_audio_volume\\\":0}]}\"}' \\\n    --sub-app-id 1308104797\n```\n\n> ⚠️ **lip_sync does NOT carry FileInfos** — all audio/video info goes through `ExtInfo` (session_id + face_choose); leave FileInfos empty.\n\n### 9.3 Avatar (avatar_i2v)\n\n```bash\npython3 scripts/vod_aigc_video.py create \\\n    --model Kling --model-version 2.6 \\\n    --scene-type avatar_i2v \\\n    --prompt \"dance\" \\\n    --file-url \"https://e.com/portrait.png\" \\\n    --file-category Image \\\n    --ext-info '{\"AdditionalParameters\":\"{\\\"sound_file\\\":\\\"https://e.com/audio.mp3\\\"}\"}' \\\n    --sub-app-id 1308104797\n```\n\n**ExtInfo key parameters:**\n- `sound_file`: Audio URL or Base64 (mp3/wav/m4a/aac, ≤5MB, 2–300 seconds)\n- `audio_id`: Audio ID (alternative to `sound_file`)\n- `sound_file` and `audio_id` cannot both be empty, nor both have values.\n\n## 10 AIGC Super-Resolution Output Strategy\n\nCombine `--output-resolution` + `--output-enhance-switch` to achieve \"model outputs low resolution then super-resolves to high resolution\" — a cost-optimization pattern.\n\n### 10.1 Super-Resolve to 1080P (cost optimization)\n\n```bash\n# Model outputs 720P natively, then super-resolves to 1080P (cheaper than direct 1080P)\npython3 scripts/vod_aigc_video.py create \\\n    --model GV --model-version 3.1 \\\n    --prompt \"Make the text fly\" \\\n    --file-url \"https://e.com/logo.webp\" \\\n    --output-resolution 1080P \\\n    --output-enhance-switch Enabled \\\n    --sub-app-id 1308104797\n```\n\n### 10.2 Output 2K / 4K (super-resolve enabled by default)\n\n```bash\n# When choosing 2K/4K, EnhanceSwitch defaults to Enabled (no need to set explicitly)\npython3 scripts/vod_aigc_video.py create \\\n    --model GV --model-version 3.1 \\\n    --prompt \"Smiling, walking towards me\" \\\n    --output-resolution 2K \\\n    --sub-app-id 1308104797\n```\n\n### 10.3 Resolution Support per Model\n\n| Model | Supported Resolutions |\n|---|---|\n| Kling | **720P, 1080P, 4K** (default: 720P; interface tested: all versions accept 4K) |\n| Jimeng | **ExtInfo `width`/`height` only** (does not support OutputConfig.Resolution) |\n| Hailuo | 768P, 1080P (default: 768P); H3 verified to output 2560x1440 |\n| Vidu | 720P, 1080P (default: 720P) |\n| GV | 3.1 series: 720P, 1080P (default: 720P); **omni: 720P, 1080P, 2K, 4K** (2K/4K are super-resolution output) |\n| H2 | **720P, 1080P, 2K, 4K** (1.0/1.1 identical) |\n| OS | 720P standard; also supports ExtInfo `width`/`height` |\n| PixVerse | **360P, 540P, 720P, 1080P, 4K** (interface tested: all versions accept 4K) |\n| Seedance | 1.0-pro-fast multiple tiers; 1.5-pro does NOT support 1080P |\n| Mingmou | **ExtInfo `width`/`height` only** |\n| Hunyuan 1.5 | **ExtInfo `size` only** |\n| Hunyuan 3d_2.0 + 3d_scene | 1080P (recommended) |\n\n> The table above shows the native output capabilities. **All models** can chain `--output-enhance-switch Enabled` to output 2K/4K (auto-enabled when 2K/4K is selected).\n\n---\n\n## 11 Custom Resolution via ExtInfo (width/height/size)\n\nSome video models do not support the standard `--output-resolution` parameter and require custom resolution via `--ext-info` (similar to image-side §13).\n\n### 11.1 Jimeng 3.0pro Custom width/height\n\n```bash\npython3 scripts/vod_aigc_video.py create \\\n    --model Jimeng --model-version 3.0pro \\\n    --prompt \"Ancient costume swordswoman dance\" \\\n    --ext-info '{\"AdditionalParameters\": \"{\\\"width\\\":1920, \\\"height\\\":1080}\"}' \\\n    --output-storage-mode Temporary \\\n    --sub-app-id 1308104797\n```\n\n### 11.2 Hunyuan 1.5 Custom size\n\n```bash\npython3 scripts/vod_aigc_video.py create \\\n    --model Hunyuan --model-version 1.5 \\\n    --prompt \"Floating clouds\" \\\n    --ext-info '{\"AdditionalParameters\": \"{\\\"size\\\":\\\"1280x720\\\"}\"}' \\\n    --output-storage-mode Temporary \\\n    --sub-app-id 1308104797\n```\n\n### 11.3 OS 2.0 Custom width/height\n\n```bash\npython3 scripts/vod_aigc_video.py create \\\n    --model OS --model-version 2.0 \\\n    --prompt \"abstract liquid flowing\" \\\n    --ext-info '{\"AdditionalParameters\": \"{\\\"width\\\":1920, \\\"height\\\":1080}\"}' \\\n    --output-storage-mode Temporary \\\n    --sub-app-id 1308104797\n```\n\n### 11.4 Mingmou 1.0 Custom width/height\n\n```bash\npython3 scripts/vod_aigc_video.py create \\\n    --model Mingmou --model-version 1.0 \\\n    --prompt \"City skyline at dusk\" \\\n    --ext-info '{\"AdditionalParameters\": \"{\\\"width\\\":1920, \\\"height\\\":1080}\"}' \\\n    --output-storage-mode Temporary \\\n    --sub-app-id 1308104797\n```\n\n### 11.5 ExtInfo Custom Resolution vs OutputConfig\n\n| Model | Standard `--output-resolution` | ExtInfo Custom Resolution |\n|---|---|---|\n| **Jimeng 3.0pro** | ❌ Not supported | ✅ `width`/`height` |\n| **Hunyuan 1.5** | ❌ Not supported | ✅ `size` |\n| **OS 2.0** | ✅ 720P native | ✅ `width`/`height` also supported |\n| **Mingmou 1.0** | ❌ Not supported | ✅ `width`/`height` |\n| Other models (Kling/Vidu/GV/Hailuo/PixVerse/Seedance) | ✅ Supported | — |\n\n---\n\n## 12 Key Constraints Discovered via Interface Testing\n\n> All constraints verified via real interface submissions (2026-06-30)\n\n### 12.1 ModelName Interface Names (verified)\n\n| ModelName | Interface Result | Notes |\n|---|---|---|\n| `Seedance` | ✅ Accepted | **Real interface name** (ByteDance Doubao video series) |\n| `SV` | ❌ **Rejected**: `ModelName SV is invalid` | Earlier skill docs incorrect, fixed |\n| `GV` | ✅ Accepted | Google Veo (no aliases — `Veo`/`GoogleVeo` all rejected) |\n| `Hailuo` | ✅ Accepted | MiniMax Hailuo (no aliases — `MiniMax` rejected) |\n| `Kling`/`Jimeng`/`Vidu`/`Hunyuan`/`Mingmou`/`OS`/`PixVerse` | ✅ Accepted | Single canonical name, no aliases |\n\n### 12.2 Vidu q3-mix / q3-drama Constraints\n\nInterface tested:\n\n```\nmodel Vidu q3-mix only supports reference generation, all FileInfos must have Usage=Reference\nmodel Vidu q3-drama only supports reference generation, FileInfos or SubjectInfos is required\n```\n\n**Correct usage**:\n\n```bash\npython3 scripts/vod_aigc_video.py create \\\n    --model Vidu --model-version q3-mix \\\n    --prompt \"Cinematic shot with natural lighting\" \\\n    --file-url \"https://e.com/ref.jpg\" \\\n    --file-usage Reference \\\n    --sub-app-id 1308104797\n```\n\n---\n\nFile v1.1.3:references/vod_create_aigc_advanced_custom_element.md\n\n# vod_create_aigc_advanced_custom_element — Parameters & Examples\n> This file is generated by splitting references, corresponding script: `scripts/vod_create_aigc_advanced_custom_element.py`\n\n### ⚠️ Common Parameter Mistakes\n\n| Incorrect Usage | Correct Usage | Description |\n|---------|---------|------|\n| `--name element-name` | `--element-name element-name` | Custom element parameters use the `element-` prefix |\n| `--image-list url1,url2` | `--element-image-list '{\"frontal_image\":\"...\",\"refer_images\":[...]}'` | Reference images use `--element-image-list` in JSON format |\n\n## Table of Contents\n\n**Parameter Reference**\n- [Common Parameters](#common-parameters)\n- [create Parameters](#create-p\n\nArchive v1.1.2: 41 files, 210211 bytes\n\nFiles: LICENSE.txt (1068b), references/vod_aigc_chat.md (13123b), references/vod_aigc_image.md (35279b), references/vod_aigc_token.md (5562b), references/vod_aigc_video.md (33739b), references/vod_create_aigc_advanced_custom_element.md (13536b), references/vod_create_scene_aigc_video_task.md (11243b), references/vod_describe_media.md (17113b), references/vod_describe_sub_app_ids.md (4469b), references/vod_describe_task.md (6543b), references/vod_import_media_knowledge.md (5628b), references/vod_process_image.md (6834b), references/vod_process_media.md (22571b), references/vod_pull_upload.md (5636b), references/vod_scene_aigc_image.md (15866b), references/vod_search_media_by_semantics.md (4346b), references/vod_search_media.md (10998b), references/vod_upload.md (9429b), scripts/requirements.txt (1043b), scripts/vod_aigc_chat.py (44311b), scripts/vod_aigc_image.py (22247b), scripts/vod_aigc_token.py (21356b), scripts/vod_aigc_video.py (31686b), scripts/vod_auto_upgrade.py (4539b), scripts/vod_create_aigc_advanced_custom_element.py (29847b), scripts/vod_create_scene_aigc_video_task.py (18950b), scripts/vod_describe_media.py (29340b), scripts/vod_describe_sub_app_ids.py (8567b), scripts/vod_describe_task.py (44130b), scripts/vod_import_media_knowledge.py (19811b), scripts/vod_load_env.py (17039b), scripts/vod_process_image.py (30821b), scripts/vod_process_media.py (86572b), scripts/vod_pull_upload.py (17689b), scripts/vod_scene_aigc_image.py (23303b), scripts/vod_search_media_by_semantics.py (19811b), scripts/vod_search_media.py (33146b), scripts/vod_upload.py (27142b), skill-card.md (3231b), SKILL.md (33600b), _meta.json (135b)\n\nArchive v1.1.1: 42 files, 207516 bytes\n\nFiles: desc.txt (2380b), LICENSE.txt (1068b), references/vod_aigc_chat.md (13123b), references/vod_aigc_image.md (35279b), references/vod_aigc_token.md (5562b), references/vod_aigc_video.md (29508b), references/vod_create_aigc_advanced_custom_element.md (13536b), references/vod_create_scene_aigc_video_task.md (11243b), references/vod_describe_media.md (17113b), references/vod_describe_sub_app_ids.md (4469b), references/vod_describe_task.md (6543b), references/vod_import_media_knowledge.md (5628b), references/vod_process_image.md (6834b), references/vod_process_media.md (22571b), references/vod_pull_upload.md (5636b), references/vod_scene_aigc_image.md (15866b), references/vod_search_media_by_semantics.md (4346b), references/vod_search_media.md (10998b), references/vod_upload.md (9429b), scripts/requirements.txt (1043b), scripts/vod_aigc_chat.py (44311b), scripts/vod_aigc_image.py (22166b), scripts/vod_aigc_token.py (20811b), scripts/vod_aigc_video.py (26775b), scripts/vod_auto_upgrade.py (4539b), scripts/vod_create_aigc_advanced_custom_element.py (29771b), scripts/vod_create_scene_aigc_video_task.py (18869b), scripts/vod_describe_media.py (29259b), scripts/vod_describe_sub_app_ids.py (7859b), scripts/vod_describe_task.py (44054b), scripts/vod_import_media_knowledge.py (19735b), scripts/vod_load_env.py (17039b), scripts/vod_process_image.py (30745b), scripts/vod_process_media.py (86506b), scripts/vod_pull_upload.py (17623b), scripts/vod_scene_aigc_image.py (23222b), scripts/vod_search_media_by_semantics.py (19735b), scripts/vod_search_media.py (33070b), scripts/vod_upload.py (27061b), skill-card.md (3465b), SKILL.md (33201b), _meta.json (135b)\n\nArchive v1.0.9: 42 files, 183635 bytes\n\nFiles: desc.txt (2380b), LICENSE.txt (1068b), references/vod_aigc_chat.md (13103b), references/vod_aigc_image.md (14325b), references/vod_aigc_token.md (5553b), references/vod_aigc_video.md (12679b), references/vod_create_aigc_advanced_custom_element.md (13528b), references/vod_create_scene_aigc_video_task.md (11230b), references/vod_describe_media.md (17084b), references/vod_describe_sub_app_ids.md (4457b), references/vod_describe_task.md (6532b), references/vod_import_media_knowledge.md (5615b), references/vod_process_image.md (6826b), references/vod_process_media.md (22543b), references/vod_pull_upload.md (5624b), references/vod_scene_aigc_image.md (15847b), references/vod_search_media_by_semantics.md (4336b), references/vod_search_media.md (10961b), references/vod_upload.md (9412b), scripts/requirements.txt (1043b), scripts/vod_aigc_chat.py (44292b), scripts/vod_aigc_image.py (18571b), scripts/vod_aigc_token.py (20480b), scripts/vod_aigc_video.py (22899b), scripts/vod_auto_upgrade.py (2238b), scripts/vod_create_aigc_advanced_custom_element.py (26487b), scripts/vod_create_scene_aigc_video_task.py (15536b), scripts/vod_describe_media.py (28738b), scripts/vod_describe_sub_app_ids.py (7740b), scripts/vod_describe_task.py (43550b), scripts/vod_import_media_knowledge.py (16396b), scripts/vod_load_env.py (14167b), scripts/vod_process_image.py (30203b), scripts/vod_process_media.py (81888b), scripts/vod_pull_upload.py (14318b), scripts/vod_scene_aigc_image.py (19888b), scripts/vod_search_media_by_semantics.py (19208b), scripts/vod_search_media.py (32541b), scripts/vod_upload.py (23710b), skill-card.md (3735b), SKILL.md (31304b), _meta.json (135b)\n\nArchive v1.0.8: 41 files, 178350 bytes\n\nFiles: desc.txt (2380b), LICENSE.txt (1068b), references/vod_aigc_chat.md (11501b), references/vod_aigc_image.md (14241b), references/vod_aigc_token.md (5678b), references/vod_aigc_video.md (11741b), references/vod_create_aigc_advanced_custom_element.md (13703b), references/vod_create_scene_aigc_video_task.md (11415b), references/vod_describe_media.md (17189b), references/vod_describe_sub_app_ids.md (4457b), references/vod_describe_task.md (6564b), references/vod_import_media_knowledge.md (5790b), references/vod_process_image.md (6826b), references/vod_process_media.md (22587b), references/vod_pull_upload.md (5624b), references/vod_scene_aigc_image.md (16022b), references/vod_search_media_by_semantics.md (4336b), references/vod_search_media.md (11086b), references/vod_upload.md (9430b), scripts/requirements.txt (1044b), scripts/vod_aigc_chat.py (25657b), scripts/vod_aigc_image.py (18109b), scripts/vod_aigc_token.py (19981b), scripts/vod_aigc_video.py (20382b), scripts/vod_create_aigc_advanced_custom_element.py (26085b), scripts/vod_create_scene_aigc_video_task.py (15464b), scripts/vod_describe_media.py (28666b), scripts/vod_describe_sub_app_ids.py (7668b), scripts/vod_describe_task.py (43478b), scripts/vod_import_media_knowledge.py (15994b), scripts/vod_load_env.py (14167b), scripts/vod_process_image.py (29801b), scripts/vod_process_media.py (81382b), scripts/vod_pull_upload.py (14246b), scripts/vod_scene_aigc_image.py (19816b), scripts/vod_search_media_by_semantics.py (18806b), scripts/vod_search_media.py (32139b), scripts/vod_upload.py (23638b), skill-card.md (3765b), SKILL.md (31186b), _meta.json (135b)\n\nArchive v1.0.6: 39 files, 168675 bytes\n\nFiles: desc.txt (2380b), LICENSE.txt (1068b), references/vod_aigc_chat.md (11501b), references/vod_aigc_image.md (14241b), references/vod_aigc_token.md (5678b), references/vod_aigc_video.md (11741b), references/vod_create_aigc_advanced_custom_element.md (13703b), references/vod_create_scene_aigc_video_task.md (11415b), references/vod_describe_media.md (17189b), references/vod_describe_sub_app_ids.md (4457b), references/vod_describe_task.md (6564b), references/vod_import_media_knowledge.md (5790b), references/vod_process_image.md (6826b), references/vod_process_media.md (22587b), references/vod_pull_upload.md (5624b), references/vod_scene_aigc_image.md (16022b), references/vod_search_media_by_semantics.md (4336b), references/vod_search_media.md (11086b), references/vod_upload.md (9430b), scripts/vod_aigc_chat.py (24327b), scripts/vod_aigc_image.py (17131b), scripts/vod_aigc_token.py (19425b), scripts/vod_aigc_video.py (19404b), scripts/vod_create_aigc_advanced_custom_element.py (25328b), scripts/vod_create_scene_aigc_video_task.py (14582b), scripts/vod_describe_media.py (27688b), scripts/vod_describe_sub_app_ids.py (6786b), scripts/vod_describe_task.py (41153b), scripts/vod_import_media_knowledge.py (14414b), scripts/vod_load_env.py (12207b), scripts/vod_process_image.py (27333b), scripts/vod_process_media.py (73886b), scripts/vod_pull_upload.py (12325b), scripts/vod_scene_aigc_image.py (18309b), scripts/vod_search_media_by_semantics.py (17370b), scripts/vod_search_media.py (30216b), scripts/vod_upload.py (20910b), SKILL.md (28443b), _meta.json (135b)","readmeExcerpt":"Skill: Tencent VOD Intl. Owner: tencent-mpaas-skills Summary: Tencent Cloud VOD (Video on Demand) command generation assistant. Must trigger whenever the user's request involves any VOD operation: [Upload] local/URL pull upload, expiration/SessionId/storage path; [Media Processing] transcode/TESHD/screenshot/sprite/enhance/real-person/drama/scene/remux/HLS/GIF Tags: latest:1.1.3 Version history: v1.1.3 | 2026-08-09T1","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"python3 scripts/vod_load_env.py --check-only"},{"language":"bash","snippet":"# Required\nTENCENTCLOUD_SECRET_ID=your-secret-id\nTENCENTCLOUD_SECRET_KEY=your-secret-key\n\n# Optional\nTENCENTCLOUD_REGION=ap-guangzhou               # default: ap-guangzhou\nTENCENTCLOUD_VOD_AIGC_TOKEN=your-aigc-token    # For AIGC LLM Chat only\nTENCENTCLOUD_VOD_SUB_APP_ID=your-sub-app-id    # Used for sub-application operations"},{"language":"bash","snippet":"export TENCENTCLOUD_SECRET_ID=\"your-secret-id\"\nexport TENCENTCLOUD_SECRET_KEY=\"your-secret-key\"\nexport TENCENTCLOUD_REGION=\"your-api-region\" # default: ap-guangzhou\nexport TENCENTCLOUD_VOD_AIGC_TOKEN=\"your-aigc-token\"   # For AIGC LLM Chat only\nexport TENCENTCLOUD_VOD_SUB_APP_ID=\"your-sub-app-id\"   # Optional, used by some scripts\n\npython3 -m pip install -r scripts/requirements.txt"},{"language":"bash","snippet":"> python3 -m pip install -r scripts/requirements.txt --upgrade\n>"},{"language":"bash","snippet":"python3 scripts/vod_aigc_audio.py create \\\n    --model Kling --scene-type sfx \\\n    --prompt \"fireworks sound during Chinese New Year celebration\" \\\n    --output-storage-mode Temporary --output-duration 6 \\\n    --sub-app-id 1308104797"},{"language":"bash","snippet":"python3 scripts/vod_aigc_audio.py create \\\n    --model Kling --scene-type sfx \\\n    --video-url \"https://example.com/ref.mp4\" \\\n    --prompt \"gentle wind sound, distant bird calls, occasional footsteps, page turning, rain hitting the window\" \\\n    --bgm-prompt \"healing piano music, soft string accompaniment, warm and soothing melody\" \\\n    --asmr-mode true \\\n    --output-duration 6 \\\n    --sub-app-id 1308104797"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: tencent-vod-intl\ndescription: \"Tencent Cloud VOD (Video on Demand) command generation assistant. Must trigger whenever the user's request involves any VOD operation: [Upload] local/URL pull upload, expiration/SessionId/storage path; [Media Processing] transcode/TESHD/screenshot/sprite/enhance/real-person/drama/scene/remux/HLS/GIF/adaptive bitrate/review/procedure; [Media Query] FileId query details/transcode/subtitles/cover/metadata; [AIGC] text2img/text2video/img2video (Kling/Hunyuan/Vidu/GG/GV/Hailuo/MJ/Qwen/SI/OG/Jimeng/Mingmou/OS/Seedance/PixVerse), LLM chat (GPT/Gemini models, streaming), scene AIGC/outfit change/image expansion/product image/custom elements; [AIGC Audio] text-to-sfx/video-to-sfx/text-to-music/BGM/ASMR (Kling/MiniMaxMusic/GL); [AIGC Token/Usage] token management, usage stats; [Search] name/semantic/knowledge base; [Image] super-res/denoise/enhance/understand; [Sub-app/Task] sub-app query, task status. Do NOT trigger: MPS operations, COS direct upload, live streaming.\"\nmetadata:\n  version: \"1.1.3\"\n---\n\n# Tencent Cloud Video on Demand (VOD) Service\n\n## Role Definition\n\nYou are a professional assistant for Tencent Cloud VOD (Video on Demand), helping users generate correct Python script commands.\n\n## Output Specification\n\n1. **Output commands only** — no explanations, no filler text\n2. Command format: `python3 scripts/<script-name>.py [subcommand] [parameters]`\n3. All scripts support `--dry-run` (simulate execution)\n4. **Links output after task completion (pre-signed download links, playback URLs, etc.) must be presented in Markdown hyperlink format**, i.e. `[description](URL)` — must not be output as code blocks or plain text.\n\n> 💰 **Cost Notice**: This Skill calls Tencent Cloud VOD services and will incur charges, including transcoding fees, AI processing fees, storage fees, etc. When a task has not yet returned a result, do not manually re-submit the request, as this will result in duplicate charges. For detailed pricing, refer to [Tencent Cloud VOD Pricing](https://cloud.tencent.com/document/product/266/2838). A cost notice **must** be given each time a **processing script** is called (transcoding/enhancement/screenshot/AIGC/image processing/knowledge base import, etc.); no notice is needed for query scripts (vod_describe_task/vod_describe_media/vod_search_media/vod_describe_sub_app_ids) or upload scripts (vod_upload/vod_pull_upload). **Before invoking any processing script, you must first restate the exact command to the user and obtain explicit confirmation (\"Proceed?\") before submission; when parameters are uncertain or for high-cost operations (e.g. AIGC video generation, long-video transcoding, batch image processing, knowledge base import), prefer running with `--dry-run` first to preview**. Users are also advised to configure budget alerts and monthly caps at the [Tencent Cloud Billing Center](https://console.cloud.tencent.com/expense) to prevent runaway spend.\n\nTencent Cloud's official Python SDK is used "},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn70yhz47k5f2a22dsfdhthzqx83k6t0\",\n  \"slug\": \"tencent-vod-intl\",\n  \"version\": \"1.1.3\",\n  \"publishedAt\": 1786282114378\n}"},{"path":"references/vod_aigc_audio.md","content":"# vod_aigc_audio.py Reference\n\nVOD AIGC audio generation task tool, based on the `CreateAigcAudioTask` API.\nSupports text-to-sound-effect / video-to-sound-effect (Kling), text-to-music (MiniMaxMusic / GL(Google Lyria)).\n\n## Parameters\n\n### Basic Parameters\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--model` | enum | ❌ | Model name: `Kling` (sound effect) / `MiniMaxMusic` / `GL` (music) |\n| `--model-version` | string | ❌ | Model version; **recommended to leave unset for Kling** (uses the system default stable version, shown empty in doc examples); `MiniMaxMusic` supports `2.0/2.5/2.6/3.0`; `GL` supports `3.0-clip/3.0-pro` |\n| `--scene-type` | enum | ❌ | Scene type: `sfx` (sound effect, Kling-only) / `music` (music, MiniMaxMusic/GL-only) |\n| `--prompt` | string | ❌ | Description (prompt) of the audio to generate |\n\n### Reference Video Parameters (video-to-sound-effect scenario)\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--video-id` | string | ❌ | VOD FileId of the reference video |\n| `--video-url` | string | ❌ | URL of the reference video |\n| `--video-infos` | string | ❌ | JSON array of multiple reference videos, format: `[{\"Type\":\"Url\",\"Url\":\"...\"}]`; mutually exclusive with `--video-id`/`--video-url` (single-file form takes precedence) |\n\n### Reference Audio Parameters (e.g. generating music from an input audio)\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--audio-id` | string | ❌ | VOD FileId of the reference audio |\n| `--audio-url` | string | ❌ | URL of the reference audio |\n| `--audio-infos` | string | ❌ | JSON array of multiple reference audios, format: `[{\"Type\":\"Url\",\"Url\":\"...\"}]` |\n\n### AdditionalParameters Convenience Parameters\n\nThe `AdditionalParameters` field of `CreateAigcAudioTask` is used to pass model-specific scenario parameters (as a JSON string). The script provides the following convenience parameters, which are automatically merged into the same JSON:\n\n| Parameter | Type | Description |\n|------|------|------|\n| `--bgm-prompt` | string | BGM generation prompt (**video-to-sound-effect scenario, Kling**), merged as `AdditionalParameters.bgm_prompt` |\n| `--asmr-mode` | enum(`true`/`false`) | Whether to enable ASMR mode (enhances detailed sound effects, good for highly immersive content), merged as `AdditionalParameters.asmr_mode` (boolean) |\n| `--lyrics` | string | Lyrics content (**text-to-music scenario, MiniMaxMusic**), merged as `AdditionalParameters.lyrics` |\n| `--additional-parameters` | string | Reserved field, raw JSON string passthrough, merged with the convenience parameters above (convenience params take precedence) |\n\n### Output Configuration Parameters\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--output-storage-mode` | enum | ❌ | Storage mode: `Permanent` / `Temporary` (default) |\n| `--output-media-name` | string | ❌ | Output file name, up to 64 characters |\n| `--output-cla"},{"path":"references/vod_aigc_chat.md","content":"# vod_aigc_chat — Detailed Parameters and Examples\n> This file is generated by splitting references, corresponding script: `scripts/vod_aigc_chat.py`\n\n### ⚠️ Common Parameter Errors\n\n| Incorrect Usage | Correct Usage | Description |\n|---------|---------|------|\n| `--message \"A\" --message \"B\"` | `--messages '[{\"role\":\"user\",\"content\":\"A\"},...]'` | For multi-turn conversations, use `--messages` JSON array; do not repeat `--message` |\n\n## Parameter Reference\n### General Parameters\n\n| Parameter | Type | Description |\n|------|------|------|\n| `--token` | string | AIGC API Token (can also be set via environment variable `TENCENTCLOUD_VOD_AIGC_TOKEN`) |\n| `--model` / `-m` | string | Model to use (default `gpt-5.1`, see supported models list) |\n| `--json` | flag | Output full response in JSON format (only valid for `chat` subcommand; `stream` subcommand uses SSE streaming output, this parameter has no effect) |\n| `--dry-run` | flag | Preview request body without sending the request |\n| `--timeout` | int | Request timeout in seconds (default 120) |\n| `--output-file` | path | Save response to specified file path (`chat` subcommand saves full response JSON; `stream` subcommand saves plain text content) |\n| `--no-usage` | flag | Do not display token usage statistics |\n\n### Message Content Parameters\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--message` / `-q` | string | ✅* | User message content (simple single-turn conversation) |\n| `--messages` | JSON | ✅* | Complete messages JSON array (multi-turn conversation, mutually exclusive with `--message`) |\n| `--system` / `-s` | string | - | System prompt (sets AI role/persona; when used together with `--messages`, this parameter is ignored — system should be embedded in the messages JSON) |\n\n### Multimodal Parameters\n\n| Parameter | Type | Description |\n|------|------|------|\n| `--image-url` | string | Image URL (supported by both OpenAI and Gemini, size limit 70MB) |\n| `--audio-base64` | string | Base64-encoded audio data (Gemini only) |\n| `--audio-format` | mp3/wav | Audio format (default mp3) |\n| `--file-url` | string | File/video URL (Gemini only, size limit 70MB) |\n| `--file-name` | string | File name (used together with `--file-url`) |\n\n### Generation Control Parameters\n\n| Parameter | Type | Description |\n|------|------|------|\n| `--temperature` / `-t` | float | Output randomness (0~2; lower = more precise, higher = more creative; if not specified, server default is used, typically 0.7) |\n| `--max-tokens` | int | Maximum number of tokens to generate |\n| `--thinking` | flag | Enable reasoning/thinking mode (CoT); **Note: not supported by gpt-5.1/5.2/4o — a warning will be printed and the parameter will be automatically ignored** |\n| `--reasoning-effort` | string | Thinking level: `none`/`minimal`/`low`/`medium`/`high`/`xhigh` |\n| `--response-format` | string | Output format: `text` (default) / `json` (JSON object) / `json_schema` (same as `json`, fallback is json_object; f"},{"path":"references/vod_aigc_image.md","content":"# vod_aigc_image — Detailed Parameters and Examples\n> This file is generated by splitting references, corresponding script: `scripts/vod_aigc_image.py`\n\n### ⚠️ Common Parameter Errors\n\n| Incorrect Usage | Correct Usage | Description |\n|---------|---------|------|\n| `--model hunyuan-3.0` | `--model Hunyuan --model-version 3.0` | Model name and version number are separate |\n| `--model vidu-q2` or `--model vidu` | `--model Vidu --model-version q2` | **Model name must be capitalized (`Vidu`), version number uses a separate `--model-version q2`** |\n| `--aspect-ratio 16:9` | `--output-aspect-ratio 16:9` | Image generation output parameters use the `output-` prefix |\n| `--resolution 2K` | `--output-resolution 2K` | Image generation resolution parameters use the `output-` prefix |\n| `--class-id 10` | `--output-class-id 10` | **Image generation output class ID must use `--output-class-id`, not `--class-id`** |\n| `--file-ids id1,id2` | `--file-infos '[{\"Type\":\"File\",\"FileId\":\"id1\"}]'` | Multiple reference images use `--file-infos` JSON array |\n| `--person-generation Disallowed` | `--output-person-generation Disallowed` | Disabling person generation uses the `output-` prefix |\n| `--num-images 4` or `-n 4` | `--output-image-count 4` | **Multi-image output (OG 1-8, Kling 1-9)**; uses the `output-` prefix |\n| `--format png` | `--output-format png` | **Specify output format (OG)**: `jpeg`/`png`; uses the `output-` prefix |\n| `--mask-url ...` or `--mask-file-id ...` | `--file-infos '[{\"Type\":\"File\",\"FileId\":\"src\"},{\"Type\":\"File\",\"FileId\":\"mask\",\"ReferenceType\":\"mask\"}]'` | **OG mask editing**: first image is the source to edit, second is the mask (`ReferenceType=\"mask\"`) |\n\n## Parameter Description\n\n### General Parameters\n\n| Parameter | Type | Description |\n|------|------|------|\n| `--sub-app-id` | int | Sub-application ID (can also be set via environment variable `TENCENTCLOUD_VOD_SUB_APP_ID`) |\n| `--region` | string | Tencent Cloud region (default: `ap-guangzhou`) |\n| `--json` | flag | JSON format output |\n| `--dry-run` | flag | Only prints request parameter preview without sending the request |\n\n### create Parameters (Create Image Generation Task)\n\n#### Model Parameters (Required)\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--model` | string | ✅ | Model name (Hunyuan/Qwen/Vidu/Kling/MJ/GG) |\n| `--model-version` | string | - | Model version (uses default version if not specified) |\n\n#### Prompt Parameters\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--prompt` | string | ✅* | Prompt for generating the image (required when no reference image is provided) |\n| `--negative-prompt` | string | - | Prompt to prevent the model from generating certain content (negative prompt) |\n| `--enhance-prompt` | string | - | Whether to automatically optimize the prompt (Enabled/Disabled) |\n\n#### Reference Image Parameters\n\n| Parameter | Type | Required | Description |\n|------|------|------|------|\n| `--file-id`"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":2263,"uniquenessScore":38,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-10T17:38:06.541Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-10T17:38:06.541Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T21:57:51.419Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}