{"id":"4ef2e167-1680-43a6-bff4-80be3e60127b","entityType":"agent","slug":"clawhub-minimax-ai-dev-minimax-multimodal","name":"Minimax-Multimodal-Toolkit","canonicalUrl":"https://www.xpersona.co/agent/clawhub-minimax-ai-dev-minimax-multimodal","canonicalPath":"/agent/clawhub-minimax-ai-dev-minimax-multimodal","generatedAt":"2026-10-09T23:14:58.337Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-09T03:48:47.536Z","emptyReason":null},"description":"Use mmx to generate text, images, video, speech, and music via the MiniMax AI platform. Use when the user wants to create media content, chat with MiniMax mo... Skill: Minimax-Multimodal-Toolkit Owner: minimax-ai-dev Summary: Use mmx to generate text, images, video, speech, and music via the MiniMax AI platform. Use when the user wants to create media content, chat with MiniMax mo... Tags: latest:1.0.2 Version history: v1.0.2 | 2026-04-08T22:01:58.739Z | user **Switch to mmx-cli: major simplification and platform change** - Replaces all separate bash scripts and API referenc","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 5.7K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s17e3wmgte783ff2e9fwb0cpgh83hcx2:minimax-multimodal","sourceUrl":"https://clawhub.ai/minimax-ai-dev/minimax-multimodal","homepage":"https://clawhub.ai/minimax-ai-dev/skills/minimax-multimodal","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/minimax-ai-dev/minimax-multimodal","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/minimax-ai-dev/skills/minimax-multimodal","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":69,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Use mmx to generate text, images, video, speech, and music via the MiniMax AI platform. Use when the user wants to create media content, chat with MiniMax mo..."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T03:48:47.536Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T03:48:47.536Z","emptyReason":null},"stars":null,"forks":null,"downloads":5708,"packageName":null,"latestVersion":"1.0.2","tractionLabel":"5.7K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T03:48:47.536Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T03:48:47.536Z","lastCrawledAt":"2026-10-09T03:48:47.536Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T03:48:47.536Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.2","createdAt":"2026-04-08T22:01:58.739Z","changelog":"**Switch to `mmx-cli`: major simplification and platform change** - Replaces all separate bash scripts and API reference docs with a unified command-line interface (`mmx`). - Removes 15 legacy shell script and markdown reference files. - Updates skill name and description to reflect the new `mmx-cli` interface. - New workflow: install `mmx-cli` via npm and use a single CLI for text, image, video, speech, and music generation. - All instructions, options, and usage details are consolidated into the new SKILL.md focused on `mmx` commands and agent-friendly flags.","fileCount":3,"zipByteSize":5521},{"version":"1.0.1","createdAt":"2026-03-24T23:29:03.918Z","changelog":"**MiniMax multimodal toolkit now uses bash scripts, adds image generation, and drops Python dependencies:** - Switched all major scripts from Python to bash, requiring only `ffmpeg`, `jq`, and `curl` (no pip/Python needed) - Added support for image generation (text-to-image, image-to-image) via new image generation scripts - All scripts and workflows updated to use `.sh` bash scripts instead of `.py` Python scripts - Requires new environment variable `MINIMAX_API_HOST` (in addition to API key) with region selection instructions - Updated and simplified prerequisites and usage instructions; Python environment setup is no longer needed","fileCount":17,"zipByteSize":62146},{"version":"1.0.0","createdAt":"2026-03-20T09:19:25.577Z","changelog":"MiniMax multimodal skill initial release — generate speech, music, video, and process media via MiniMax AI. - Supports TTS (text-to-speech), voice cloning, and custom voice design. - Music and video creation: text-to-video, image-to-video, templates, multi-scene, and more. - FFmpeg-powered media tools: convert, concatenate, trim, extract audio/video. - Enforces all outputs to agent’s minimax-output/ directory, with explicit output paths required. - Includes comprehensive rules for TTS voice segmentation (single/multi-voice) and API key setup guidance.","fileCount":26,"zipByteSize":78890}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17e3wmgte783ff2e9fwb0cpgh83hcx2:minimax-multimodal","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-minimax-ai-dev-minimax-multimodal/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-minimax-ai-dev-minimax-multimodal/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-minimax-ai-dev-minimax-multimodal/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-minimax-ai-dev-minimax-multimodal/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-minimax-ai-dev-minimax-multimodal/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-minimax-ai-dev-minimax-multimodal/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-09T23:14:58.333Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-minimax-ai-dev-minimax-multimodal/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-minimax-ai-dev-minimax-multimodal/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-minimax-ai-dev-minimax-multimodal/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-minimax-ai-dev-minimax-multimodal/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-09T03:48:47.536Z","emptyReason":null},"readme":"Skill: Minimax-Multimodal-Toolkit\n\nOwner: minimax-ai-dev\n\nSummary: Use mmx to generate text, images, video, speech, and music via the MiniMax AI platform. Use when the user wants to create media content, chat with MiniMax mo...\n\nTags: latest:1.0.2\n\nVersion history:\n\nv1.0.2 | 2026-04-08T22:01:58.739Z | user\n\n**Switch to `mmx-cli`: major simplification and platform change**\n\n- Replaces all separate bash scripts and API reference docs with a unified command-line interface (`mmx`).\n- Removes 15 legacy shell script and markdown reference files.\n- Updates skill name and description to reflect the new `mmx-cli` interface.\n- New workflow: install `mmx-cli` via npm and use a single CLI for text, image, video, speech, and music generation.\n- All instructions, options, and usage details are consolidated into the new SKILL.md focused on `mmx` commands and agent-friendly flags.\n\nv1.0.1 | 2026-03-24T23:29:03.918Z | user\n\n**MiniMax multimodal toolkit now uses bash scripts, adds image generation, and drops Python dependencies:**\n\n- Switched all major scripts from Python to bash, requiring only `ffmpeg`, `jq`, and `curl` (no pip/Python needed)\n- Added support for image generation (text-to-image, image-to-image) via new image generation scripts\n- All scripts and workflows updated to use `.sh` bash scripts instead of `.py` Python scripts\n- Requires new environment variable `MINIMAX_API_HOST` (in addition to API key) with region selection instructions\n- Updated and simplified prerequisites and usage instructions; Python environment setup is no longer needed\n\nv1.0.0 | 2026-03-20T09:19:25.577Z | user\n\nMiniMax multimodal skill initial release — generate speech, music, video, and process media via MiniMax AI.\n\n- Supports TTS (text-to-speech), voice cloning, and custom voice design.\n- Music and video creation: text-to-video, image-to-video, templates, multi-scene, and more.\n- FFmpeg-powered media tools: convert, concatenate, trim, extract audio/video.\n- Enforces all outputs to agent’s minimax-output/ directory, with explicit output paths required.\n- Includes comprehensive rules for TTS voice segmentation (single/multi-voice) and API key setup guidance.\n\nArchive index:\n\nArchive v1.0.2: 3 files, 5521 bytes\n\nFiles: skill-card.md (1995b), SKILL.md (10960b), _meta.json (137b)\n\nFile v1.0.2:SKILL.md\n\n---\nname: mmx-cli\ndescription: Use mmx to generate text, images, video, speech, and music via the MiniMax AI platform. Use when the user wants to create media content, chat with MiniMax models, perform web search, or manage MiniMax API resources from the terminal.\n---\n\n# MiniMax CLI — Agent Skill Guide\n\nUse `mmx` to generate text, images, video, speech, music, and perform web search via the MiniMax AI platform.\n\n## Prerequisites\n\n```bash\n# Install\nnpm install -g mmx-cli\n\n# Auth (persisted to ~/.mmx/credentials.json)\nmmx auth login --api-key sk-xxxxx\n\n# Or pass per-call\nmmx text chat --api-key sk-xxxxx --message \"Hello\"\n```\n\nRegion is auto-detected. Override with `--region global` or `--region cn`.\n\n---\n\n## Agent Flags\n\nAlways use these flags in non-interactive (agent/CI) contexts:\n\n| Flag | Purpose |\n|---|---|\n| `--non-interactive` | Fail fast on missing args instead of prompting |\n| `--quiet` | Suppress spinners/progress; stdout is pure data |\n| `--output json` | Machine-readable JSON output |\n| `--async` | Return task ID immediately (video generation) |\n| `--dry-run` | Preview the API request without executing |\n| `--yes` | Skip confirmation prompts |\n\n---\n\n## Commands\n\n### text chat\n\nChat completion. Default model: `MiniMax-M2.7`.\n\n```bash\nmmx text chat --message <text> [flags]\n```\n\n| Flag | Type | Description |\n|---|---|---|\n| `--message <text>` | string, **required**, repeatable | Message text. Prefix with `role:` to set role (e.g. `\"system:You are helpful\"`, `\"user:Hello\"`) |\n| `--messages-file <path>` | string | JSON file with messages array. Use `-` for stdin |\n| `--system <text>` | string | System prompt |\n| `--model <model>` | string | Model ID (default: `MiniMax-M2.7`) |\n| `--max-tokens <n>` | number | Max tokens (default: 4096) |\n| `--temperature <n>` | number | Sampling temperature (0.0, 1.0] |\n| `--top-p <n>` | number | Nucleus sampling threshold |\n| `--stream` | boolean | Stream tokens (default: on in TTY) |\n| `--tool <json-or-path>` | string, repeatable | Tool definition JSON or file path |\n\n```bash\n# Single message\nmmx text chat --message \"user:What is MiniMax?\" --output json --quiet\n\n# Multi-turn\nmmx text chat \\\n  --system \"You are a coding assistant.\" \\\n  --message \"user:Write fizzbuzz in Python\" \\\n  --output json\n\n# From file\ncat conversation.json | mmx text chat --messages-file - --output json\n```\n\n**stdout**: response text (text mode) or full response object (json mode).\n\n---\n\n### image generate\n\nGenerate images. Model: `image-01`.\n\n```bash\nmmx image generate --prompt <text> [flags]\n```\n\n| Flag | Type | Description |\n|---|---|---|\n| `--prompt <text>` | string, **required** | Image description |\n| `--aspect-ratio <ratio>` | string | e.g. `16:9`, `1:1` |\n| `--n <count>` | number | Number of images (default: 1) |\n| `--subject-ref <params>` | string | Subject reference: `type=character,image=path-or-url` |\n| `--out-dir <dir>` | string | Download images to directory |\n| `--out-prefix <prefix>` | string | Filename prefix (default: `image`) |\n\n```bash\nmmx image generate --prompt \"A cat in a spacesuit\" --output json --quiet\n# stdout: image URLs (one per line in quiet mode)\n\nmmx image generate --prompt \"Logo\" --n 3 --out-dir ./gen/ --quiet\n# stdout: saved file paths (one per line)\n```\n\n---\n\n### video generate\n\nGenerate video. Default model: `MiniMax-Hailuo-2.3`. This is an async task — by default it polls until completion.\n\n```bash\nmmx video generate --prompt <text> [flags]\n```\n\n| Flag | Type | Description |\n|---|---|---|\n| `--prompt <text>` | string, **required** | Video description |\n| `--model <model>` | string | `MiniMax-Hailuo-2.3` (default) or `MiniMax-Hailuo-2.3-Fast` |\n| `--first-frame <path-or-url>` | string | First frame image |\n| `--callback-url <url>` | string | Webhook URL for completion |\n| `--download <path>` | string | Save video to specific file |\n| `--async` | boolean | Return task ID immediately |\n| `--no-wait` | boolean | Same as `--async` |\n| `--poll-interval <seconds>` | number | Polling interval (default: 5) |\n\n```bash\n# Non-blocking: get task ID\nmmx video generate --prompt \"A robot.\" --async --quiet\n# stdout: {\"taskId\":\"...\"}\n\n# Blocking: wait and get file path\nmmx video generate --prompt \"Ocean waves.\" --download ocean.mp4 --quiet\n# stdout: ocean.mp4\n```\n\n### video task get\n\nQuery status of a video generation task.\n\n```bash\nmmx video task get --task-id <id> [--output json]\n```\n\n### video download\n\nDownload a completed video by task ID.\n\n```bash\nmmx video download --file-id <id> [--out <path>]\n```\n\n---\n\n### speech synthesize\n\nText-to-speech. Default model: `speech-2.8-hd`. Max 10k chars.\n\n```bash\nmmx speech synthesize --text <text> [flags]\n```\n\n| Flag | Type | Description |\n|---|---|---|\n| `--text <text>` | string | Text to synthesize |\n| `--text-file <path>` | string | Read text from file. Use `-` for stdin |\n| `--model <model>` | string | `speech-2.8-hd` (default), `speech-2.6`, `speech-02` |\n| `--voice <id>` | string | Voice ID (default: `English_expressive_narrator`) |\n| `--speed <n>` | number | Speed multiplier |\n| `--volume <n>` | number | Volume level |\n| `--pitch <n>` | number | Pitch adjustment |\n| `--format <fmt>` | string | Audio format (default: `mp3`) |\n| `--sample-rate <hz>` | number | Sample rate (default: 32000) |\n| `--bitrate <bps>` | number | Bitrate (default: 128000) |\n| `--channels <n>` | number | Audio channels (default: 1) |\n| `--language <code>` | string | Language boost |\n| `--subtitles` | boolean | Include subtitle timing data |\n| `--pronunciation <from/to>` | string, repeatable | Custom pronunciation |\n| `--sound-effect <effect>` | string | Add sound effect |\n| `--out <path>` | string | Save audio to file |\n| `--stream` | boolean | Stream raw audio to stdout |\n\n```bash\nmmx speech synthesize --text \"Hello world\" --out hello.mp3 --quiet\n# stdout: hello.mp3\n\necho \"Breaking news.\" | mmx speech synthesize --text-file - --out news.mp3\n```\n\n---\n\n### music generate\n\nGenerate music. Model: `music-2.5`. Responds well to rich, structured descriptions.\n\n```bash\nmmx music generate --prompt <text> [--lyrics <text>] [flags]\n```\n\n| Flag | Type | Description |\n|---|---|---|\n| `--prompt <text>` | string | Music style description (can be detailed) |\n| `--lyrics <text>` | string | Song lyrics with structure tags. Use `\"\\u65e0\\u6b4c\\u8bcd\"` for instrumental. Cannot be used with `--instrumental` |\n| `--lyrics-file <path>` | string | Read lyrics from file. Use `-` for stdin |\n| `--vocals <text>` | string | Vocal style, e.g. `\"warm male baritone\"`, `\"bright female soprano\"`, `\"duet with harmonies\"` |\n| `--genre <text>` | string | Music genre, e.g. folk, pop, jazz |\n| `--mood <text>` | string | Mood or emotion, e.g. warm, melancholic, uplifting |\n| `--instruments <text>` | string | Instruments to feature, e.g. `\"acoustic guitar, piano\"` |\n| `--tempo <text>` | string | Tempo description, e.g. fast, slow, moderate |\n| `--bpm <number>` | number | Exact tempo in beats per minute |\n| `--key <text>` | string | Musical key, e.g. C major, A minor, G sharp |\n| `--avoid <text>` | string | Elements to avoid in the generated music |\n| `--use-case <text>` | string | Use case context, e.g. `\"background music for video\"`, `\"theme song\"` |\n| `--structure <text>` | string | Song structure, e.g. `\"verse-chorus-verse-bridge-chorus\"` |\n| `--references <text>` | string | Reference tracks or artists, e.g. `\"similar to Ed Sheeran\"` |\n| `--extra <text>` | string | Additional fine-grained requirements |\n| `--instrumental` | boolean | Generate instrumental music (no vocals). Cannot be used with `--lyrics` or `--lyrics-file` |\n| `--aigc-watermark` | boolean | Embed AI-generated content watermark |\n| `--format <fmt>` | string | Audio format (default: `mp3`) |\n| `--sample-rate <hz>` | number | Sample rate (default: 44100) |\n| `--bitrate <bps>` | number | Bitrate (default: 256000) |\n| `--out <path>` | string | Save audio to file |\n| `--stream` | boolean | Stream raw audio to stdout |\n\nAt least one of `--prompt` or `--lyrics` is required.\n\n```bash\n# Simple usage\nmmx music generate --prompt \"Upbeat pop\" --lyrics \"La la la...\" --out song.mp3 --quiet\n\n# Detailed prompt with vocal characteristics\nmmx music generate --prompt \"Warm morning folk\" \\\n  --vocals \"male and female duet, harmonies in chorus\" \\\n  --instruments \"acoustic guitar, piano\" \\\n  --bpm 95 \\\n  --lyrics-file song.txt \\\n  --out duet.mp3\n\n# Instrumental (use --instrumental flag)\nmmx music generate --prompt \"Cinematic orchestral, building tension\" --instrumental --out bgm.mp3\n```\n\n---\n\n### vision describe\n\nImage understanding via VLM. Provide either `--image` or `--file-id`, not both.\n\n```bash\nmmx vision describe (--image <path-or-url> | --file-id <id>) [flags]\n```\n\n| Flag | Type | Description |\n|---|---|---|\n| `--image <path-or-url>` | string | Local path or URL (auto base64-encoded) |\n| `--file-id <id>` | string | Pre-uploaded file ID (skips base64) |\n| `--prompt <text>` | string | Question about the image (default: `\"Describe the image.\"`) |\n\n```bash\nmmx vision describe --image photo.jpg --prompt \"What breed?\" --output json\n```\n\n**stdout**: description text (text mode) or full response (json mode).\n\n---\n\n### search query\n\nWeb search via MiniMax.\n\n```bash\nmmx search query --q <query>\n```\n\n| Flag | Type | Description |\n|---|---|---|\n| `--q <query>` | string, **required** | Search query |\n\n```bash\nmmx search query --q \"MiniMax AI\" --output json --quiet\n```\n\n---\n\n### quota show\n\nDisplay Token Plan usage and remaining quotas.\n\n```bash\nmmx quota show [--output json]\n```\n\n---\n\n## Tool Schema Export\n\nExport all commands as Anthropic/OpenAI-compatible JSON tool schemas:\n\n```bash\n# All tool-worthy commands (excludes auth/config/update)\nmmx config export-schema\n\n# Single command\nmmx config export-schema --command \"video generate\"\n```\n\nUse this to dynamically register mmx commands as tools in your agent framework.\n\n---\n\n## Exit Codes\n\n| Code | Meaning |\n|---|---|\n| 0 | Success |\n| 1 | General error |\n| 2 | Usage error (bad flags, missing args) |\n| 3 | Authentication error |\n| 4 | Quota exceeded |\n| 5 | Timeout |\n| 10 | Content filter triggered |\n\n---\n\n## Piping Patterns\n\n```bash\n# stdout is always clean data — safe to pipe\nmmx text chat --message \"Hi\" --output json | jq '.content'\n\n# stderr has progress/spinners — discard if needed\nmmx video generate --prompt \"Waves\" 2>/dev/null\n\n# Chain: generate image → describe it\nURL=$(mmx image generate --prompt \"A sunset\" --quiet)\nmmx vision describe --image \"$URL\" --quiet\n\n# Async video workflow\nTASK=$(mmx video generate --prompt \"A robot\" --async --quiet | jq -r '.taskId')\nmmx video task get --task-id \"$TASK\" --output json\nmmx video download --task-id \"$TASK\" --out robot.mp4\n```\n\n---\n\n## Configuration Precedence\n\nCLI flags → environment variables → `~/.mmx/config.json` → defaults.\n\n```bash\n# Persistent config\nmmx config set --key region --value cn\nmmx config show\n\n# Environment\nexport MINIMAX_API_KEY=sk-xxxxx\nexport MINIMAX_REGION=cn\n```\n\nFile v1.0.2:_meta.json\n\n{\n  \"ownerId\": \"kn70p6rdfg6k3598at7gm6r5fd82h5rs\",\n  \"slug\": \"minimax-multimodal\",\n  \"version\": \"1.0.2\",\n  \"publishedAt\": 1775685718739\n}\n\nFile v1.0.2:skill-card.md\n\n## Description:\n\nUse mmx to generate text, images, video, speech, and music via the MiniMax AI platform.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[minimax-ai-dev](https://clawhub.ai/user/minimax-ai-dev)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and agents use this skill to call the MiniMax CLI for multimodal generation, image understanding, web search, quota checks, and related API-resource workflows from the terminal.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill relies on a global npm install for mmx-cli, and the release evidence notes that the package is unpinned.\n\nMitigation: Verify the npm package and version before installation, and install only in environments where that package is trusted.\n\nRisk: The skill shows API-key usage patterns that can expose MiniMax credentials if copied directly into shell history or logs.\n\nMitigation: Prefer environment variables or secret-manager workflows, and avoid placing real API keys directly in commands.\n\nRisk: The --yes flag can bypass confirmations before quota-consuming operations.\n\nMitigation: Use --dry-run or review command parameters before using --yes in automated workflows.\n\n## Reference(s):\n\n\n## Skill Output:\n\n**Output Type(s):** [Guidance, Shell commands, Configuration, Text, JSON, Files]\n\n**Output Format:** [Markdown guidance with inline shell commands and expected CLI output formats]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [The skill guides mmx CLI calls that may return text, JSON, task IDs, file paths, media URLs, or generated media files depending on the command.]\n\n## Skill Version(s):\n\n1.0.2 (source: release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v1.0.1: 17 files, 62146 bytes\n\nFiles: references/image-api.md (3639b), references/music-api.md (1719b), references/tts-guide.md (3164b), references/tts-voice-catalog.md (33087b), references/video-api.md (4873b), references/video-prompt-guide.md (3678b), scripts/check_environment.sh (3579b), scripts/image/generate_image.sh (9690b), scripts/media_tools.sh (18341b), scripts/music/generate_music.sh (9571b), scripts/tts/generate_voice.sh (27857b), scripts/video/add_bgm.sh (7835b), scripts/video/generate_long_video.sh (17129b), scripts/video/generate_template_video.sh (7014b), scripts/video/generate_video.sh (11485b), SKILL.md (30136b), _meta.json (137b)\n\nFile v1.0.1:SKILL.md\n\n---\nname: minimax-multimodal-toolkit\ndescription: MiniMax multimodal model skill — use MiniMax  Multi-Modal models for speech, music, video, and image. Create voice, music, video, and images with MiniMax AI: TTS (text-to-speech, voice cloning, voice design, multi-segment), music (songs, instrumentals), video (text-to-video, image-to-video, start-end frame, subject reference, templates, long-form multi-scene), image (text-to-image, image-to-image with character reference), and media processing (convert, concat, trim, extract). Use when the user mentions MiniMax, multimodal generation, or wants speech/music/video/image AI, MiniMax APIs, or FFmpeg workflows alongside MiniMax outputs.\n---\n\n# MiniMax Multi-Modal Toolkit\n\nGenerate voice, music, video, and image content via MiniMax APIs — the unified entry for **MiniMax multimodal** use cases (audio + music + video + image). Includes voice cloning & voice design for custom voices, image generation with character reference, and FFmpeg-based media tools for audio/video format conversion, concatenation, trimming, and extraction.\n\n## Output Directory\n\n**All generated files MUST be saved to `minimax-output/` under the AGENT'S current working directory (NOT the skill directory).** Every script call MUST include an explicit `--output` / `-o` argument pointing to this location. Never omit the output argument or rely on script defaults.\n\n**Rules:**\n1. Before running any script, ensure `minimax-output/` exists in the agent's working directory (create if needed: `mkdir -p minimax-output`)\n2. Always use absolute or relative paths from the agent's working directory: `--output minimax-output/video.mp4`\n3. **Never** `cd` into the skill directory to run scripts — run from the agent's working directory using the full script path\n4. Intermediate/temp files (segment audio, video segments, extracted frames) are automatically placed in `minimax-output/tmp/`. They can be cleaned up when no longer needed: `rm -rf minimax-output/tmp`\n\n## Prerequisites\n\n```bash\nbrew install ffmpeg jq              # macOS (or apt install ffmpeg jq on Linux)\nbash scripts/check_environment.sh\n```\n\nNo Python or pip required — all scripts are pure bash using `curl`, `ffmpeg`, `jq`, and `xxd`.\n\n### API Host Configuration\n\nMiniMax provides two service endpoints for different regions. Set `MINIMAX_API_HOST` before running any script:\n\n| Region | Platform URL | API Host Value |\n|--------|-------------|----------------|\n| China Mainland（中国大陆） | https://platform.minimaxi.com | `https://api.minimaxi.com` |\n| Global（全球） | https://platform.minimax.io | `https://api.minimax.io` |\n\n```bash\n# China Mainland\nexport MINIMAX_API_HOST=\"https://api.minimaxi.com\"\n\n# or Global\nexport MINIMAX_API_HOST=\"https://api.minimax.io\"\n```\n\n**IMPORTANT — When API Host is missing:**\nBefore running any script, check if `MINIMAX_API_HOST` is set in the environment. If it is NOT configured:\n1. Ask the user which service endpoint their MiniMax account uses:\n   - **China Mainland** → `https://api.minimaxi.com`\n   - **Global** → `https://api.minimax.io`\n2. Instruct and help user to set it via `export MINIMAX_API_HOST=\"https://api.minimaxi.com\"` (or the global variant) in their terminal or add it to their shell profile (`~/.zshrc` / `~/.bashrc`) for persistence\n\n### API Key Configuration\n\nSet the `MINIMAX_API_KEY` environment variable before running any script:\n\n```bash\nexport MINIMAX_API_KEY=\"your-api-key-here\"\n```\n\nThe key starts with `sk-api-` or `sk-cp-`, obtainable from https://platform.minimaxi.com (China) or https://platform.minimax.io (Global)\n\n**IMPORTANT — When API Key is missing:**\nBefore running any script, check if `MINIMAX_API_KEY` is set in the environment. If it is NOT configured:\n1. Ask the user to provide their MiniMax API key\n2. Instruct and help user to set it via `export MINIMAX_API_KEY=\"sk-...\"` in their terminal or add it to their shell profile (`~/.zshrc` / `~/.bashrc`) for persistence\n\n## Key Capabilities\n\n| Capability | Description | Entry point |\n|------------|-------------|-------------|\n| TTS | Text-to-speech synthesis with multiple voices and emotions | `scripts/tts/generate_voice.sh` |\n| Voice Cloning | Clone a voice from an audio sample (10s–5min) | `scripts/tts/generate_voice.sh clone` |\n| Voice Design | Create a custom voice from a text description | `scripts/tts/generate_voice.sh design` |\n| Music Generation | Generate songs with lyrics or instrumental tracks | `scripts/music/generate_music.sh` |\n| Image Generation | Text-to-image, image-to-image with character reference | `scripts/image/generate_image.sh` |\n| Video Generation | Text-to-video, image-to-video, subject reference, templates | `scripts/video/generate_video.sh` |\n| Long Video | Multi-scene chained video with crossfade transitions | `scripts/video/generate_long_video.sh` |\n| Media Tools | Audio/video format conversion, concatenation, trimming, extraction | `scripts/media_tools.sh` |\n\n## TTS (Text-to-Speech)\n\nEntry point: `scripts/tts/generate_voice.sh`\n\n### IMPORTANT: Single voice vs Multi-segment — Choose the right approach\n\n| User intent | Approach |\n|-------------|----------|\n| Single voice / no multi-character need | `tts` command — generate the entire text in one call |\n| Multiple characters / narrator + dialogue | `generate` command with segments.json |\n\n**Default behavior:** When the user simply asks to generate speech/voice and does NOT mention multiple voices or characters, use the `tts` command directly with a single appropriate voice. Do NOT split into segments or use the multi-segment pipeline — just pass the full text to `tts` in one call.\n\nOnly use multi-segment `generate` when:\n- The user explicitly needs multiple voices/characters\n- The text requires narrator + character dialogue separation\n- The text exceeds **10,000 characters** (API limit per request) — in this case, split into segments with the same voice\n\n### Single-voice generation (DEFAULT)\n\n```bash\nbash scripts/tts/generate_voice.sh tts \"Hello world\" -o minimax-output/hello.mp3\nbash scripts/tts/generate_voice.sh tts \"你好世界\" -v female-shaonv -o minimax-output/hello_cn.mp3\n```\n\n### Multi-segment generation (multi-voice / audiobook / podcast)\n\n**Complete workflow — follow ALL steps in order:**\n\n1. **Write segments.json** — split text into segments with voice assignments (see format and rules below)\n2. **Run `generate` command** — this reads segments.json, generates audio for EACH segment via TTS API, then merges them into a single output file with crossfade\n\n```bash\n# Step 1: Write segments.json to minimax-output/\n# (use the Write tool to create minimax-output/segments.json)\n\n# Step 2: Generate audio from segments.json — this is the CRITICAL step\n# It generates each segment individually and merges them into one file\nbash scripts/tts/generate_voice.sh generate minimax-output/segments.json \\\n  -o minimax-output/output.mp3 --crossfade 200\n```\n\n**Do NOT skip Step 2.** Writing segments.json alone does nothing — you MUST run the `generate` command to actually produce audio.\n\n### Voice management\n\n```bash\n# List all available voices\nbash scripts/tts/generate_voice.sh list-voices\n\n# Voice cloning (from audio sample, 10s–5min)\nbash scripts/tts/generate_voice.sh clone sample.mp3 --voice-id my-voice\n\n# Voice design (from text description)\nbash scripts/tts/generate_voice.sh design \"A warm female narrator voice\" --voice-id narrator\n```\n\n### Audio processing\n\n```bash\nbash scripts/tts/generate_voice.sh merge part1.mp3 part2.mp3 -o minimax-output/combined.mp3\nbash scripts/tts/generate_voice.sh convert input.wav -o minimax-output/output.mp3\n```\n\n### TTS Models\n\n| Model | Notes |\n|-------|-------|\n| speech-2.8-hd | Recommended, auto emotion matching |\n| speech-2.8-turbo | Faster variant |\n| speech-2.6-hd | Previous gen, manual emotion |\n| speech-2.6-turbo | Previous gen, faster |\n\n### segments.json Format\n\nDefault crossfade between segments: **200ms** (`--crossfade 200`).\n\n```json\n[\n  { \"text\": \"Hello!\", \"voice_id\": \"female-shaonv\", \"emotion\": \"\" },\n  { \"text\": \"Welcome.\", \"voice_id\": \"male-qn-qingse\", \"emotion\": \"happy\" }\n]\n```\n\nLeave `emotion` empty for speech-2.8 models (auto-matched from text).\n\n### IMPORTANT: Multi-Segment Script Generation Rules (Audiobooks, Podcasts, etc.)\n\nWhen generating segments.json for audiobooks, podcasts, or any multi-character narration, you MUST split narration text from character dialogue into separate segments with distinct voices.\n\n**Rule: Narration and dialogue are ALWAYS separate segments.**\n\nA sentence like `\"Tom said: The weather is great today!\"` must be split into two segments:\n- Segment 1 (narrator voice): `\"Tom said:\"`\n- Segment 2 (character voice): `\"The weather is great today!\"`\n\n**Example — Audiobook with narrator + 2 characters:**\n\n```json\n[\n  { \"text\": \"Morning sunlight streamed into the classroom as students filed in one by one.\", \"voice_id\": \"narrator-voice\", \"emotion\": \"\" },\n  { \"text\": \"Tom smiled and turned to Lisa:\", \"voice_id\": \"narrator-voice\", \"emotion\": \"\" },\n  { \"text\": \"The weather is amazing today! Let's go to the park after school!\", \"voice_id\": \"tom-voice\", \"emotion\": \"happy\" },\n  { \"text\": \"Lisa thought for a moment, then replied:\", \"voice_id\": \"narrator-voice\", \"emotion\": \"\" },\n  { \"text\": \"Sure, but I need to drop off my backpack at home first.\", \"voice_id\": \"lisa-voice\", \"emotion\": \"\" },\n  { \"text\": \"They exchanged a smile and went back to listening to the lecture.\", \"voice_id\": \"narrator-voice\", \"emotion\": \"\" }\n]\n```\n\n**Key principles:**\n1. **Narrator** uses a consistent neutral narrator voice throughout\n2. **Each character** has a dedicated voice_id, maintained consistently across all their dialogue\n3. **Split at dialogue boundaries** — `\"He said:\"` is narrator, the quoted content is the character\n4. **Do NOT merge** narrator text and character speech into a single segment\n5. For characters without pre-existing voice_ids, use voice cloning or voice design to create them first, then reference the created voice_id in segments\n\n## Music Generation\n\nEntry point: `scripts/music/generate_music.sh`\n\n### IMPORTANT: Instrumental vs Lyrics — When to use which\n\n| Scenario | Mode | Action |\n|----------|------|--------|\n| BGM for video / voice / podcast | Instrumental (default) | Use `--instrumental` directly, do NOT ask user |\n| User explicitly asks to \"create music\" / \"make a song\" | Ask user first | Ask whether they want instrumental or with lyrics |\n\n**When adding background music to video or voice content**, always default to instrumental mode (`--instrumental`). Do not ask the user — BGM should never have vocals competing with the main content.\n\n**When the user explicitly asks to create/generate music as the primary task**, ask them whether they want:\n- Instrumental (pure music, no vocals)\n- With lyrics (song with vocals — user provides or you help write lyrics)\n\n```bash\n# Instrumental (for BGM or when user chooses instrumental)\nbash scripts/music/generate_music.sh \\\n  --instrumental \\\n  --prompt \"ambient electronic, atmospheric\" \\\n  --output minimax-output/ambient.mp3 --download\n\n# Song with lyrics (when user chooses vocal music)\nbash scripts/music/generate_music.sh \\\n  --lyrics \"[verse]\\nHello world\\n[chorus]\\nLa la la\" \\\n  --prompt \"indie folk, melancholic\" \\\n  --output minimax-output/song.mp3 --download\n\n# With style fields\nbash scripts/music/generate_music.sh \\\n  --lyrics \"[verse]\\nLyrics here\" \\\n  --genre \"pop\" --mood \"upbeat\" --tempo \"fast\" \\\n  --output minimax-output/pop_track.mp3 --download\n```\n\n### Music Model\n\nDefault model: `music-2.5`\n\n`music-2.5` does **not** support `--instrumental` directly. When instrumental music is needed, the script automatically applies a workaround:\n- Sets lyrics to `[intro] [outro]` (empty structural tags, no actual vocals), appends `pure music, no lyrics` to the prompt\n\nThis produces instrumental-style output without requiring manual intervention. You can always use `--instrumental` and the script handles the rest.\n\n## Image Generation\n\nEntry point: `scripts/image/generate_image.sh`\n\nModel: `image-01` — photorealistic image generation from text prompts, with optional character reference for image-to-image.\n\n### IMPORTANT: Mode Selection — t2i vs i2i\n\n| User intent | Mode |\n|-------------|------|\n| Generate image from text description (default) | `t2i` — text-to-image |\n| Generate image with a character reference photo (keep same person) | `i2i` — image-to-image |\n\n**Default behavior:** When the user asks to generate/create an image without mentioning a reference photo, use `t2i` mode (default). Only use `i2i` mode when the user provides a character reference image or explicitly asks to base the image on an existing person's appearance.\n\n### IMPORTANT: Aspect Ratio — Infer from user context\n\nDo NOT always default to `1:1`. Analyze the user's request and choose the most appropriate aspect ratio:\n\n| User intent / context | Recommended ratio | Resolution |\n|-----------------------|-------------------|------------|\n| 头像、图标、社交媒体头像、avatar、icon、profile pic | `1:1` | 1024×1024 |\n| 风景、横幅、桌面壁纸、landscape、banner、desktop wallpaper | `16:9` | 1280×720 |\n| 传统照片、经典比例、classic photo | `4:3` | 1152×864 |\n| 摄影作品、杂志封面、photography、magazine | `3:2` | 1248×832 |\n| 人像竖图、海报、portrait photo、poster | `2:3` | 832×1248 |\n| 竖版海报、书籍封面、tall poster、book cover | `3:4` | 864×1152 |\n| 手机壁纸、社交媒体故事、phone wallpaper、story、reel | `9:16` | 720×1280 |\n| 超宽全景、电影画幅、panoramic、cinematic ultrawide | `21:9` | 1344×576 |\n| 未指定特定需求 / ambiguous | `1:1` | 1024×1024 |\n\n### IMPORTANT: Image Count — When to generate multiple images\n\n| User intent | Count (`-n`) |\n|-------------|--------------|\n| Default / single image request | `1` (default) |\n| 用户说\"几张\"、\"多张\"、\"一些\" / \"a few\", \"several\" | `3` |\n| 用户说\"多种方案\"、\"备选\" / \"variations\", \"options\" | `3`–`4` |\n| 用户明确指定数量 | Use the specified number (1–9) |\n\n### Text-to-Image Examples\n\n```bash\n# Basic text-to-image\nbash scripts/image/generate_image.sh \\\n  --prompt \"A cat sitting on a rooftop at sunset, cinematic lighting, warm tones, photorealistic\" \\\n  -o minimax-output/cat.png\n\n# Landscape with inferred aspect ratio\nbash scripts/image/generate_image.sh \\\n  --prompt \"Mountain landscape with misty valleys, photorealistic, golden hour\" \\\n  --aspect-ratio 16:9 \\\n  -o minimax-output/landscape.png\n\n# Phone wallpaper (portrait 9:16)\nbash scripts/image/generate_image.sh \\\n  --prompt \"Aurora borealis over a snowy forest, vivid colors, magical atmosphere\" \\\n  --aspect-ratio 9:16 \\\n  -o minimax-output/wallpaper.png\n\n# Multiple variations\nbash scripts/image/generate_image.sh \\\n  --prompt \"Abstract geometric art, vibrant colors\" \\\n  -n 3 \\\n  -o minimax-output/art.png\n\n# With prompt optimizer\nbash scripts/image/generate_image.sh \\\n  --prompt \"A man standing on Venice Beach, 90s documentary style\" \\\n  --aspect-ratio 16:9 --prompt-optimizer \\\n  -o minimax-output/beach.png\n\n# Custom dimensions (must be multiple of 8)\nbash scripts/image/generate_image.sh \\\n  --prompt \"Product photo of a luxury watch on marble surface\" \\\n  --width 1024 --height 768 \\\n  -o minimax-output/watch.png\n```\n\n### Image-to-Image (Character Reference)\n\nUse a reference photo to generate images with the same character in new scenes. Best results with a single front-facing portrait. Supported formats: JPG, JPEG, PNG (max 10MB).\n\n```bash\n# Character reference — place same person in a new scene\nbash scripts/image/generate_image.sh \\\n  --mode i2i \\\n  --prompt \"A girl looking into the distance from a library window, warm afternoon light\" \\\n  --ref-image face.jpg \\\n  --aspect-ratio 16:9 \\\n  -o minimax-output/girl_library.png\n\n# Multiple character variations\nbash scripts/image/generate_image.sh \\\n  --mode i2i \\\n  --prompt \"A woman in a red dress at a gala event, elegant, cinematic\" \\\n  --ref-image face.jpg -n 3 \\\n  -o minimax-output/gala.png\n```\n\n### Aspect Ratio Reference\n\n| Ratio | Resolution | Best for |\n|-------|------------|----------|\n| `1:1` | 1024×1024 | Default, avatars, icons, social media |\n| `16:9` | 1280×720 | Landscape, banner, desktop wallpaper |\n| `4:3` | 1152×864 | Classic photo, presentations |\n| `3:2` | 1248×832 | Photography, magazine layout |\n| `2:3` | 832×1248 | Portrait photo, poster |\n| `3:4` | 864×1152 | Book cover, tall poster |\n| `9:16` | 720×1280 | Phone wallpaper, social story/reel |\n| `21:9` | 1344×576 | Ultra-wide panoramic, cinematic |\n\n### Key Options\n\n| Option | Description |\n|--------|-------------|\n| `--prompt TEXT` | Image description, max 1500 chars (required) |\n| `--aspect-ratio RATIO` | Aspect ratio (see table above). Infer from user context |\n| `--width PX` / `--height PX` | Custom size, 512–2048, must be multiple of 8, both required together. Overridden by `--aspect-ratio` if both set |\n| `-n N` | Number of images to generate, 1–9 (default 1) |\n| `--seed N` | Random seed for reproducibility. Same seed + same params → similar results |\n| `--prompt-optimizer` | Enable automatic prompt optimization by the API |\n| `--ref-image FILE` | Character reference image for i2i mode (local file or URL, JPG/JPEG/PNG, max 10MB) |\n| `--no-download` | Print image URLs instead of downloading files |\n| `--aigc-watermark` | Add AIGC watermark to generated images |\n\n## Video Generation\n\n### IMPORTANT: Single vs Multi-Segment — Choose the right script\n\n| User intent | Script to use |\n|-------------|---------------|\n| Default / no special request | `scripts/video/generate_video.sh` (single segment, **10s, 768P**) |\n| User explicitly asks for \"long video\", \"multi-scene\", \"story\", or duration > 10s | `scripts/video/generate_long_video.sh` (multi-segment) |\n\n**Default behavior:** Always use single-segment `generate_video.sh` with **duration 10s and resolution 768P** unless the user explicitly asks for a long video, multi-scene video, or specifies a total duration exceeding 10 seconds. Do NOT automatically split into multiple segments — a single 10s video is the standard output. Only use `generate_long_video.sh` when the user clearly needs multi-scene or longer content.\n\nEntry point (single video): `scripts/video/generate_video.sh`\nEntry point (long/multi-scene): `scripts/video/generate_long_video.sh`\n\n### Video Model Constraints (MUST follow)\n\n**Duration limits by model and resolution:**\n\n| Model | 720P | 768P | 1080P |\n|-------|------|------|-------|\n| MiniMax-Hailuo-2.3 | - | 6s or **10s** | 6s only |\n| MiniMax-Hailuo-2.3-Fast | - | 6s or **10s** | 6s only |\n| MiniMax-Hailuo-02 | - | 6s or **10s** | 6s only |\n| T2V-01 / T2V-01-Director | 6s only | - | - |\n| I2V-01 / I2V-01-Director / I2V-01-live | 6s only | - | - |\n| S2V-01 (ref) | 6s only | - | - |\n\n**Resolution options by model and duration:**\n\n| Model | 6s | 10s |\n|-------|-----|-----|\n| MiniMax-Hailuo-2.3 | 768P (default), 1080P | 768P only |\n| MiniMax-Hailuo-2.3-Fast | 768P (default), 1080P | 768P only |\n| MiniMax-Hailuo-02 | 512P, 768P (default), 1080P | 512P, 768P (default) |\n| Other models | 720P (default) | Not supported |\n\n**Key rules:**\n- **Default: 10s + 768P** (best balance of length and quality for MiniMax-Hailuo-2.3)\n- 1080P only supports 6s duration — if user requests 1080P, set `--duration 6`\n- 10s duration only works with 768P (or 512P on Hailuo-02) — never combine 10s + 1080P\n- Older models (T2V-01, I2V-01, S2V-01) only support 6s at 720P\n\n### IMPORTANT: Prompt Optimization (MUST follow before generating any video)\n\nBefore calling any video generation script, you MUST optimize the user's prompt by reading and applying `references/video-prompt-guide.md`. Never pass the user's raw description directly as `--prompt`.\n\n**Optimization steps:**\n\n1. **Apply the Professional Formula**: `Main subject + Scene + Movement + Camera motion + Aesthetic atmosphere`\n   - BAD: `\"A puppy in a park\"`\n   - GOOD: `\"A golden retriever puppy runs toward the camera on a sun-dappled grass path in a park, [跟随] smooth tracking shot, warm golden hour lighting, shallow depth of field, joyful atmosphere\"`\n\n2. **Add camera instructions** using `[指令]` syntax: `[推进]`, `[拉远]`, `[跟随]`, `[固定]`, `[左摇]`, etc.\n\n3. **Include aesthetic details**: lighting (golden hour, dramatic side lighting), color grading (warm tones, cinematic), texture (dust particles, rain droplets), atmosphere (intimate, epic, peaceful)\n\n4. **Keep to 1-2 key actions** for 6-10 second videos — do not overcrowd with events\n\n5. **For i2v mode** (image-to-video): Focus prompt on **movement and change only**, since the image already establishes the visual. Do NOT re-describe what's in the image.\n   - BAD: `\"A lake with mountains\"` (just repeating the image)\n   - GOOD: `\"Gentle ripples spread across the water surface, a breeze rustles the distant trees, [固定] fixed camera, soft morning light, peaceful and serene\"`\n\n6. **For multi-segment long videos**: Each segment's prompt must be self-contained and optimized individually. The i2v segments (segment 2+) should describe motion/change relative to the previous segment's ending frame.\n\n```bash\n# Text-to-video (default: 10s, 768P)\nbash scripts/video/generate_video.sh \\\n  --mode t2v \\\n  --prompt \"A golden retriever puppy bounds toward the camera on a sunlit grass path, [跟随] tracking shot, warm golden hour, shallow depth of field, joyful\" \\\n  --output minimax-output/puppy.mp4\n\n# Text-to-video with 1080P (must use --duration 6)\nbash scripts/video/generate_video.sh \\\n  --mode t2v \\\n  --prompt \"A golden retriever puppy bounds toward the camera\" \\\n  --duration 6 --resolution 1080P \\\n  --output minimax-output/puppy_hd.mp4\n\n# Image-to-video (prompt focuses on MOTION, not image content)\nbash scripts/video/generate_video.sh \\\n  --mode i2v \\\n  --prompt \"The petals begin to sway gently in the breeze, soft light shifts across the surface, [固定] fixed framing, dreamy pastel tones\" \\\n  --first-frame photo.jpg \\\n  --output minimax-output/animated.mp4\n\n# Start-end frame interpolation (sef mode uses MiniMax-Hailuo-02)\nbash scripts/video/generate_video.sh \\\n  --mode sef \\\n  --first-frame start.jpg --last-frame end.jpg \\\n  --output minimax-output/transition.mp4\n\n# Subject reference (face consistency, ref mode uses S2V-01, 6s only)\nbash scripts/video/generate_video.sh \\\n  --mode ref \\\n  --prompt \"A young woman in a white dress walks slowly through a sunlit garden, [跟随] smooth tracking, warm natural lighting, cinematic depth of field\" \\\n  --subject-image face.jpg \\\n  --duration 6 \\\n  --output minimax-output/person.mp4\n```\n\n### Long-form Video (Multi-scene)\n\nMulti-scene long videos chain segments together: the first segment generates via text-to-video (t2v), then each subsequent segment uses the last frame of the previous segment as its first frame (i2v). Segments are joined with crossfade transitions for smooth continuity. Default is 10 seconds per segment.\n\n**Workflow:**\n1. Segment 1: t2v — generated purely from the optimized text prompt\n2. Segment 2+: i2v — the previous segment's last frame becomes `first_frame_image`, prompt describes **motion and change from that ending state**\n3. All segments are concatenated with 0.5s crossfade transitions to eliminate jump cuts\n4. Optional: AI-generated background music is overlaid\n\n**Prompt rules for each segment:**\n- Each segment prompt MUST be independently optimized using the Professional Formula\n- Segment 1 (t2v): Full scene description with subject, scene, camera, atmosphere\n- Segment 2+ (i2v): Focus on **what changes and moves** from the previous ending frame. Do NOT repeat the visual description — the first frame already provides it\n- Maintain visual consistency: keep lighting, color grading, and style keywords consistent across segments\n- Each segment covers only 10 seconds of action — keep it focused\n\n```bash\n# Example: 3-segment story with optimized per-segment prompts (default: 10s/segment, 768P)\nbash scripts/video/generate_long_video.sh \\\n  --scenes \\\n    \"A lone astronaut stands on a red desert planet surface, wind blowing dust particles, [推进] slow push in toward the visor, dramatic rim lighting, cinematic sci-fi atmosphere\" \\\n    \"The astronaut turns and begins walking toward a distant glowing structure on the horizon, dust swirling around boots, [跟随] tracking from behind, vast desolate landscape, golden light from the structure\" \\\n    \"The astronaut reaches the structure entrance, a massive doorway pulses with blue energy, [推进] slow push in toward the doorway, light reflects off the visor, awe-inspiring epic scale\" \\\n  --music-prompt \"cinematic orchestral ambient, slow build, sci-fi atmosphere\" \\\n  --output minimax-output/long_video.mp4\n\n# With custom settings\nbash scripts/video/generate_long_video.sh \\\n  --scenes \"Scene 1 prompt\" \"Scene 2 prompt\" \\\n  --segment-duration 10 \\\n  --resolution 768P \\\n  --crossfade 0.5 \\\n  --music-prompt \"calm ambient background music\" \\\n  --output minimax-output/long_video.mp4\n```\n\n### Add Background Music\n\n```bash\nbash scripts/video/add_bgm.sh \\\n  --video input.mp4 \\\n  --generate-bgm --instrumental \\\n  --music-prompt \"soft piano background\" \\\n  --bgm-volume 0.3 \\\n  --output minimax-output/output_with_bgm.mp4\n```\n\n### Template Video\n\n```bash\nbash scripts/video/generate_template_video.sh \\\n  --template-id 392753057216684038 \\\n  --media photo.jpg \\\n  --output minimax-output/template_output.mp4\n```\n\n### Video Models\n\n| Mode | Default Model | Default Duration | Default Resolution | Notes |\n|------|--------------|-----------------|-------------------|-------|\n| t2v | MiniMax-Hailuo-2.3 | 10s | 768P | Latest text-to-video |\n| i2v | MiniMax-Hailuo-2.3 | 10s | 768P | Latest image-to-video |\n| sef | MiniMax-Hailuo-02 | 6s | 768P | Start-end frame |\n| ref | S2V-01 | 6s | 720P | Subject reference, 6s only |\n\n## Media Tools (Audio/Video Processing)\n\nEntry point: `scripts/media_tools.sh`\n\nStandalone FFmpeg-based utilities for format conversion, concatenation, extraction, trimming, and audio overlay. Use these when the user needs to process existing media files without generating new content via MiniMax API.\n\n### Video Format Conversion\n\n```bash\n# Convert between formats (mp4, mov, webm, mkv, avi, ts, flv)\nbash scripts/media_tools.sh convert-video input.webm -o output.mp4\nbash scripts/media_tools.sh convert-video input.mp4 -o output.mov\n\n# With quality / resolution / fps options\nbash scripts/media_tools.sh convert-video input.mp4 -o output.mp4 \\\n  --crf 18 --preset medium --resolution 1920x1080 --fps 30\n```\n\n### Audio Format Conversion\n\n```bash\n# Convert between formats (mp3, wav, flac, ogg, aac, m4a, opus, wma)\nbash scripts/media_tools.sh convert-audio input.wav -o output.mp3\nbash scripts/media_tools.sh convert-audio input.mp3 -o output.flac \\\n  --bitrate 320k --sample-rate 48000 --channels 2\n```\n\n### Video Concatenation\n\n```bash\n# Concatenate with crossfade transition (default 0.5s)\nbash scripts/media_tools.sh concat-video seg1.mp4 seg2.mp4 seg3.mp4 -o merged.mp4\n\n# Hard cut (no crossfade)\nbash scripts/media_tools.sh concat-video seg1.mp4 seg2.mp4 -o merged.mp4 --crossfade 0\n```\n\n### Audio Concatenation\n\n```bash\n# Simple concatenation\nbash scripts/media_tools.sh concat-audio part1.mp3 part2.mp3 -o combined.mp3\n\n# With crossfade\nbash scripts/media_tools.sh concat-audio part1.mp3 part2.mp3 -o combined.mp3 --crossfade 1\n```\n\n### Extract Audio from Video\n\n```bash\n# Extract as mp3\nbash scripts/media_tools.sh extract-audio video.mp4 -o audio.mp3\n\n# Extract as wav with higher bitrate\nbash scripts/media_tools.sh extract-audio video.mp4 -o audio.wav --bitrate 320k\n```\n\n### Video Trimming\n\n```bash\n# Trim by start/end time (seconds)\nbash scripts/media_tools.sh trim-video input.mp4 -o clip.mp4 --start 5 --end 15\n\n# Trim by start + duration\nbash scripts/media_tools.sh trim-video input.mp4 -o clip.mp4 --start 10 --duration 8\n```\n\n### Add Audio to Video (Overlay / Replace)\n\n```bash\n# Mix audio with existing video audio\nbash scripts/media_tools.sh add-audio --video video.mp4 --audio bgm.mp3 -o output.mp4 \\\n  --volume 0.3 --fade-in 2 --fade-out 3\n\n# Replace original audio entirely\nbash scripts/media_tools.sh add-audio --video video.mp4 --audio narration.mp3 -o output.mp4 \\\n  --replace\n```\n\n### Media File Info\n\n```bash\nbash scripts/media_tools.sh probe input.mp4\n```\n\n## Script Architecture\n\n```\nscripts/\n├── check_environment.sh         # Env verification (curl, ffmpeg, jq, xxd, API key)\n├── media_tools.sh               # Audio/video conversion, concat, trim, extract\n├── tts/\n│   └── generate_voice.sh        # Unified TTS CLI (tts, clone, design, list-voices, generate, merge, convert)\n├── music/\n│   └── generate_music.sh        # Music generation CLI\n├── image/\n│   └── generate_image.sh        # Image generation CLI (2 modes: t2i, i2i)\n└── video/\n    ├── generate_video.sh        # Video generation CLI (4 modes: t2v, i2v, sef, ref)\n    ├── generate_long_video.sh   # Multi-scene long video\n    ├── generate_template_video.sh # Template-based video\n    └── add_bgm.sh              # Background music overlay\n```\n\n## References\n\nRead these for detailed API parameters, voice catalogs, and prompt engineering:\n\n- [tts-guide.md](references/tts-guide.md) — TTS setup, voice management, audio processing, segment format, troubleshooting\n- [tts-voice-catalog.md](references/tts-voice-catalog.md) — Full voice catalog with IDs, descriptions, and parameter reference\n- [music-api.md](references/music-api.md) — Music generation API: endpoints, parameters, response format\n- [image-api.md](references/image-api.md) — Image generation API: text-to-image, image-to-image, parameters\n- [video-api.md](references/video-api.md) — Video API: endpoints, models, parameters, camera instructions, templates\n- [video-prompt-guide.md](references/video-prompt-guide.md) — Video prompt engineering: formulas, styles, image-to-video tips\n\nFile v1.0.1:_meta.json\n\n{\n  \"ownerId\": \"kn70p6rdfg6k3598at7gm6r5fd82h5rs\",\n  \"slug\": \"minimax-multimodal\",\n  \"version\": \"1.0.1\",\n  \"publishedAt\": 1774394943918\n}\n\nFile v1.0.1:references/image-api.md\n\n# MiniMax Image Generation API (image-01)\n\nSource: https://platform.minimaxi.com/docs/api-reference/image-generation-t2i and https://platform.minimaxi.com/docs/api-reference/image-generation-i2i\n\n## Endpoint\n\n`POST https://api.minimaxi.com/v1/image_generation`\n\n## Auth\n\n`Authorization: Bearer <MINIMAX_API_KEY>`\n\n## Request (JSON)\n\nRequired:\n- `model`: string — `image-01`\n- `prompt`: string (max 1500 chars) — text description of the desired image\n\nOptional:\n- `aspect_ratio`: string — image aspect ratio, default `1:1`. Options:\n  - `1:1` (1024×1024)\n  - `16:9` (1280×720)\n  - `4:3` (1152×864)\n  - `3:2` (1248×832)\n  - `2:3` (832×1248)\n  - `3:4` (864×1152)\n  - `9:16` (720×1280)\n  - `21:9` (1344×576)\n- `width`: integer — custom width in pixels. Range [512, 2048], must be multiple of 8. Overridden by `aspect_ratio` if both set.\n- `height`: integer — custom height in pixels. Same rules as `width`. Both `width` and `height` must be set together.\n- `response_format`: string — `url` (default, valid 24h) or `base64`\n- `n`: integer (1–9, default 1) — number of images to generate\n- `seed`: integer — random seed for reproducibility\n- `prompt_optimizer`: boolean (default `false`) — enable automatic prompt optimization\n- `aigc_watermark`: boolean (default `false`) — add AIGC watermark\n\n### Subject Reference (image-to-image)\n\n- `subject_reference`: array — character reference for image-to-image generation\n  - `type`: string — currently only `character` (portrait)\n  - `image_file`: string — reference image as public URL or Base64 Data URL (`data:image/jpeg;base64,...`). For best results, use a single person front-facing photo. Formats: JPG, JPEG, PNG. Max size: 10MB.\n\n## Example — Text-to-Image\n\n```json\n{\n  \"model\": \"image-01\",\n  \"prompt\": \"A man in a white t-shirt, full-body, standing front view, outdoors, with the Venice Beach sign in the background, Los Angeles. Fashion photography in 90s documentary style, film grain, photorealistic.\",\n  \"aspect_ratio\": \"16:9\",\n  \"response_format\": \"url\",\n  \"n\": 3,\n  \"prompt_optimizer\": true\n}\n```\n\n## Example — Image-to-Image (Character Reference)\n\n```json\n{\n  \"model\": \"image-01\",\n  \"prompt\": \"A girl looking into the distance from a library window\",\n  \"aspect_ratio\": \"16:9\",\n  \"subject_reference\": [\n    {\n      \"type\": \"character\",\n      \"image_file\": \"https://example.com/face.jpg\"\n    }\n  ],\n  \"n\": 2\n}\n```\n\n## Response\n\n```json\n{\n  \"id\": \"03ff3cd0820949eb8a410056b5f21d38\",\n  \"data\": {\n    \"image_urls\": [\"https://...\", \"https://...\", \"https://...\"],\n    \"image_base64\": null\n  },\n  \"metadata\": {\n    \"success_count\": 3,\n    \"failed_count\": 0\n  },\n  \"base_resp\": {\n    \"status_code\": 0,\n    \"status_msg\": \"success\"\n  }\n}\n```\n\n- `data.image_urls`: array of image URLs (when `response_format` is `url`, valid 24h)\n- `data.image_base64`: array of Base64 strings (when `response_format` is `base64`)\n- `metadata.success_count`: number of successfully generated images\n- `metadata.failed_count`: number of images blocked by content safety\n\n## Status Codes\n\n| Code | Meaning |\n|------|---------|\n| 0 | Success |\n| 1002 | Rate limited, retry later |\n| 1004 | Auth failed, check API key |\n| 1008 | Insufficient balance |\n| 1026 | Prompt contains sensitive content |\n| 2013 | Invalid parameters |\n| 2049 | Invalid API key |\n\n## Notes\n\n- The API is synchronous — images are returned directly in the response (no polling needed).\n- URL format image links expire after 24 hours.\n- For image-to-image: upload a single front-facing portrait for best character reference results.\n- `width`/`height` are overridden by `aspect_ratio` if both provided.\n\nFile v1.0.1:references/music-api.md\n\n# MiniMax Music Generation API (music-2.5)\n\nSource: https://platform.minimaxi.com/docs/api-reference/music-generation\n\n## Endpoint\n\n`POST https://api.minimaxi.com/v1/music_generation`\n\n## Auth\n\n`Authorization: Bearer <MINIMAX_API_KEY>`\n\n## Request (JSON)\n\nRequired:\n- `model`: string — `music-2.5`\n- `lyrics`: string (1–3500 chars) — required. Use `\\n` for line breaks. Structure tags: `[Verse]`, `[Chorus]`, `[Bridge]`, `[Intro]`, `[Outro]`, etc.\n\nOptional:\n- `prompt`: string (0–2000 chars) — style description, optional but recommended.\n- `lyrics_optimizer`: boolean — auto-generate lyrics from prompt when lyrics is empty.\n- `stream`: boolean (default `false`)\n- `output_format`: `hex` (default) or `url`. URL valid for 24 hours.\n- `aigc_watermark`: boolean — top-level field, non-streaming only.\n- `audio_setting`:\n  - `sample_rate`: 16000, 24000, 32000, 44100\n  - `bitrate`: 32000, 64000, 128000, 256000\n  - `format`: mp3, wav, pcm\n\n## Example\n\n```json\n{\n  \"model\": \"music-2.5\",\n  \"prompt\": \"indie folk, melancholic, introspective\",\n  \"lyrics\": \"[verse]\\n...\\n[chorus]\\n...\",\n  \"aigc_watermark\": false,\n  \"audio_setting\": {\n    \"sample_rate\": 44100,\n    \"bitrate\": 256000,\n    \"format\": \"mp3\"\n  }\n}\n```\n\n## Response\n\n- `data.audio`: hex string or URL depending on `output_format`\n- `data.status`: 1 (generating), 2 (complete)\n- `extra_info`: duration, sample_rate, channels, bitrate, size\n- `base_resp.status_code`: 0 on success\n\n## Notes\n\n- `music-2.5` does not support `is_instrumental`. For instrumental music, use lyrics `[intro] [outro]` and add `pure music, no lyrics` to the prompt.\n- `prompt` is optional but recommended for better style control.\n- `stream=true` only supports `hex` output.\n\nFile v1.0.1:references/tts-guide.md\n\n# TTS Guide\n\n## Setup\n\n```bash\ncd skills/MiniMaxStudio\npip install -r requirements.txt\nbrew install ffmpeg   # macOS (or: sudo apt install ffmpeg)\nexport MINIMAX_API_KEY=\"your-api-key\"   # sk-api-xxx or sk-cp-xxx\npython scripts/check_environment.py\n```\n\n## Quick Test\n\n```bash\npython scripts/tts/generate_voice.py tts \"Hello, this is a test.\" -o test.mp3\n```\n\n## Voice Management\n\nList available voices:\n\n```bash\npython scripts/tts/generate_voice.py list-voices\n```\n\n### Voice Cloning\n\nCreate a custom voice from an audio sample:\n\n```bash\npython scripts/tts/generate_voice.py clone audio.mp3 --voice-id my-custom-voice\n\n# With preview\npython scripts/tts/generate_voice.py clone audio.mp3 --voice-id my-voice --preview \"Test text\" --preview-output preview.mp3\n```\n\nRequirements: 10s–5min duration, ≤20MB, mp3/wav/m4a format.\n\n### Voice Design\n\nDesign a voice from a text description:\n\n```bash\npython scripts/tts/generate_voice.py design \"A warm, gentle female voice\" --voice-id designed-voice\n```\n\nCustom voices expire after 7 days if not used with TTS.\n\n## Audio Processing\n\n### Merge\n\n```bash\npython scripts/tts/generate_voice.py merge file1.mp3 file2.mp3 -o combined.mp3\npython scripts/tts/generate_voice.py merge a.mp3 b.mp3 -o merged.mp3 --crossfade 300\n```\n\n### Convert\n\n```bash\npython scripts/tts/generate_voice.py convert input.wav -o output.mp3\npython scripts/tts/generate_voice.py convert input.wav -o output.mp3 --format mp3 --bitrate 192k --sample-rate 32000\n```\n\nFFmpeg required. Supported formats: mp3, wav, flac, ogg, m4a, aac, wma, opus, pcm.\n\n## Segment-Based TTS\n\nFor multi-voice, multi-emotion workflows using a `segments.json` file:\n\n```bash\n# Validate\npython scripts/tts/generate_voice.py validate segments.json --verbose\n\n# Generate\npython scripts/tts/generate_voice.py generate segments.json -o output.mp3 --crossfade 200\n```\n\n### segments.json Format\n\n```json\n[\n  { \"text\": \"Hello!\", \"voice_id\": \"female-shaonv\", \"emotion\": \"\" },\n  { \"text\": \"How are you?\", \"voice_id\": \"male-qn-qingse\", \"emotion\": \"happy\" }\n]\n```\n\n- `text` (required): Text to synthesize\n- `voice_id` (required): Voice ID\n- `emotion` (optional): For speech-2.8 models, leave empty for auto-matching. Valid values: happy, sad, angry, fearful, disgusted, surprised, calm, fluent, whisper\n\n## Troubleshooting\n\n| Error | Solution |\n|-------|----------|\n| `MINIMAX_API_KEY is required` | `export MINIMAX_API_KEY=\"key\"` |\n| `FFmpeg not installed` | `brew install ffmpeg` |\n| `Voice not found` | `python scripts/tts/generate_voice.py list-voices` |\n| `401 Unauthorized` | Check API key validity |\n| `429 Too Many Requests` | Add delays between requests |\n\n## API Details\n\n- **Endpoint**: `POST /v1/t2a_v2`\n- **Base URL**: `https://api.minimaxi.com`\n- **Auth**: `Authorization: Bearer {MINIMAX_API_KEY}`\n- **Models**: speech-2.8-hd (recommended), speech-2.8-turbo, speech-2.6-hd, speech-2.6-turbo, speech-02-hd, speech-02-turbo, speech-01-hd, speech-01-turbo\n- **Text limit**: 10,000 characters per request\n- **Pause marker**: `<#x#>` where x is seconds (0.01–99.99)\n- **Interjection tags** (speech-2.8 only): `(laughs)`, `(chuckle)`, `(coughs)`, `(sighs)`, `(breath)`, etc.\n\nFile v1.0.1:references/tts-voice-catalog.md\n\n# TTS Voice Catalog\n\n## Contents\n\n- [Voice Selection Guide](#voice-selection-guide)\n- [System Voices by Language](#system-voices-by-language)\n- [Voice Parameters](#voice-parameters)\n- [Custom Voices](#custom-voices)\n\n---\n\n## Voice Selection Guide\n\n### Decision Flow\n\n```\nContent type?\n├── Narration / Audiobook  → audiobook_female_1, audiobook_male_1\n├── News / Announcement    → Chinese (Mandarin)_News_Anchor, Chinese (Mandarin)_Male_Announcer\n├── Documentary            → doc_commentary\n└── Other                  → Select by: Gender → Age → Language → Personality\n```\n\n### Recommended Professional Voices\n\n| Scenario | Recommended | Characteristics |\n|----------|-------------|-----------------|\n| Narration / Audiobook | `audiobook_female_1`, `audiobook_male_1` | Clear articulation, good pacing, sustained performance |\n| News / Announcement | `Chinese (Mandarin)_News_Anchor`, `Chinese (Mandarin)_Male_Announcer` | Authoritative, professional pacing |\n| Documentary | `doc_commentary` | Professional, clear, consistent |\n\n### Selection Priority\n\n1. **Gender** (mandatory match) — male voices for male characters, female for female\n2. **Age** — Children / Youth / Adult / Elderly\n3. **Language** (must match content language)\n4. **Personality/tone** — choose best fit from matching candidates\n\n---\n\n## System Voices by Language\n\nGender: M = Male, F = Female, N = Neutral/Character\nAge: C = Child, Y = Youth, A = Adult, E = Elder\n\n### Chinese Mandarin (普通话)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `male-qn-qingse` | 青涩青年 | M | Y | Youthful, inexperienced | Campus, coming-of-age |\n| `male-qn-badao` | 霸道青年 | M | Y | Arrogant, dominant | Drama, romance |\n| `male-qn-daxuesheng` | 青年大学生 | M | Y | University student | Campus, educational |\n| `male-qn-jingying` | 精英青年 | M | A | Elite, ambitious | Business, professional |\n| `female-shaonv` | 少女 | F | Y | Young maiden | Romance, youth |\n| `female-yujie` | 御姐 | F | A | Mature, elegant | Romance, professional |\n| `female-chengshu` | 成熟女性 | F | A | Mature, reliable | Sophisticated, news |\n| `female-tianmei` | 甜美女性 | F | A | Sweet, pleasant | Soft, gentle |\n| `clever_boy` | 聪明男童 | M | C | Smart, witty | Children's, educational |\n| `cute_boy` | 可爱男童 | M | C | Adorable | Kids, animations |\n| `lovely_girl` | 萌萌女童 | F | C | Cute, sweet | Children's stories |\n| `cartoon_pig` | 卡通猪小琪 | N | C | Cartoon character | Animations, comedy |\n| `bingjiao_didi` | 病娇弟弟 | M | Y | Tsundere brother | Romance, character |\n| `junlang_nanyou` | 俊朗男友 | M | Y | Handsome boyfriend | Romance, dating |\n| `chunzhen_xuedi` | 纯真学弟 | M | Y | Innocent junior | Campus, youth |\n| `lengdan_xiongzhang` | 冷淡学长 | M | Y | Cool senior | Campus, romance |\n| `badao_shaoye` | 霸道少爷 | M | A | Arrogant young master | Drama, character |\n| `tianxin_xiaoling` | 甜心小玲 | F | Y | Sweet Xiao Ling | Character, animations |\n| `qiaopi_mengmei` | 俏皮萌妹 | F | Y | Playful cute girl | Comedy, light-hearted |\n| `wumei_yujie` | 妩媚御姐 | F | A | Charming mature woman | Romance, mature |\n| `diadia_xuemei` | 嗲嗲学妹 | F | Y | Flirty junior girl | Romance, dating |\n| `danya_xuejie` | 淡雅学姐 | F | Y | Elegant senior girl | Campus, romance |\n| `Arrogant_Miss` | 嚣张小姐 | F | A | Arrogant young lady | Drama, character |\n| `Robot_Armor` | 机械战甲 | N | A | Robotic armor | Sci-fi, games |\n| `audiobook_male_1` | 有声书男1 | M | A | Warm, engaging narrator | Audiobooks, stories |\n| `audiobook_female_1` | 有声书女1 | F | A | Gentle, expressive narrator | Audiobooks, stories |\n| `doc_commentary` | 纪录片解说 | M | A | Professional narrator | Documentary |\n| `Chinese (Mandarin)_News_Anchor` | 新闻女声 | F | A | News anchor | News, broadcasts |\n| `Chinese (Mandarin)_Male_Announcer` | 播报男声 | M | A | Male announcer | Announcements |\n| `Chinese (Mandarin)_Radio_Host` | 电台男主播 | M | A | Radio host | Podcasts, radio |\n| `Chinese (Mandarin)_Reliable_Executive` | 沉稳高管 | M | A | Reliable executive | Corporate, business |\n| `Chinese (Mandarin)_Gentleman` | 温润男声 | M | A | Gentle, refined | Narration, storytelling |\n| `Chinese (Mandarin)_Unrestrained_Young_Man` | 不羁青年 | M | Y | Unrestrained, casual | Entertainment |\n| `Chinese (Mandarin)_Southern_Young_Man` | 南方小哥 | M | Y | Southern accent | Regional, casual |\n| `Chinese (Mandarin)_Gentle_Youth` | 温润青年 | M | Y | Gentle young man | Narration, calm |\n| `Chinese (Mandarin)_Sincere_Adult` | 真诚青年 | M | Y | Sincere, genuine | Honest, genuine |\n| `Chinese (Mandarin)_Straightforward_Boy` | 率真弟弟 | M | Y | Frank, direct | Casual, direct |\n| `Chinese (Mandarin)_Pure-hearted_Boy` | 清澈邻家弟弟 | M | Y | Pure-hearted neighbor | Innocent, wholesome |\n| `Chinese (Mandarin)_Stubborn_Friend` | 嘴硬竹马 | M | Y | Stubborn childhood friend | Drama, character |\n| `Chinese (Mandarin)_Lyrical_Voice` | 抒情男声 | M | A | Lyrical, singing | Music, singing |\n| `Chinese (Mandarin)_Mature_Woman` | 傲娇御姐 | F | A | Tsundere mature woman | Romance, character |\n| `Chinese (Mandarin)_Sweet_Lady` | 甜美女声 | F | A | Sweet lady | Soft, gentle |\n| `Chinese (Mandarin)_Warm_Bestie` | 温暖闺蜜 | F | A | Warm bestie | Friendly, supportive |\n| `Chinese (Mandarin)_Warm_Girl` | 温暖少女 | F | Y | Warm young girl | Friendly, supportive |\n| `Chinese (Mandarin)_Soft_Girl` | 柔和少女 | F | Y | Soft, gentle | Calm, soothing |\n| `Chinese (Mandarin)_Crisp_Girl` | 清脆少女 | F | Y | Crisp, clear | Bright, clear |\n| `Chinese (Mandarin)_Gentle_Senior` | 温柔学姐 | F | Y | Gentle senior girl | Campus, supportive |\n| `Chinese (Mandarin)_Wise_Women` | 阅历姐姐 | F | A | Experienced, wise | Advice, guidance |\n| `Chinese (Mandarin)_HK_Flight_Attendant` | 港普空姐 | F | A | HK accent flight attendant | Regional, entertainment |\n| `Chinese (Mandarin)_Cute_Spirit` | 憨憨萌兽 | N | C | Cute cartoon spirit | Animations, children's |\n| `Chinese (Mandarin)_Humorous_Elder` | 搞笑大爷 | M | E | Humorous old man | Comedy, entertainment |\n| `Chinese (Mandarin)_Kind-hearted_Elder` | 花甲奶奶 | F | E | Kind elderly lady | Stories, warm |\n| `Chinese (Mandarin)_Kind-hearted_Antie` | 热心大婶 | F | E | Kind-hearted auntie | Warm, friendly |\n\n### Chinese Cantonese (粤语)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `Cantonese_ProfessionalHost（F)` | 专业女主持 | F | A | Professional host | Broadcasts, hosting |\n| `Cantonese_GentleLady` | 温柔女声 | F | A | Gentle female | Soft, warm |\n| `Cantonese_ProfessionalHost（M)` | 专业男主持 | M | A | Professional host | Broadcasts, hosting |\n| `Cantonese_PlayfulMan` | 活泼男声 | M | A | Playful male | Entertainment, casual |\n| `Cantonese_CuteGirl` | 可爱女孩 | F | C | Cute girl | Children's, animations |\n| `Cantonese_KindWoman` | 善良女声 | F | A | Kind female | Warm, friendly |\n\n### English\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `English_Trustworthy_Man` | Trustworthy Man | M | A | Reliable, sincere | Business, narration |\n| `English_Graceful_Lady` | Graceful Lady | F | A | Elegant, refined | Formal, professional |\n| `English_Aussie_Bloke` | Aussie Bloke | M | A | Casual Australian | Casual, entertainment |\n| `English_Whispering_girl` | Whispering Girl | F | Y | Soft whisper | Romance, intimate |\n| `English_Diligent_Man` | Diligent Man | M | A | Earnest, hardworking | Motivational, educational |\n| `English_Gentle-voiced_man` | Gentle-voiced Man | M | E | Soft-spoken, kind | Calm, supportive |\n| `English_Sweet_Girl` | Sweet Girl | F | C | Sweet, innocent | Children's, friendly |\n| `Charming_Lady` | Charming Lady | F | A | Elegant, sophisticated | Professional, romance |\n| `Attractive_Girl` | Attractive Girl | F | Y | Engaging female | Entertainment, marketing |\n| `Serene_Woman` | Serene Woman | F | A | Calm, peaceful | Meditation, relaxation |\n| `Santa_Claus` | Santa Claus | M | E | Festive, jolly | Holiday, children's |\n| `Charming_Santa` | Charming Santa | M | E | Smooth, charismatic | Holiday, entertainment |\n| `Grinch` | Grinch | M | A | Whiny, mischievous | Comedy, holiday |\n| `Rudolph` | Rudolph | N | C | Cute, nasal reindeer | Children's, holiday |\n| `Arnold` | Arnold | M | A | Deep, robotic | Sci-fi, action |\n| `Cute_Elf` | Cute Elf | N | C | Playful, tiny elf | Fantasy, children's |\n\n### Japanese (日本語)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `Japanese_IntellectualSenior` | Intellectual Senior | M | E | Wise, knowledgeable | Narration, educational |\n| `Japanese_DecisivePrincess` | Decisive Princess | F | A | Confident, royal | Animation, games |\n| `Japanese_LoyalKnight` | Loyal Knight | M | A | Brave, faithful | Fantasy, games |\n| `Japanese_DominantMan` | Dominant Man | M | A | Powerful, commanding | Action, leadership |\n| `Japanese_SeriousCommander` | Serious Commander | M | A | Stern, authoritative | Military, games |\n| `Japanese_ColdQueen` | Cold Queen | F | A | Distant, majestic | Drama, fantasy |\n| `Japanese_DependableWoman` | Dependable Woman | F | A | Reliable, supportive | Guidance |\n| `Japanese_GentleButler` | Gentle Butler | M | A | Polite, refined | Comedy, animation |\n| `Japanese_KindLady` | Kind Lady | F | A | Warm, gentle | Comforting |\n| `Japanese_CalmLady` | Calm Lady | F | A | Composed, serene | Meditation, relaxation |\n| `Japanese_OptimisticYouth` | Optimistic Youth | M | Y | Cheerful, positive | Youth, motivation |\n| `Japanese_GenerousIzakayaOwner` | Generous Izakaya Owner | M | A | Friendly, welcoming | Casual, comedy |\n| `Japanese_SportyStudent` | Sporty Student | M | Y | Energetic, athletic | Sports, youth |\n| `Japanese_InnocentBoy` | Innocent Boy | M | C | Pure, naive | Children's |\n| `Japanese_GracefulMaiden` | Graceful Maiden | F | Y | Elegant, gentle | Romance, drama |\n\n### Korean (한국어)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `Korean_SweetGirl` | Sweet Girl | F | C | Sweet, adorable | Children's, romance |\n| `Korean_CheerfulBoyfriend` | Cheerful Boyfriend | M | Y | Energetic, loving | Romance, dating |\n| `Korean_EnchantingSister` | Enchanting Sister | F | A | Charming, captivating | Family, drama |\n| `Korean_ShyGirl` | Shy Girl | F | Y | Timid, reserved | Comedy, romance |\n| `Korean_ReliableSister` | Reliable Sister | F | A | Trustworthy, dependable | Guidance |\n| `Korean_StrictBoss` | Strict Boss | M | A | Authoritative, demanding | Business, drama |\n| `Korean_SassyGirl` | Sassy Girl | F | Y | Bold, witty | Comedy, entertainment |\n| `Korean_ChildhoodFriendGirl` | Childhood Friend Girl | F | Y | Familiar, friendly | Romance, nostalgia |\n| `Korean_PlayboyCharmer` | Playboy Charmer | M | A | Smooth, flirtatious | Romance, entertainment |\n| `Korean_ElegantPrincess` | Elegant Princess | F | A | Graceful, royal | Animation, fantasy |\n| `Korean_BraveFemaleWarrior` | Brave Female Warrior | F | A | Courageous | Action, fantasy |\n| `Korean_BraveYouth` | Brave Youth | M | Y | Heroic | Action, youth |\n| `Korean_CalmLady` | Calm Lady | F | A | Composed, serene | Meditation, relaxation |\n| `Korean_EnthusiasticTeen` | Enthusiastic Teen | M | Y | Excited, energetic | Youth |\n| `Korean_SoothingLady` | Soothing Lady | F | A | Calming, comforting | Relaxation |\n| `Korean_IntellectualSenior` | Intellectual Senior | M | E | Wise, knowledgeable | Educational, narration |\n| `Korean_LonelyWarrior` | Lonely Warrior | M | A | Solitary, melancholic | Drama, fantasy |\n| `Korean_MatureLady` | Mature Lady | F | A | Sophisticated | Professional, drama |\n| `Korean_InnocentBoy` | Innocent Boy | M | C | Pure, naive | Children's |\n| `Korean_CharmingSister` | Charming Sister | F | A | Attractive, delightful | Family, romance |\n| `Korean_AthleticStudent` | Athletic Student | M | Y | Sporty, energetic | Sports, youth |\n| `Korean_BraveAdventurer` | Brave Adventurer | M | A | Courageous explorer | Adventure, fantasy |\n| `Korean_CalmGentleman` | Calm Gentleman | M | A | Composed, refined | Formal, professional |\n| `Korean_WiseElf` | Wise Elf | M | E | Ancient, mystical | Fantasy, narration |\n| `Korean_CheerfulCoolJunior` | Cheerful Cool Junior | M | Y | Popular, friendly | Youth, entertainment |\n| `Korean_DecisiveQueen` | Decisive Queen | F | A | Commanding | Drama, fantasy |\n| `Korean_ColdYoungMan` | Cold Young Man | M | Y | Distant, aloof | Drama, romance |\n| `Korean_MysteriousGirl` | Mysterious Girl | F | Y | Enigmatic, secretive | Mystery, drama |\n| `Korean_QuirkyGirl` | Quirky Girl | F | Y | Eccentric, unique | Comedy |\n| `Korean_ConsiderateSenior` | Considerate Senior | M | E | Thoughtful, caring | Warm, supportive |\n| `Korean_CheerfulLittleSister` | Cheerful Little Sister | F | C | Playful, adorable | Family, comedy |\n| `Korean_DominantMan` | Dominant Man | M | A | Powerful, commanding | Leadership, action |\n| `Korean_AirheadedGirl` | Airheaded Girl | F | Y | Bubbly, spacey | Comedy |\n| `Korean_ReliableYouth` | Reliable Youth | M | Y | Trustworthy, dependable | Supportive |\n| `Korean_FriendlyBigSister` | Friendly Big Sister | F | A | Warm, protective | Family, support |\n| `Korean_GentleBoss` | Gentle Boss | M | A | Kind, understanding | Business |\n| `Korean_ColdGirl` | Cold Girl | F | Y | Aloof, distant | Drama, romance |\n| `Korean_HaughtyLady` | Haughty Lady | F | A | Arrogant, proud | Drama, comedy |\n| `Korean_CharmingElderSister` | Charming Elder Sister | F | A | Graceful | Romance, family |\n| `Korean_IntellectualMan` | Intellectual Man | M | A | Smart, knowledgeable | Educational |\n| `Korean_CaringWoman` | Caring Woman | F | A | Nurturing | Supportive, warm |\n| `Korean_WiseTeacher` | Wise Teacher | M | E | Experienced | Educational |\n| `Korean_ConfidentBoss` | Confident Boss | M | A | Self-assured, capable | Business, leadership |\n| `Korean_AthleticGirl` | Athletic Girl | F | Y | Sporty, energetic | Sports, fitness |\n| `Korean_PossessiveMan` | Possessive Man | M | A | Intense, protective | Romance, drama |\n| `Korean_GentleWoman` | Gentle Woman | F | A | Soft-spoken, kind | Calm |\n| `Korean_CockyGuy` | Cocky Guy | M | Y | Confident, arrogant | Comedy |\n| `Korean_ThoughtfulWoman` | Thoughtful Woman | F | A | Reflective, caring | Drama |\n| `Korean_OptimisticYouth` | Optimistic Youth | M | Y | Positive, hopeful | Motivation |\n\n### Spanish (Español)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `Spanish_Narrator` | Narrator | M | A | Professional narrator | Documentaries |\n| `Spanish_CaptivatingStoryteller` | Captivating Storyteller | M | A | Engaging narrator | Audiobooks |\n| `Spanish_WiseScholar` | Wise Scholar | M | A | Knowledgeable | Educational |\n| `Spanish_SereneWoman` | Serene Woman | F | A | Calm, peaceful | Relaxation |\n| `Spanish_MaturePartner` | Mature Partner | M | A | Sophisticated | Romance, drama |\n| `Spanish_ConfidentWoman` | Confident Woman | F | A | Self-assured | Professional |\n| `Spanish_DeterminedManager` | Determined Manager | M | A | Ambitious, driven | Business |\n| `Spanish_BossyLeader` | Bossy Leader | M | A | Commanding | Leadership |\n| `Spanish_ReservedYoungMan` | Reserved Young Man | M | Y | Quiet, introverted | Drama |\n| `Spanish_ThoughtfulMan` | Thoughtful Man | M | A | Reflective | Educational |\n| `Spanish_RationalMan` | Rational Man | M | A | Logical, analytical | Business |\n| `Spanish_Deep-tonedMan` | Deep-toned Man | M | A | Deep, resonant | Commanding |\n| `Spanish_Jovialman` | Jovial Man | M | A | Cheerful, friendly | Entertainment |\n| `Spanish_Steadymentor` | Steady Mentor | M | A | Reliable mentor | Guidance |\n| `Spanish_ReliableMan` | Reliable Man | M | A | Trustworthy | Professional |\n| `Spanish_RomanticHusband` | Romantic Husband | M | A | Loving, romantic | Romance |\n| `Spanish_Comedian` | Comedian | M | A | Humorous | Comedy |\n| `Spanish_Debator` | Debator | M | A | Persuasive | Debate |\n| `Spanish_ToughBoss` | Tough Boss | M | A | Harsh, demanding | Business, drama |\n| `Spanish_AngryMan` | Angry Man | M | A | Frustrated | Drama, comedy |\n| `Spanish_PowerfulSoldier` | Powerful Soldier | M | A | Strong, brave | Action, military |\n| `Spanish_PassionateWarrior` | Passionate Warrior | M | A | Fierce, dedicated | Action, fantasy |\n| `Spanish_PowerfulVeteran` | Powerful Veteran | M | A | Experienced | Military |\n| `Spanish_SensibleManager` | Sensible Manager | M | A | Practical | Business |\n| `Spanish_Kind-heartedGirl` | Kind-hearted Girl | F | C | Warm, compassionate | Children's |\n| `Spanish_SophisticatedLady` | Sophisticated Lady | F | A | Elegant, refined | Formal |\n| `Spanish_FrankLady` | Frank Lady | F | A | Direct, honest | Comedy |\n| `Spanish_Fussyhostess` | Fussy Hostess | F | A | Demanding | Comedy, drama |\n| `Spanish_Wiselady` | Wise Lady | F | E | Experienced, wise | Guidance |\n| `Spanish_ThoughtfulLady` | Thoughtful Lady | F | A | Considerate | Advice |\n| `Spanish_AssertiveQueen` | Assertive Queen | F | A | Commanding | Drama, fantasy |\n| `Spanish_CaringGirlfriend` | Caring Girlfriend | F | Y | Nurturing | Romance |\n| `Spanish_ChattyGirl` | Chatty Girl | F | Y | Talkative, sociable | Comedy |\n| `Spanish_CompellingGirl` | Compelling Girl | F | Y | Persuasive | Marketing |\n| `Spanish_WhimsicalGirl` | Whimsical Girl | F | C | Playful, imaginative | Children's |\n| `Spanish_Intonategirl` | Intonate Girl | F | Y | Musical, melodic | Singing |\n| `Spanish_SincereTeen` | Sincere Teen | M | Y | Honest, genuine | Youth |\n| `Spanish_Strong-WilledBoy` | Strong-willed Boy | M | Y | Determined | Youth, motivation |\n| `Spanish_EnergeticBoy` | Energetic Boy | M | C | Active, lively | Youth, sports |\n| `Spanish_StrictBoss` | Strict Boss | M | A | Strict | Business |\n| `Spanish_HumorousElder` | Humorous Elder | M | E | Funny | Comedy |\n| `Spanish_SereneElder` | Serene Elder | M | E | Calm, peaceful | Meditation |\n| `Spanish_SantaClaus` | Santa Claus | M | E | Festive | Holiday |\n| `Spanish_Rudolph` | Rudolph | N | C | Reindeer | Holiday |\n| `Spanish_Arnold` | Arnold | M | A | Robotic | Sci-fi |\n| `Spanish_Ghost` | Ghost | N | A | Spooky | Horror |\n| `Spanish_AnimeCharacter` | Anime Character | N | Y | Anime-style | Animation |\n\n### Portuguese (Português)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `Portuguese_Narrator` | Narrator | M | A | Professional narrator | Documentaries |\n| `Portuguese_CaptivatingStoryteller` | Captivating Storyteller | M | A | Engaging narrator | Audiobooks |\n| `Portuguese_WiseScholar` | Wise Scholar | M | A | Knowledgeable | Educational |\n| `Portuguese_Deep-VoicedGentleman` | Deep-voiced Gentleman | M | A | Deep, rich | Commanding |\n| `Portuguese_ReservedYoungMan` | Reserved Young Man | M | Y | Quiet, introverted | Drama |\n| `Portuguese_ThoughtfulMan` | Thoughtful Man | M | A | Reflective | Educational |\n| `Portuguese_RationalMan` | Rational Man | M | A | Logical | Business |\n| `Portuguese_Jovialman` | Jovial Man | M | A | Cheerful | Entertainment |\n| `Portuguese_Steadymentor` | Steady Mentor | M | A | Reliable mentor | Guidance |\n| `Portuguese_ReliableMan` | Reliable Man | M | A | Trustworthy | Professional |\n| `Portuguese_RomanticHusband` | Romantic Husband | M | A | Loving | Romance |\n| `Portuguese_Comedian` | Comedian | M | A | Humorous | Comedy |\n| `Portuguese_Debator` | Debator | M | A | Persuasive | Debate |\n| `Portuguese_ToughBoss` | Tough Boss | M | A | Demanding | Business |\n| `Portuguese_StrictBoss` | Strict Boss | M | A | Strict | Business |\n| `Portuguese_AngryMan` | Angry Man | M | A | Frustrated | Drama |\n| `Portuguese_Godfather` | Godfather | M | A | Authoritative | Drama |\n| `Portuguese_PowerfulSoldier` | Powerful Soldier | M | A | Strong, brave | Action |\n| `Portuguese_PowerfulVeteran` | Powerful Veteran | M | A | Experienced | Military |\n| `Portuguese_SensibleManager` | Sensible Manager | M | A | Practical | Business |\n| `Portuguese_DeterminedManager` | Determined Manager | M | A | Driven | Business |\n| `Portuguese_BossyLeader` | Bossy Leader | M | A | Commanding | Leadership |\n| `Portuguese_CalmLeader` | Calm Leader | M | A | Composed, steady | Leadership |\n| `Portuguese_FascinatingBoy` | Fascinating Boy | M | Y | Charming | Romance |\n| `Portuguese_Strong-WilledBoy` | Strong-willed Boy | M | Y | Determined | Youth |\n| `Portuguese_EnergeticBoy` | Energetic Boy | M | C | Active, lively | Youth |\n| `Portuguese_FragileBoy` | Fragile Boy | M | Y | Sensitive | Drama |\n| `Portuguese_MaturePartner` | Mature Partner | M | A | Sophisticated | Romance |\n| `Portuguese_HumorousElder` | Humorous Elder | M | E | Funny | Comedy |\n| `Portuguese_SereneElder` | Serene Elder | M | E | Calm | Meditation |\n| `Portuguese_ConfidentWoman` | Confident Woman | F | A | Self-assured | Professional |\n| `Portuguese_SereneWoman` | Serene Woman | F | A | Calm, peaceful | Relaxation |\n| `Portuguese_SentimentalLady` | Sentimental Lady | F | A | Emotional | Drama, romance |\n| `Portuguese_Wiselady` | Wise Lady | F | E | Wise | Guidance |\n| `Portuguese_GorgeousLady` | Gorgeous Lady | F | A | Beautiful | Romance |\n| `Portuguese_LovelyLady` | Lovely Lady | F | A | Sweet, endearing | Warm |\n| `Portuguese_Pompouslady` | Pompous Lady | F | A | Self-important | Comedy |\n| `Portuguese_CharmingQueen` | Charming Queen | F | A | Elegant | Drama, fantasy |\n| `Portuguese_AssertiveQueen` | Assertive Queen | F | A | Commanding | Drama, fantasy |\n| `Portuguese_CharmingLady` | Charming Lady | F | A | Sophisticated | Professional |\n| `Portuguese_InspiringLady` | Inspiring Lady | F | A | Motivating | Motivation |\n| `Portuguese_StressedLady` | Stressed Lady | F | A | Anxious | Comedy |\n| `Portuguese_FrankLady` | Frank Lady | F | A | Direct, honest | Comedy |\n| `Portuguese_Fussyhostess` | Fussy Hostess | F | A | Demanding | Comedy |\n| `Portuguese_ThoughtfulLady` | Thoughtful Lady | F | A | Considerate | Advice |\n| `Portuguese_GentleTeacher` | Gentle Teacher | F | A | Kind, patient | Educational |\n| `Portuguese_Kind-heartedGirl` | Kind-hearted Girl | F | C | Warm | Children's |\n| `Portuguese_SweetGirl` | Sweet Girl | F | Y | Sweet, adorable | Romance |\n| `Portuguese_AttractiveGirl` | Attractive Girl | F | Y | Charming | Entertainment |\n| `Portuguese_PlayfulGirl` | Playful Girl | F | Y | Fun-loving | Comedy |\n| `Portuguese_SmartYoungGirl` | Smart Young Girl | F | Y | Intelligent | Educational |\n| `Portuguese_UpsetGirl` | Upset Girl | F | Y | Distressed | Drama |\n| `Portuguese_ElegantGirl` | Elegant Girl | F | Y | Graceful | Formal |\n| `Portuguese_CompellingGirl` | Compelling Girl | F | Y | Persuasive | Marketing |\n| `Portuguese_WhimsicalGirl` | Whimsical Girl | F | C | Playful | Children's |\n| `Portuguese_ChattyGirl` | Chatty Girl | F | Y | Talkative | Comedy |\n| `Portuguese_NaughtySchoolgirl` | Naughty Schoolgirl | F | Y | Mischievous | Comedy |\n| `Portuguese_SadTeen` | Sad Teen | F | Y | Melancholic | Drama |\n| `Portuguese_CaringGirlfriend` | Caring Girlfriend | F | Y | Nurturing | Romance |\n| `Portuguese_FriendlyNeighbor` | Friendly Neighbor | F | A | Warm, helpful | Community |\n| `Portuguese_Dramatist` | Dramatist | M | A | Theatrical | Drama |\n| `Portuguese_TheatricalActor` | Theatrical Actor | M | A | Dramatic | Entertainment |\n| `Portuguese_Conscientiousinstructor` | Conscientious Instructor | M | A | Diligent | Training |\n| `Portuguese_PlayfulSpirit` | Playful Spirit | N | C | Cheerful spirit | Fantasy |\n| `Portuguese_SantaClaus` | Santa Claus | M | E | Festive | Holiday |\n| `Portuguese_Rudolph` | Rudolph | N | C | Reindeer | Holiday |\n| `Portuguese_Arnold` | Arnold | M | A | Robotic | Sci-fi |\n| `Portuguese_CharmingSanta` | Charming Santa | M | E | Charismatic | Holiday |\n| `Portuguese_Grinch` | Grinch | M | A | Mischievous | Comedy |\n| `Portuguese_Ghost` | Ghost | N | A | Spooky | Horror |\n| `Portuguese_GrimReaper` | Grim Reaper | N | A | Dark, ominous | Horror |\n\n### French (Français)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `French_Male_Speech_New` | Level-Headed Man | M | A | Calm, reasonable | Professional |\n| `French_Female_News Anchor` | Patient Female Presenter | F | A | Clear, patient | News |\n| `French_CasualMan` | Casual Man | M | A | Relaxed, informal | Casual |\n| `French_MovieLeadFemale` | Movie Lead Female | F | A | Dramatic, expressive | Drama |\n| `French_FemaleAnchor` | Female Anchor | F | A | Professional anchor | News |\n\n### Indonesian (Bahasa Indonesia)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `Indonesian_SweetGirl` | Sweet Girl | F | C | Sweet, adorable | Children's |\n| `Indonesian_ReservedYoungMan` | Reserved Young Man | M | Y | Quiet, introverted | Drama |\n| `Indonesian_CharmingGirl` | Charming Girl | F | Y | Attractive | Romance |\n| `Indonesian_CalmWoman` | Calm Woman | F | A | Composed, peaceful | Relaxation |\n| `Indonesian_ConfidentWoman` | Confident Woman | F | A | Self-assured | Professional |\n| `Indonesian_CaringMan` | Caring Man | M | A | Nurturing | Family |\n| `Indonesian_BossyLeader` | Bossy Leader | M | A | Commanding | Leadership |\n| `Indonesian_DeterminedBoy` | Determined Boy | M | Y | Ambitious | Youth |\n| `Indonesian_GentleGirl` | Gentle Girl | F | Y | Soft-spoken | Calm |\n\n### German (Deutsch)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `German_FriendlyMan` | Friendly Man | M | A | Warm, approachable | Casual |\n| `German_SweetLady` | Sweet Lady | F | A | Pleasant, kind | Warm |\n| `German_PlayfulMan` | Playful Man | M | A | Fun-loving | Comedy |\n\n### Russian (Русский)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `Russian_HandsomeChildhoodFriend` | Handsome Childhood Friend | M | Y | Charming | Romance |\n| `Russian_BrightHeroine` | Bright Queen | F | A | Lively, strong | Drama |\n| `Russian_AmbitiousWoman` | Ambitious Woman | F | A | Driven | Professional |\n| `Russian_ReliableMan` | Reliable Man | M | A | Trustworthy | Professional |\n| `Russian_CrazyQueen` | Crazy Girl | F | Y | Wild, unpredictable | Comedy |\n| `Russian_PessimisticGirl` | Pessimistic Girl | F | Y | Gloomy | Comedy |\n| `Russian_AttractiveGuy` | Attractive Guy | M | A | Charming | Romance |\n| `Russian_Bad-temperedBoy` | Bad-tempered Boy | M | Y | Irritable, grumpy | Comedy |\n\n### Italian (Italiano)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `Italian_BraveHeroine` | Brave Heroine | F | A | Courageous | Action |\n| `Italian_Narrator` | Narrator | M | A | Professional narrator | Storytelling |\n| `Italian_WanderingSorcerer` | Wandering Sorcerer | M | A | Mysterious | Fantasy |\n| `Italian_DiligentLeader` | Diligent Leader | M | A | Hardworking | Leadership |\n\n### Arabic (العربية)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `Arabic_CalmWoman` | Calm Woman | F | A | Composed | Relaxation |\n| `Arabic_FriendlyGuy` | Friendly Guy | M | A | Warm | Casual |\n\n### Turkish (Türkçe)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `Turkish_CalmWoman` | Calm Woman | F | A | Composed | Relaxation |\n| `Turkish_Trustworthyman` | Trustworthy Man | M | A | Reliable | Professional |\n\n### Ukrainian (Українська)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `Ukrainian_CalmWoman` | Calm Woman | F | A | Composed | Relaxation |\n| `Ukrainian_WiseScholar` | Wise Scholar | M | A | Knowledgeable | Educational |\n\n### Dutch (Nederlands)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `Dutch_kindhearted_girl` | Kind-hearted Girl | F | C | Warm | Children's |\n| `Dutch_bossy_leader` | Bossy Leader | M | A | Commanding | Leadership |\n\n### Vietnamese (Tiếng Việt)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `Vietnamese_kindhearted_girl` | Kind-hearted Girl | F | C | Warm | Children's |\n\n### Thai (ภาษาไทย)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `Thai_male_1_sample8` | Serene Man | M | A | Calm, peaceful | Relaxation |\n| `Thai_male_2_sample2` | Friendly Man | M | A | Warm | Casual |\n| `Thai_female_1_sample1` | Confident Woman | F | A | Self-assured | Professional |\n| `Thai_female_2_sample2` | Energetic Woman | F | A | Active, lively | Motivation |\n\n### Polish (Polski)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `Polish_male_1_sample4` | Male Narrator | M | A | Professional | Narration |\n| `Polish_male_2_sample3` | Male Anchor | M | A | Professional | News |\n| `Polish_female_1_sample1` | Calm Woman | F | A | Composed | Relaxation |\n| `Polish_female_2_sample3` | Casual Woman | F | A | Relaxed | Casual |\n\n### Romanian (Română)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `Romanian_male_1_sample2` | Reliable Man | M | A | Trustworthy | Professional |\n| `Romanian_male_2_sample1` | Energetic Youth | M | Y | Active, lively | Youth |\n| `Romanian_female_1_sample4` | Optimistic Youth | F | Y | Positive | Motivation |\n| `Romanian_female_2_sample1` | Gentle Woman | F | A | Soft-spoken | Calm |\n\n### Greek (Ελληνικά)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `greek_male_1a_v1` | Thoughtful Mentor | M | A | Reflective, wise | Guidance |\n| `Greek_female_1_sample1` | Gentle Lady | F | A | Soft-spoken | Calm |\n| `Greek_female_2_sample3` | Girl Next Door | F | Y | Friendly | Casual |\n\n### Czech (Čeština)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `czech_male_1_v1` | Assured Presenter | M | A | Confident | Presentations |\n| `czech_female_5_v7` | Steadfast Narrator | F | A | Reliable | Storytelling |\n| `czech_female_2_v2` | Elegant Lady | F | A | Graceful | Formal |\n\n### Finnish (Suomi)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `finnish_male_3_v1` | Upbeat Man | M | A | Cheerful | Motivation |\n| `finnish_male_1_v2` | Friendly Boy | M | Y | Warm | Children's |\n| `finnish_female_4_v1` | Assertive Woman | F | A | Confident | Professional |\n\n### Hindi (हिन्दी)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `hindi_male_1_v2` | Trustworthy Advisor | M | A | Reliable, wise | Guidance |\n| `hindi_female_2_v1` | Tranquil Woman | F | A | Calm, peaceful | Meditation |\n| `hindi_female_1_v2` | News Anchor | F | A | Professional | News |\n\n---\n\n## Voice Parameters\n\n### VoiceSetting\n\n```python\nfrom scripts.tts.utils import VoiceSetting\n\nvoice = VoiceSetting(\n    voice_id=\"male-qn-qingse\",\n    speed=1.0,       # 0.5–2.0 (default 1.0)\n    volume=1.0,      # 0.1–10.0 (default 1.0)\n    pitch=0,         # -12 to +12 (default 0)\n    emotion=\"\",      # Leave empty for speech-2.8 auto-matching (recommended)\n)\n```\n\n### Speed\n\n| Value | Effect |\n|-------|--------|\n| 0.75 | Slower, deliberate (news, tutorials) |\n| 1.0 | Normal pace |\n| 1.25 | Slightly faster (energetic) |\n| 1.5+ | Fast (time-sensitive) |\n\n### Emotion\n\n| Value | Description | Model Support |\n|-------|-------------|---------------|\n| *(empty)* | Auto-match from text | speech-2.8 (recommended) |\n| `happy` | Cheerful, upbeat | All |\n| `sad` | Melancholic, somber | All |\n| `angry` | Intense, frustrated | All |\n| `fearful` | Anxious, nervous | All |\n| `disgusted` | Repulsed | All |\n| `surprised` | Astonished | All |\n| `calm` | Neutral tone | All |\n| `fluent` | Natural, lively | speech-2.6 only |\n| `whisper` | Soft, gentle | speech-2.6 only |\n\n---\n\n## Custom Voices\n\n### Voice Cloning\n\nCreate custom voices from audio samples:\n- Source: 10s–5min, mp3/wav/m4a, ≤20MB, clear single speaker\n- Best: 30–60s of clean speech with varied intonation\n\n### Voice Design\n\nGenerate voices from text descriptions:\n- Include: gender, age, vocal characteristics, tone, use case\n- Example: \"A warm, grandmotherly voice with gentle pacing, perfect for bedtime stories\"\n\nCustom voices expire after 7 days if not used with TTS. List all voices: `python scripts/tts/generate_voice.py list-voices`\n\nFile v1.0.1:references/video-api.md\n\n# MiniMax Video Generation API Documentation\n\n## API Endpoints\n\n| Endpoint | Method | Description |\n|----------|--------|-------------|\n| `/v1/video_generation` | POST | Create video generation task (all 4 modes) |\n| `/v1/query/video_generation` | GET | Query task status |\n| `/v1/files/retrieve` | GET | Get video download URL |\n| `/v1/video_template_generation` | POST | Create template-based video task |\n| `/v1/query/video_template_generation` | GET | Query template task status |\n\n**Base URL:** `https://api.minimaxi.com`\n**Auth:** `Authorization: Bearer {MINIMAX_API_KEY}`\n\n---\n\n## Video Generation Models\n\n### Text-to-Video (T2V) Models\n| Model | Resolution | Duration | Notes |\n|-------|-----------|----------|-------|\n| MiniMax-Hailuo-2.3 | 768P (default), 1080P | 6s (1080P), 6/10s (768P) | Recommended, latest |\n| MiniMax-Hailuo-2.3-Fast | 768P (default), 1080P | 6s (1080P), 6/10s (768P) | Fast variant |\n| MiniMax-Hailuo-02 | 512P, 768P (default), 1080P | 6s (1080P), 6/10s (512P/768P) | Previous gen |\n| T2V-01-Director | 720P | 6s | Director control |\n| T2V-01 | 720P | 6s | Base model |\n\n### Image-to-Video (I2V) Models\n| Model | Resolution | Duration | Notes |\n|-------|-----------|----------|-------|\n| MiniMax-Hailuo-2.3 | 768P, 1080P | 6s | Recommended |\n| MiniMax-Hailuo-2.3-Fast | 768P, 1080P | 6s | Fast variant |\n| MiniMax-Hailuo-02 | 512P, 768P, 1080P | 6/10s | Previous gen |\n| I2V-01-Director | 720P | 6s | Director control |\n| I2V-01-live | 720P | 6s | Live photo style |\n| I2V-01 | 720P | 6s | Base model |\n\n### Start-End Frame Model\n| Model | Notes |\n|-------|-------|\n| MiniMax-Hailuo-02 | Only model supporting start-end frame |\n\n### Subject Reference Model\n| Model | Notes |\n|-------|-------|\n| S2V-01 | Face consistency across video |\n\n---\n\n## Request Parameters\n\n### Common Parameters (All Modes)\n| Parameter | Type | Required | Default | Description |\n|-----------|------|----------|---------|-------------|\n| model | string | Yes | - | Model name |\n| prompt | string | Depends | - | Video description, max 2000 chars |\n| duration | int | No | 6 | Video length in seconds |\n| resolution | string | No | 768P/720P | Video resolution |\n| prompt_optimizer | bool | No | true | Auto-optimize prompt |\n| fast_pretreatment | bool | No | false | Shorten optimizer duration |\n| callback_url | string | No | - | Webhook URL |\n| aigc_watermark | bool | No | false | Add watermark |\n\n### Image-to-Video Parameters\n| Parameter | Type | Required | Description |\n|-----------|------|----------|-------------|\n| first_frame_image | string | Yes | Starting frame (URL or base64 data URL) |\n\n**Image requirements:** JPG/JPEG/PNG/WebP, < 20MB, short side > 300px, aspect ratio 2:5–5:2.\n\n### Start-End Frame Parameters\n| Parameter | Type | Required | Description |\n|-----------|------|----------|-------------|\n| first_frame_image | string | Yes | Starting frame |\n| last_frame_image | string | Yes | Ending frame |\n\n### Subject Reference Parameters\n| Parameter | Type | Required | Description |\n|-----------|------|----------|-------------|\n| subject_reference | array | Yes | Array of subject objects |\n\nEach object has `type` and `image` (array of image URLs):\n```json\n[{ \"type\": \"character\", \"image\": [\"<image_url>\"] }]\n```\n\n---\n\n## Camera Instructions\n\nSupported in `[指令]` syntax for Hailuo-2.3, Hailuo-02, and Director models:\n\n| Category | Instructions |\n|----------|-------------|\n| Pan | `[左移]`, `[右移]` |\n| Rotation | `[左摇]`, `[右摇]` |\n| Push/Pull | `[推进]`, `[拉远]` |\n| Elevation | `[上升]`, `[下降]` |\n| Tilt | `[上摇]`, `[下摇]` |\n| Zoom | `[变焦推近]`, `[变焦拉远]` |\n| Other | `[晃动]`, `[跟随]`, `[固定]` |\n\nCombine for simultaneous: `[左摇,上升]` (max 3). Sequential: `...[推进], then ...[拉远]`\n\n---\n\n## Response\n\n**Query status:** `Preparing`, `Queueing`, `Processing`, `Success`, `Fail`\n\n**Error codes:** 0 (success), 1002 (rate limited), 1004 (auth failed), 1008 (insufficient balance), 1026 (sensitive content), 2013 (invalid params), 2049 (invalid API key)\n\n---\n\n## Video Templates\n\n| Template | ID | Input | Description |\n|----------|-----|-------|-------------|\n| Diving | 392753057216684038 | Image | Diving motion |\n| Rings | 393881433990066176 | Image | Gymnastics rings |\n| Survival | 393769180141805569 | Image + Text | Outdoor survival |\n| Labubu | 394246956137422856 | Image | Labubu character |\n| McDonald's Delivery | 393879757702918151 | Image | Pet courier |\n| Tibetan Portrait | 393766210733957121 | Image | Cultural portrait |\n| Female Model Ads | 393866076583718914 | Image | Female fashion |\n| Male Model Ads | 393876118804459526 | Image | Male fashion |\n| Winter Romance | 393857704283172856 | Image | Snowy portrait |\n| Four Seasons | 398574688191234048 | Image | Seasonal portrait |\n| Helpless Moments | 394125185182695432 | Text only | Comedic animation |\n\nFile v1.0.1:references/video-prompt-guide.md\n\n# Video Prompt Writing Guide\n\n## Prompt Structure\n\n### Basic Formula\n**Main subject + Scene/Space + Movement/Change**\n\nExamples:\n- \"A puppy runs toward the camera in a sunny park\"\n- \"A woman walks in the rain holding an umbrella on a city street\"\n- \"A stream flows through a green valley with morning mist\"\n\n### Professional Formula\n**Main subject + Scene + Movement + Camera motion + Aesthetic atmosphere**\n\nExamples:\n- \"A couple sits on a park bench, warm golden hour lighting, [固定] framing, intimate and romantic atmosphere\"\n- \"A young man in a suit eats noodles at a street stall, [拉远] revealing the busy night market, warm tones, cinematic\"\n- \"A dancer performs contemporary dance in an empty studio, [跟随] smooth tracking, dramatic side lighting\"\n\n---\n\n## Key Principles\n\n1. **More precise language → more accurate video**\n2. **Richer description → better generation quality**\n3. **Keep prompts focused on 5-6 seconds of action** — do not describe too many events\n4. **Combine shot types with mood descriptors** for professional output\n\n---\n\n## Camera Instructions Usage\n\n### Simultaneous Camera Movement\nPlace multiple instructions in one bracket:\n- `[左摇,上升]` — pan left while rising\n- `[推进,下摇]` — push in while tilting down\n\n### Sequential Camera Movement\nPlace instructions at different points in the prompt:\n- \"The camera starts with [推进] toward the face, then [拉远] to reveal the full scene\"\n\n---\n\n## Style-Specific Prompt Tips\n\n### Realistic / Cinematic Style\n- Mention lighting: \"golden hour\", \"overcast sky\", \"dramatic side lighting\"\n- Color grading: \"warm tones\", \"cool desaturated palette\", \"high contrast\"\n- Texture: \"rain droplets on glass\", \"dust particles in sunlight\"\n- Cinematic terms: \"shallow depth of field\", \"anamorphic lens flare\"\n\n### Animation Style\n- Substyle: \"2D anime\", \"3D Pixar-style\", \"watercolor animation\", \"stop-motion\"\n- Character design: \"big expressive eyes\", \"chibi proportions\"\n- Effects: \"sparkle particles\", \"speed lines\", \"dramatic wind effects\"\n\n### Product / Commercial Style\n- Product details: \"smooth surface\", \"premium materials\", \"elegant design\"\n- Studio lighting: \"soft box lighting\", \"rim light\", \"gradient background\"\n- Motion: \"slow rotation\", \"smooth reveal\", \"gentle float\"\n\n### Fantasy / Sci-Fi Style\n- World elements: \"floating islands\", \"neon cyberpunk city\", \"enchanted forest\"\n- VFX: \"magic particles\", \"holographic displays\", \"energy beams\"\n- Scale: \"vast landscape\", \"towering structures\", \"infinite horizon\"\n\n### Nature / Documentary Style\n- Terminology: \"macro shot\", \"time-lapse\", \"wildlife behavior\"\n- Phenomena: \"morning dew\", \"sunset colors\", \"storm clouds\"\n- Precision: \"slow motion at 240fps\", \"underwater perspective\"\n\n---\n\n## Image-to-Video Prompt Tips\n\nFocus on **movement and change** since the image establishes the visual:\n- Image of still lake → \"Gentle ripples spread across the water surface, a breeze rustles the trees, [固定] fixed camera, peaceful\"\n- Image of portrait → \"The person slowly smiles and turns their head, natural blinking, [推进] subtle push in, warm lighting\"\n\n---\n\n## Prompt Building Checklist\n\n1. **Subject**: Appearance, clothing, color, expression, posture\n2. **Action**: 1-2 key temporal actions (\"first...then...\")\n3. **Scene**: Setting with foreground + background + atmosphere\n4. **Camera**: `[运镜指令]` for precise control\n5. **Aesthetic**: Lighting, color, texture, cinematic quality\n\n## Common Mistakes\n\n1. Too many events for 6-second videos\n2. Conflicting camera instructions\n3. Vague descriptions\n4. Static descriptions without motion\n5. Missing aesthetic layer\n6. Overlong prompts (keep under 200 words)\n\nArchive v1.0.0: 26 files, 78890 bytes\n\nFiles: references/music-api.md (1883b), references/tts-guide.md (3164b), references/tts-voice-catalog.md (33087b), references/video-api.md (4813b), references/video-prompt-guide.md (3678b), requirements.txt (84b), scripts/check_environment.py (3215b), scripts/env_loader.py (1454b), scripts/media_tools.py (25589b), scripts/music/generate_music.py (9623b), scripts/music/utils_audio.py (540b), scripts/tts/async_tts.py (7775b), scripts/tts/audio_processing.py (34050b), scripts/tts/generate_voice.py (14654b), scripts/tts/segment_tts.py (16120b), scripts/tts/sync_tts.py (5589b), scripts/tts/utils.py (8877b), scripts/tts/voice_clone.py (5928b), scripts/tts/voice_design.py (3913b), scripts/tts/voice_management.py (3322b), scripts/video/add_bgm.py (9146b), scripts/video/generate_long_video.py (18284b), scripts/video/generate_template_video.py (7073b), scripts/video/generate_video.py (10192b), SKILL.md (23321b), _meta.json (137b)\n\nFile v1.0.0:SKILL.md\n\n---\nname: minimax-multimodal-toolkit\ndescription: MiniMax multimodal model skill — use MiniMax  Multi-Modal models for speech, music, and video. Create voice, music, and video with MiniMax AI: TTS (text-to-speech, voice cloning, voice design, multi-segment), music (songs, instrumentals), video (text-to-video, image-to-video, start-end frame, subject reference, templates, long-form multi-scene), and media processing (convert, concat, trim, extract). Use when the user mentions MiniMax, multimodal generation, or wants speech/music/video AI, MiniMax APIs, or FFmpeg workflows alongside MiniMax outputs.\n---\n\n# MiniMax Multi-Modal Toolkit\n\nGenerate voice, music, and video content via MiniMax APIs — the unified entry for **MiniMax multimodal** use cases (audio + music + video). Includes voice cloning & voice design for custom voices, and FFmpeg-based media tools for audio/video format conversion, concatenation, trimming, and extraction.\n\n## Output Directory\n\n**All generated files MUST be saved to `minimax-output/` under the AGENT'S current working directory (NOT the skill directory).** Every script call MUST include an explicit `--output` / `-o` argument pointing to this location. Never omit the output argument or rely on script defaults.\n\n**Rules:**\n1. Before running any script, ensure `minimax-output/` exists in the agent's working directory (create if needed: `mkdir -p minimax-output`)\n2. Always use absolute or relative paths from the agent's working directory: `--output minimax-output/video.mp4`\n3. **Never** `cd` into the skill directory to run scripts — run from the agent's working directory using the full script path\n4. Intermediate/temp files (segment audio, video segments, extracted frames) are automatically placed in `minimax-output/tmp/`. They can be cleaned up when no longer needed: `rm -rf minimax-output/tmp`\n\n## Prerequisites\n\n```bash\npip install -r requirements.txt   # requests, websockets, ffmpeg-python\nbrew install ffmpeg                # macOS\npython scripts/check_environment.py\n```\n\n### API Key Configuration\n\nSet the `MINIMAX_API_KEY` environment variable before running any script:\n\n```bash\nexport MINIMAX_API_KEY=\"your-api-key-here\"\n```\n\nThe key starts with `sk-api-` or `sk-cp-`, obtainable from https://platform.minimaxi.com\n\n**IMPORTANT — When API Key is missing:**\nBefore running any script, check if `MINIMAX_API_KEY` is set in the environment. If it is NOT configured:\n1. Ask the user to provide their MiniMax API key\n2. Instruct and help user to set it via `export MINIMAX_API_KEY=\"sk-...\"` in their terminal or add it to their shell profile (`~/.zshrc` / `~/.bashrc`) for persistence\n\n## Key Capabilities\n\n| Capability | Description | Entry point |\n|------------|-------------|-------------|\n| TTS | Text-to-speech synthesis with multiple voices and emotions | `scripts/tts/generate_voice.py` |\n| Voice Cloning | Clone a voice from an audio sample (10s–5min) | `scripts/tts/generate_voice.py clone` |\n| Voice Design | Create a custom voice from a text description | `scripts/tts/generate_voice.py design` |\n| Music Generation | Generate songs with lyrics or instrumental tracks | `scripts/music/generate_music.py` |\n| Video Generation | Text-to-video, image-to-video, subject reference, templates | `scripts/video/generate_video.py` |\n| Long Video | Multi-scene chained video with crossfade transitions | `scripts/video/generate_long_video.py` |\n| Media Tools | Audio/video format conversion, concatenation, trimming, extraction | `scripts/media_tools.py` |\n\n## TTS (Text-to-Speech)\n\nEntry point: `scripts/tts/generate_voice.py`\n\n### IMPORTANT: Single voice vs Multi-segment — Choose the right approach\n\n| User intent | Approach |\n|-------------|----------|\n| Single voice / no multi-character need | `tts` command — generate the entire text in one call |\n| Multiple characters / narrator + dialogue | `generate` command with segments.json |\n\n**Default behavior:** When the user simply asks to generate speech/voice and does NOT mention multiple voices or characters, use the `tts` command directly with a single appropriate voice. Do NOT split into segments or use the multi-segment pipeline — just pass the full text to `tts` in one call.\n\nOnly use multi-segment `generate` when:\n- The user explicitly needs multiple voices/characters\n- The text requires narrator + character dialogue separation\n- The text exceeds **10,000 characters** (API limit per request) — in this case, split into segments with the same voice\n\n### Single-voice generation (DEFAULT)\n\n```bash\npython scripts/tts/generate_voice.py tts \"Hello world\" -o minimax-output/hello.mp3\npython scripts/tts/generate_voice.py tts \"你好世界\" -v female-shaonv -o minimax-output/hello_cn.mp3\n```\n\n### Multi-segment generation (multi-voice / audiobook / podcast)\n\n**Complete workflow — follow ALL steps in order:**\n\n1. **Write segments.json** — split text into segments with voice assignments (see format and rules below)\n2. **Run `generate` command** — this reads segments.json, generates audio for EACH segment via TTS API, then merges them into a single output file with crossfade\n\n```bash\n# Step 1: Write segments.json to minimax-output/\n# (use the Write tool to create minimax-output/segments.json)\n\n# Step 2: Generate audio from segments.json — this is the CRITICAL step\n# It generates each segment individually and merges them into one file\npython scripts/tts/generate_voice.py generate minimax-output/segments.json \\\n  -o minimax-output/output.mp3 --crossfade 200\n```\n\n**Do NOT skip Step 2.** Writing segments.json alone does nothing — you MUST run the `generate` command to actually produce audio.\n\n### Voice management\n\n```bash\n# List all available voices\npython scripts/tts/generate_voice.py list-voices\n\n# Voice cloning (from audio sample, 10s–5min)\npython scripts/tts/generate_voice.py clone sample.mp3 --voice-id my-voice\n\n# Voice design (from text description)\npython scripts/tts/generate_voice.py design \"A warm female narrator voice\" --voice-id narrator\n```\n\n### Audio processing\n\n```bash\npython scripts/tts/generate_voice.py merge part1.mp3 part2.mp3 -o minimax-output/combined.mp3\npython scripts/tts/generate_voice.py convert input.wav -o minimax-output/output.mp3\n```\n\n### TTS Models\n\n| Model | Notes |\n|-------|-------|\n| speech-2.8-hd | Recommended, auto emotion matching |\n| speech-2.8-turbo | Faster variant |\n| speech-2.6-hd | Previous gen, manual emotion |\n| speech-2.6-turbo | Previous gen, faster |\n\n### segments.json Format\n\nDefault crossfade between segments: **200ms** (`--crossfade 200`).\n\n```json\n[\n  { \"text\": \"Hello!\", \"voice_id\": \"female-shaonv\", \"emotion\": \"\" },\n  { \"text\": \"Welcome.\", \"voice_id\": \"male-qn-qingse\", \"emotion\": \"happy\" }\n]\n```\n\nLeave `emotion` empty for speech-2.8 models (auto-matched from text).\n\n### IMPORTANT: Multi-Segment Script Generation Rules (Audiobooks, Podcasts, etc.)\n\nWhen generating segments.json for audiobooks, podcasts, or any multi-character narration, you MUST split narration text from character dialogue into separate segments with distinct voices.\n\n**Rule: Narration and dialogue are ALWAYS separate segments.**\n\nA sentence like `\"Tom said: The weather is great today!\"` must be split into two segments:\n- Segment 1 (narrator voice): `\"Tom said:\"`\n- Segment 2 (character voice): `\"The weather is great today!\"`\n\n**Example — Audiobook with narrator + 2 characters:**\n\n```json\n[\n  { \"text\": \"Morning sunlight streamed into the classroom as students filed in one by one.\", \"voice_id\": \"narrator-voice\", \"emotion\": \"\" },\n  { \"text\": \"Tom smiled and turned to Lisa:\", \"voice_id\": \"narrator-voice\", \"emotion\": \"\" },\n  { \"text\": \"The weather is amazing today! Let's go to the park after school!\", \"voice_id\": \"tom-voice\", \"emotion\": \"happy\" },\n  { \"text\": \"Lisa thought for a moment, then replied:\", \"voice_id\": \"narrator-voice\", \"emotion\": \"\" },\n  { \"text\": \"Sure, but I need to drop off my backpack at home first.\", \"voice_id\": \"lisa-voice\", \"emotion\": \"\" },\n  { \"text\": \"They exchanged a smile and went back to listening to the lecture.\", \"voice_id\": \"narrator-voice\", \"emotion\": \"\" }\n]\n```\n\n**Key principles:**\n1. **Narrator** uses a consistent neutral narrator voice throughout\n2. **Each character** has a dedicated voice_id, maintained consistently across all their dialogue\n3. **Split at dialogue boundaries** — `\"He said:\"` is narrator, the quoted content is the character\n4. **Do NOT merge** narrator text and character speech into a single segment\n5. For characters without pre-existing voice_ids, use voice cloning or voice design to create them first, then reference the created voice_id in segments\n\n## Music Generation\n\nEntry point: `scripts/music/generate_music.py`\n\n### IMPORTANT: Instrumental vs Lyrics — When to use which\n\n| Scenario | Mode | Action |\n|----------|------|--------|\n| BGM for video / voice / podcast | Instrumental (default) | Use `--instrumental` directly, do NOT ask user |\n| User explicitly asks to \"create music\" / \"make a song\" | Ask user first | Ask whether they want instrumental or with lyrics |\n\n**When adding background music to video or voice content**, always default to instrumental mode (`--instrumental`). Do not ask the user — BGM should never have vocals competing with the main content.\n\n**When the user explicitly asks to create/generate music as the primary task**, ask them whether they want:\n- Instrumental (pure music, no vocals)\n- With lyrics (song with vocals — user provides or you help write lyrics)\n\n```bash\n# Instrumental (for BGM or when user chooses instrumental)\npython scripts/music/generate_music.py \\\n  --instrumental \\\n  --prompt \"ambient electronic, atmospheric\" \\\n  --output minimax-output/ambient.mp3 --download\n\n# Song with lyrics (when user chooses vocal music)\npython scripts/music/generate_music.py \\\n  --lyrics \"[verse]\\nHello world\\n[chorus]\\nLa la la\" \\\n  --prompt \"indie folk, melancholic\" \\\n  --output minimax-output/song.mp3 --download\n\n# With style fields\npython scripts/music/generate_music.py \\\n  --lyrics \"[verse]\\nLyrics here\" \\\n  --genre \"pop\" --mood \"upbeat\" --tempo \"fast\" \\\n  --output minimax-output/pop_track.mp3 --download\n```\n\n### Music Models\n\n| Model | Notes |\n|-------|-------|\n| music-2.5+ | Recommended, supports `--instrumental` |\n| music-2.5 | Previous version |\n\n## Video Generation\n\n### IMPORTANT: Single vs Multi-Segment — Choose the right script\n\n| User intent | Script to use |\n|-------------|---------------|\n| Default / no special request | `scripts/video/generate_video.py` (single segment, **10s, 768P**) |\n| User explicitly asks for \"long video\", \"multi-scene\", \"story\", or duration > 10s | `scripts/video/generate_long_video.py` (multi-segment) |\n\n**Default behavior:** Always use single-segment `generate_video.py` with **duration 10s and resolution 768P** unless the user explicitly asks for a long video, multi-scene video, or specifies a total duration exceeding 10 seconds. Do NOT automatically split into multiple segments — a single 10s video is the standard output. Only use `generate_long_video.py` when the user clearly needs multi-scene or longer content.\n\nEntry point (single video): `scripts/video/generate_video.py`\nEntry point (long/multi-scene): `scripts/video/generate_long_video.py`\n\n### Video Model Constraints (MUST follow)\n\n**Duration limits by model and resolution:**\n\n| Model | 720P | 768P | 1080P |\n|-------|------|------|-------|\n| MiniMax-Hailuo-2.3 | - | 6s or **10s** | 6s only |\n| MiniMax-Hailuo-2.3-Fast | - | 6s or **10s** | 6s only |\n| MiniMax-Hailuo-02 | - | 6s or **10s** | 6s only |\n| T2V-01 / T2V-01-Director | 6s only | - | - |\n| I2V-01 / I2V-01-Director / I2V-01-live | 6s only | - | - |\n| S2V-01 (ref) | 6s only | - | - |\n\n**Resolution options by model and duration:**\n\n| Model | 6s | 10s |\n|-------|-----|-----|\n| MiniMax-Hailuo-2.3 | 768P (default), 1080P | 768P only |\n| MiniMax-Hailuo-2.3-Fast | 768P (default), 1080P | 768P only |\n| MiniMax-Hailuo-02 | 512P, 768P (default), 1080P | 512P, 768P (default) |\n| Other models | 720P (default) | Not supported |\n\n**Key rules:**\n- **Default: 10s + 768P** (best balance of length and quality for MiniMax-Hailuo-2.3)\n- 1080P only supports 6s duration — if user requests 1080P, set `--duration 6`\n- 10s duration only works with 768P (or 512P on Hailuo-02) — never combine 10s + 1080P\n- Older models (T2V-01, I2V-01, S2V-01) only support 6s at 720P\n\n### IMPORTANT: Prompt Optimization (MUST follow before generating any video)\n\nBefore calling any video generation script, you MUST optimize the user's prompt by reading and applying `references/video-prompt-guide.md`. Never pass the user's raw description directly as `--prompt`.\n\n**Optimization steps:**\n\n1. **Apply the Professional Formula**: `Main subject + Scene + Movement + Camera motion + Aesthetic atmosphere`\n   - BAD: `\"A puppy in a park\"`\n   - GOOD: `\"A golden retriever puppy runs toward the camera on a sun-dappled grass path in a park, [跟随] smooth tracking shot, warm golden hour lighting, shallow depth of field, joyful atmosphere\"`\n\n2. **Add camera instructions** using `[指令]` syntax: `[推进]`, `[拉远]`, `[跟随]`, `[固定]`, `[左摇]`, etc.\n\n3. **Include aesthetic details**: lighting (golden hour, dramatic side lighting), color grading (warm tones, cinematic), texture (dust particles, rain droplets), atmosphere (intimate, epic, peaceful)\n\n4. **Keep to 1-2 key actions** for 6-10 second videos — do not overcrowd with events\n\n5. **For i2v mode** (image-to-video): Focus prompt on **movement and change only**, since the image already establishes the visual. Do NOT re-describe what's in the image.\n   - BAD: `\"A lake with mountains\"` (just repeating the image)\n   - GOOD: `\"Gentle ripples spread across the water surface, a breeze rustles the distant trees, [固定] fixed camera, soft morning light, peaceful and serene\"`\n\n6. **For multi-segment long videos**: Each segment's prompt must be self-contained and optimized individually. The i2v segments (segment 2+) should describe motion/change relative to the previous segment's ending frame.\n\n```bash\n# Text-to-video (default: 10s, 768P)\npython scripts/video/generate_video.py \\\n  --mode t2v \\\n  --prompt \"A golden retriever puppy bounds toward the camera on a sunlit grass path, [跟随] tracking shot, warm golden hour, shallow depth of field, joyful\" \\\n  --output minimax-output/puppy.mp4\n\n# Text-to-video with 1080P (must use --duration 6)\npython scripts/video/generate_video.py \\\n  --mode t2v \\\n  --prompt \"A golden retriever puppy bounds toward the camera\" \\\n  --duration 6 --resolution 1080P \\\n  --output minimax-output/puppy_hd.mp4\n\n# Image-to-video (prompt focuses on MOTION, not image content)\npython scripts/video/generate_video.py \\\n  --mode i2v \\\n  --prompt \"The petals begin to sway gently in the breeze, soft light shifts across the surface, [固定] fixed framing, dreamy pastel tones\" \\\n  --first-frame photo.jpg \\\n  --output minimax-output/animated.mp4\n\n# Start-end frame interpolation (sef mode uses MiniMax-Hailuo-02)\npython scripts/video/generate_video.py \\\n  --mode sef \\\n  --first-frame start.jpg --last-frame end.jpg \\\n  --output minimax-output/transition.mp4\n\n# Subject reference (face consistency, ref mode uses S2V-01, 6s only)\npython scripts/video/generate_video.py \\\n  --mode ref \\\n  --prompt \"A young woman in a white dress walks slowly through a sunlit garden, [跟随] smooth tracking, warm natural lighting, cinematic depth of field\" \\\n  --subject-image face.jpg \\\n  --duration 6 \\\n  --output minimax-output/person.mp4\n```\n\n### Long-form Video (Multi-scene)\n\nMulti-scene long videos chain segments together: the first segment generates via text-to-video (t2v), then each subsequent segment uses the last frame of the previous segment as its first frame (i2v). Segments are joined with crossfade transitions for smooth continuity. Default is 10 seconds per segment.\n\n**Workflow:**\n1. Segment 1: t2v — generated purely from the optimized text prompt\n2. Segment 2+: i2v — the previous segment's last frame becomes `first_frame_image`, prompt describes **motion and change from that ending state**\n3. All segments are concatenated with 0.5s crossfade transitions to eliminate jump cuts\n4. Optional: AI-generated background music is overlaid\n\n**Prompt rules for each segment:**\n- Each segment prompt MUST be independently optimized using the Professional Formula\n- Segment 1 (t2v): Full scene description with subject, scene, camera, atmosphere\n- Segment 2+ (i2v): Focus on **what changes and moves** from the previous ending frame. Do NOT repeat the visual description — the first frame already provides it\n- Maintain visual consistency: keep lighting, color grading, and style keywords consistent across segments\n- Each segment covers only 10 seconds of action — keep it focused\n\n```bash\n# Example: 3-segment story with optimized per-segment prompts (default: 10s/segment, 768P)\npython scripts/video/generate_long_video.py \\\n  --scenes \\\n    \"A lone astronaut stands on a red desert planet surface, wind blowing dust particles, [推进] slow push in toward the visor, dramatic rim lighting, cinematic sci-fi atmosphere\" \\\n    \"The astronaut turns and begins walking toward a distant glowing structure on the horizon, dust swirling around boots, [跟随] tracking from behind, vast desolate landscape, golden light from the structure\" \\\n    \"The astronaut reaches the structure entrance, a massive doorway pulses with blue energy, [推进] slow push in toward the doorway, light reflects off the visor, awe-inspiring epic scale\" \\\n  --music-prompt \"cinematic orchestral ambient, slow build, sci-fi atmosphere\" \\\n  --output minimax-output/long_video.mp4\n\n# With custom settings\npython scripts/video/generate_long_video.py \\\n  --scenes \"Scene 1 prompt\" \"Scene 2 prompt\" \\\n  --segment-duration 10 \\\n  --resolution 768P \\\n  --crossfade 0.5 \\\n  --music-prompt \"calm ambient background music\" \\\n  --output minimax-output/long_video.mp4\n```\n\n### Add Background Music\n\n```bash\npython scripts/video/add_bgm.py \\\n  --video input.mp4 \\\n  --generate-bgm --instrumental \\\n  --music-prompt \"soft piano background\" \\\n  --bgm-volume 0.3 \\\n  --output minimax-output/output_with_bgm.mp4\n```\n\n### Template Video\n\n```bash\npython scripts/video/generate_template_video.py \\\n  --template-id 392753057216684038 \\\n  --media photo.jpg \\\n  --output minimax-output/template_output.mp4\n```\n\n### Video Models\n\n| Mode | Default Model | Default Duration | Default Resolution | Notes |\n|------|--------------|-----------------|-------------------|-------|\n| t2v | MiniMax-Hailuo-2.3 | 10s | 768P | Latest text-to-video |\n| i2v | MiniMax-Hailuo-2.3 | 10s | 768P | Latest image-to-video |\n| sef | MiniMax-Hailuo-02 | 6s | 768P | Start-end frame |\n| ref | S2V-01 | 6s | 720P | Subject reference, 6s only |\n\n## Media Tools (Audio/Video Processing)\n\nEntry point: `scripts/media_tools.py`\n\nStandalone FFmpeg-based utilities for format conversion, concatenation, extraction, trimming, and audio overlay. Use these when the user needs to process existing media files without generating new content via MiniMax API.\n\n### Video Format Conversion\n\n```bash\n# Convert between formats (mp4, mov, webm, mkv, avi, ts, flv)\npython scripts/media_tools.py convert-video input.webm -o output.mp4\npython scripts/media_tools.py convert-video input.mp4 -o output.mov\n\n# With quality / resolution / fps options\npython scripts/media_tools.py convert-video input.mp4 -o output.mp4 \\\n  --crf 18 --preset medium --resolution 1920x1080 --fps 30\n```\n\n### Audio Format Conversion\n\n```bash\n# Convert between formats (mp3, wav, flac, ogg, aac, m4a, opus, wma)\npython scripts/media_tools.py convert-audio input.wav -o output.mp3\npython scripts/media_tools.py convert-audio input.mp3 -o output.flac \\\n  --bitrate 320k --sample-rate 48000 --channels 2\n```\n\n### Video Concatenation\n\n```bash\n# Concatenate with crossfade transition (default 0.5s)\npython scripts/media_tools.py concat-video seg1.mp4 seg2.mp4 seg3.mp4 -o merged.mp4\n\n# Hard cut (no crossfade)\npython scripts/media_tools.py concat-video seg1.mp4 seg2.mp4 -o merged.mp4 --crossfade 0\n```\n\n### Audio Concatenation\n\n```bash\n# Simple concatenation\npython scripts/media_tools.py concat-audio part1.mp3 part2.mp3 -o combined.mp3\n\n# With crossfade\npython scripts/media_tools.py concat-audio part1.mp3 part2.mp3 -o combined.mp3 --crossfade 1\n```\n\n### Extract Audio from Video\n\n```bash\n# Extract as mp3\npython scripts/media_tools.py extract-audio video.mp4 -o audio.mp3\n\n# Extract as wav with higher bitrate\npython scripts/media_tools.py extract-audio video.mp4 -o audio.wav --bitrate 320k\n```\n\n### Video Trimming\n\n```bash\n# Trim by start/end time (seconds)\npython scripts/media_tools.py trim-video input.mp4 -o clip.mp4 --start 5 --end 15\n\n# Trim by start + duration\npython scripts/media_tools.py trim-video input.mp4 -o clip.mp4 --start 10 --duration 8\n```\n\n### Add Audio to Video (Overlay / Replace)\n\n```bash\n# Mix audio with existing video audio\npython scripts/media_tools.py add-audio --video video.mp4 --audio bgm.mp3 -o output.mp4 \\\n  --volume 0.3 --fade-in 2 --fade-out 3\n\n# Replace original audio entirely\npython scripts/media_tools.py add-audio --video video.mp4 --audio narration.mp3 -o output.mp4 \\\n  --replace\n```\n\n### Media File Info\n\n```bash\npython scripts/media_tools.py probe input.mp4\n```\n\n## Script Architecture\n\n```\nscripts/\n├── check_environment.py          # Env verification\n├── env_loader.py                # Optional .env fallback loader\n├── media_tools.py               # Audio/video conversion, concat, trim, extract\n├── tts/\n│   ├── generate_voice.py         # CLI entry point\n│   ├── sync_tts.py               # Synchronous TTS API\n│   ├── async_tts.py              # Async (task-based) TTS API\n│   ├── segment_tts.py            # Multi-segment pipeline\n│   ├── audio_processing.py       # FFmpeg audio processing\n│   ├── voice_clone.py            # Voice cloning API\n│   ├── voice_design.py           # Voice design API\n│   ├── voice_management.py       # Voice CRUD operations\n│   └── utils.py                  # Shared: API config, VoiceSetting, AudioSetting\n├── music/\n│   ├── generate_music.py         # Music generation CLI\n│   └── utils_audio.py            # Audio format utilities\n└── video/\n    ├── generate_video.py         # Video generation CLI (4 modes)\n    ├── generate_long_video.py    # Multi-scene long video\n    ├── generate_template_video.py # Template-based video\n    └── add_bgm.py               # Background music overlay\n```\n\n## References\n\nRead these for detailed API parameters, voice catalogs, and prompt engineering:\n\n- [tts-guide.md](references/tts-guide.md) — TTS setup, voice management, audio processing, segment format, troubleshooting\n- [tts-voice-catalog.md](references/tts-voice-catalog.md) — Full voice catalog with IDs, descriptions, and parameter reference\n- [music-api.md](references/music-api.md) — Music generation API: endpoints, parameters, response format\n- [video-api.md](references/video-api.md) — Video API: endpoints, models, parameters, camera instructions, templates\n- [video-prompt-guide.md](references/video-prompt-guide.md) — Video prompt engineering: formulas, styles, image-to-video tips\n\nFile v1.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn70p6rdfg6k3598at7gm6r5fd82h5rs\",\n  \"slug\": \"minimax-multimodal\",\n  \"version\": \"1.0.0\",\n  \"publishedAt\": 1773998365577\n}\n\nFile v1.0.0:references/music-api.md\n\n# MiniMax Music Generation API (music-2.5 / music-2.5+)\n\nSource: https://platform.minimaxi.com/docs/api-reference/music-generation\n\n## Endpoint\n\n`POST https://api.minimaxi.com/v1/music_generation`\n\n## Auth\n\n`Authorization: Bearer <MINIMAX_API_KEY>`\n\n## Request (JSON)\n\nRequired:\n- `model`: string — `music-2.5+` (recommended) or `music-2.5`\n- `lyrics`: string (1–3500 chars) — required for non-instrumental. Use `\\n` for line breaks. Structure tags: `[Verse]`, `[Chorus]`, `[Bridge]`, `[Intro]`, `[Outro]`, etc.\n\nOptional:\n- `prompt`: string (0–2000 chars) — style description. Required for `music-2.5+` instrumental mode.\n- `is_instrumental`: boolean — `music-2.5+` only. When true, generates pure music (no vocals); `lyrics` is not required.\n- `lyrics_optimizer`: boolean — auto-generate lyrics from prompt when lyrics is empty.\n- `stream`: boolean (default `false`)\n- `output_format`: `hex` (default) or `url`. URL valid for 24 hours.\n- `aigc_watermark`: boolean — top-level field, non-streaming only.\n- `audio_setting`:\n  - `sample_rate`: 16000, 24000, 32000, 44100\n  - `bitrate`: 32000, 64000, 128000, 256000\n  - `format`: mp3, wav, pcm\n\n## Example\n\n```json\n{\n  \"model\": \"music-2.5+\",\n  \"prompt\": \"indie folk, melancholic, introspective\",\n  \"lyrics\": \"[verse]\\n...\\n[chorus]\\n...\",\n  \"is_instrumental\": false,\n  \"aigc_watermark\": false,\n  \"audio_setting\": {\n    \"sample_rate\": 44100,\n    \"bitrate\": 256000,\n    \"format\": \"mp3\"\n  }\n}\n```\n\n## Response\n\n- `data.audio`: hex string or URL depending on `output_format`\n- `data.status`: 1 (generating), 2 (complete)\n- `extra_info`: duration, sample_rate, channels, bitrate, size\n- `base_resp.status_code`: 0 on success\n\n## Notes\n\n- `music-2.5+` with `is_instrumental=true`: `prompt` is required, `lyrics` can be omitted.\n- `music-2.5`: `prompt` is optional but recommended.\n- `stream=true` only supports `hex` output.\n\nFile v1.0.0:references/tts-guide.md\n\n# TTS Guide\n\n## Setup\n\n```bash\ncd skills/MiniMaxStudio\npip install -r requirements.txt\nbrew install ffmpeg   # macOS (or: sudo apt install ffmpeg)\nexport MINIMAX_API_KEY=\"your-api-key\"   # sk-api-xxx or sk-cp-xxx\npython scripts/check_environment.py\n```\n\n## Quick Test\n\n```bash\npython scripts/tts/generate_voice.py tts \"Hello, this is a test.\" -o test.mp3\n```\n\n## Voice Management\n\nList available voices:\n\n```bash\npython scripts/tts/generate_voice.py list-voices\n```\n\n### Voice Cloning\n\nCreate a custom voice from an audio sample:\n\n```bash\npython scripts/tts/generate_voice.py clone audio.mp3 --voice-id my-custom-voice\n\n# With preview\npython scripts/tts/generate_voice.py clone audio.mp3 --voice-id my-voice --preview \"Test text\" --preview-output preview.mp3\n```\n\nRequirements: 10s–5min duration, ≤20MB, mp3/wav/m4a format.\n\n### Voice Design\n\nDesign a voice from a text description:\n\n```bash\npython scripts/tts/generate_voice.py design \"A warm, gentle female voice\" --voice-id designed-voice\n```\n\nCustom voices expire after 7 days if not used with TTS.\n\n## Audio Processing\n\n### Merge\n\n```bash\npython scripts/tts/generate_voice.py merge file1.mp3 file2.mp3 -o combined.mp3\npython scripts/tts/generate_voice.py merge a.mp3 b.mp3 -o merged.mp3 --crossfade 300\n```\n\n### Convert\n\n```bash\npython scripts/tts/generate_voice.py convert input.wav -o output.mp3\npython scripts/tts/generate_voice.py convert input.wav -o output.mp3 --format mp3 --bitrate 192k --sample-rate 32000\n```\n\nFFmpeg required. Supported formats: mp3, wav, flac, ogg, m4a, aac, wma, opus, pcm.\n\n## Segment-Based TTS\n\nFor multi-voice, multi-emotion workflows using a `segments.json` file:\n\n```bash\n# Validate\npython scripts/tts/generate_voice.py validate segments.json --verbose\n\n# Generate\npython scripts/tts/generate_voice.py generate segments.json -o output.mp3 --crossfade 200\n```\n\n### segments.json Format\n\n```json\n[\n  { \"text\": \"Hello!\", \"voice_id\": \"female-shaonv\", \"emotion\": \"\" },\n  { \"text\": \"How are you?\", \"voice_id\": \"male-qn-qingse\", \"emotion\": \"happy\" }\n]\n```\n\n- `text` (required): Text to synthesize\n- `voice_id` (required): Voice ID\n- `emotion` (optional): For speech-2.8 models, leave empty for auto-matching. Valid values: happy, sad, angry, fearful, disgusted, surprised, calm, fluent, whisper\n\n## Troubleshooting\n\n| Error | Solution |\n|-------|----------|\n| `MINIMAX_API_KEY is required` | `export MINIMAX_API_KEY=\"key\"` |\n| `FFmpeg not installed` | `brew install ffmpeg` |\n| `Voice not found` | `python scripts/tts/generate_voice.py list-voices` |\n| `401 Unauthorized` | Check API key validity |\n| `429 Too Many Requests` | Add delays between requests |\n\n## API Details\n\n- **Endpoint**: `POST /v1/t2a_v2`\n- **Base URL**: `https://api.minimaxi.com`\n- **Auth**: `Authorization: Bearer {MINIMAX_API_KEY}`\n- **Models**: speech-2.8-hd (recommended), speech-2.8-turbo, speech-2.6-hd, speech-2.6-turbo, speech-02-hd, speech-02-turbo, speech-01-hd, speech-01-turbo\n- **Text limit**: 10,000 characters per request\n- **Pause marker**: `<#x#>` where x is seconds (0.01–99.99)\n- **Interjection tags** (speech-2.8 only): `(laughs)`, `(chuckle)`, `(coughs)`, `(sighs)`, `(breath)`, etc.\n\nFile v1.0.0:references/tts-voice-catalog.md\n\n# TTS Voice Catalog\n\n## Contents\n\n- [Voice Selection Guide](#voice-selection-guide)\n- [System Voices by Language](#system-voices-by-language)\n- [Voice Parameters](#voice-parameters)\n- [Custom Voices](#custom-voices)\n\n---\n\n## Voice Selection Guide\n\n### Decision Flow\n\n```\nContent type?\n├── Narration / Audiobook  → audiobook_female_1, audiobook_male_1\n├── News / Announcement    → Chinese (Mandarin)_News_Anchor, Chinese (Mandarin)_Male_Announcer\n├── Documentary            → doc_commentary\n└── Other                  → Select by: Gender → Age → Language → Personality\n```\n\n### Recommended Professional Voices\n\n| Scenario | Recommended | Characteristics |\n|----------|-------------|-----------------|\n| Narration / Audiobook | `audiobook_female_1`, `audiobook_male_1` | Clear articulation, good pacing, sustained performance |\n| News / Announcement | `Chinese (Mandarin)_News_Anchor`, `Chinese (Mandarin)_Male_Announcer` | Authoritative, professional pacing |\n| Documentary | `doc_commentary` | Professional, clear, consistent |\n\n### Selection Priority\n\n1. **Gender** (mandatory match) — male voices for male characters, female for female\n2. **Age** — Children / Youth / Adult / Elderly\n3. **Language** (must match content language)\n4. **Personality/tone** — choose best fit from matching candidates\n\n---\n\n## System Voices by Language\n\nGender: M = Male, F = Female, N = Neutral/Character\nAge: C = Child, Y = Youth, A = Adult, E = Elder\n\n### Chinese Mandarin (普通话)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `male-qn-qingse` | 青涩青年 | M | Y | Youthful, inexperienced | Campus, coming-of-age |\n| `male-qn-badao` | 霸道青年 | M | Y | Arrogant, dominant | Drama, romance |\n| `male-qn-daxuesheng` | 青年大学生 | M | Y | University student | Campus, educational |\n| `male-qn-jingying` | 精英青年 | M | A | Elite, ambitious | Business, professional |\n| `female-shaonv` | 少女 | F | Y | Young maiden | Romance, youth |\n| `female-yujie` | 御姐 | F | A | Mature, elegant | Romance, professional |\n| `female-chengshu` | 成熟女性 | F | A | Mature, reliable | Sophisticated, news |\n| `female-tianmei` | 甜美女性 | F | A | Sweet, pleasant | Soft, gentle |\n| `clever_boy` | 聪明男童 | M | C | Smart, witty | Children's, educational |\n| `cute_boy` | 可爱男童 | M | C | Adorable | Kids, animations |\n| `lovely_girl` | 萌萌女童 | F | C | Cute, sweet | Children's stories |\n| `cartoon_pig` | 卡通猪小琪 | N | C | Cartoon character | Animations, comedy |\n| `bingjiao_didi` | 病娇弟弟 | M | Y | Tsundere brother | Romance, character |\n| `junlang_nanyou` | 俊朗男友 | M | Y | Handsome boyfriend | Romance, dating |\n| `chunzhen_xuedi` | 纯真学弟 | M | Y | Innocent junior | Campus, youth |\n| `lengdan_xiongzhang` | 冷淡学长 | M | Y | Cool senior | Campus, romance |\n| `badao_shaoye` | 霸道少爷 | M | A | Arrogant young master | Drama, character |\n| `tianxin_xiaoling` | 甜心小玲 | F | Y | Sweet Xiao Ling | Character, animations |\n| `qiaopi_mengmei` | 俏皮萌妹 | F | Y | Playful cute girl | Comedy, light-hearted |\n| `wumei_yujie` | 妩媚御姐 | F | A | Charming mature woman | Romance, mature |\n| `diadia_xuemei` | 嗲嗲学妹 | F | Y | Flirty junior girl | Romance, dating |\n| `danya_xuejie` | 淡雅学姐 | F | Y | Elegant senior girl | Campus, romance |\n| `Arrogant_Miss` | 嚣张小姐 | F | A | Arrogant young lady | Drama, character |\n| `Robot_Armor` | 机械战甲 | N | A | Robotic armor | Sci-fi, games |\n| `audiobook_male_1` | 有声书男1 | M | A | Warm, engaging narrator | Audiobooks, stories |\n| `audiobook_female_1` | 有声书女1 | F | A | Gentle, expressive narrator | Audiobooks, stories |\n| `doc_commentary` | 纪录片解说 | M | A | Professional narrator | Documentary |\n| `Chinese (Mandarin)_News_Anchor` | 新闻女声 | F | A | News anchor | News, broadcasts |\n| `Chinese (Mandarin)_Male_Announcer` | 播报男声 | M | A | Male announcer | Announcements |\n| `Chinese (Mandarin)_Radio_Host` | 电台男主播 | M | A | Radio host | Podcasts, radio |\n| `Chinese (Mandarin)_Reliable_Executive` | 沉稳高管 | M | A | Reliable executive | Corporate, business |\n| `Chinese (Mandarin)_Gentleman` | 温润男声 | M | A | Gentle, refined | Narration, storytelling |\n| `Chinese (Mandarin)_Unrestrained_Young_Man` | 不羁青年 | M | Y | Unrestrained, casual | Entertainment |\n| `Chinese (Mandarin)_Southern_Young_Man` | 南方小哥 | M | Y | Southern accent | Regional, casual |\n| `Chinese (Mandarin)_Gentle_Youth` | 温润青年 | M | Y | Gentle young man | Narration, calm |\n| `Chinese (Mandarin)_Sincere_Adult` | 真诚青年 | M | Y | Sincere, genuine | Honest, genuine |\n| `Chinese (Mandarin)_Straightforward_Boy` | 率真弟弟 | M | Y | Frank, direct | Casual, direct |\n| `Chinese (Mandarin)_Pure-hearted_Boy` | 清澈邻家弟弟 | M | Y | Pure-hearted neighbor | Innocent, wholesome |\n| `Chinese (Mandarin)_Stubborn_Friend` | 嘴硬竹马 | M | Y | Stubborn childhood friend | Drama, character |\n| `Chinese (Mandarin)_Lyrical_Voice` | 抒情男声 | M | A | Lyrical, singing | Music, singing |\n| `Chinese (Mandarin)_Mature_Woman` | 傲娇御姐 | F | A | Tsundere mature woman | Romance, character |\n| `Chinese (Mandarin)_Sweet_Lady` | 甜美女声 | F | A | Sweet lady | Soft, gentle |\n| `Chinese (Mandarin)_Warm_Bestie` | 温暖闺蜜 | F | A | Warm bestie | Friendly, supportive |\n| `Chinese (Mandarin)_Warm_Girl` | 温暖少女 | F | Y | Warm young girl | Friendly, supportive |\n| `Chinese (Mandarin)_Soft_Girl` | 柔和少女 | F | Y | Soft, gentle | Calm, soothing |\n| `Chinese (Mandarin)_Crisp_Girl` | 清脆少女 | F | Y | Crisp, clear | Bright, clear |\n| `Chinese (Mandarin)_Gentle_Senior` | 温柔学姐 | F | Y | Gentle senior girl | Campus, supportive |\n| `Chinese (Mandarin)_Wise_Women` | 阅历姐姐 | F | A | Experienced, wise | Advice, guidance |\n| `Chinese (Mandarin)_HK_Flight_Attendant` | 港普空姐 | F | A | HK accent flight attendant | Regional, entertainment |\n| `Chinese (Mandarin)_Cute_Spirit` | 憨憨萌兽 | N | C | Cute cartoon spirit | Animations, children's |\n| `Chinese (Mandarin)_Humorous_Elder` | 搞笑大爷 | M | E | Humorous old man | Comedy, entertainment |\n| `Chinese (Mandarin)_Kind-hearted_Elder` | 花甲奶奶 | F | E | Kind elderly lady | Stories, warm |\n| `Chinese (Mandarin)_Kind-hearted_Antie` | 热心大婶 | F | E | Kind-hearted auntie | Warm, friendly |\n\n### Chinese Cantonese (粤语)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `Cantonese_ProfessionalHost（F)` | 专业女主持 | F | A | Professional host | Broadcasts, hosting |\n| `Cantonese_GentleLady` | 温柔女声 | F | A | Gentle female | Soft, warm |\n| `Cantonese_ProfessionalHost（M)` | 专业男主持 | M | A | Professional host | Broadcasts, hosting |\n| `Cantonese_PlayfulMan` | 活泼男声 | M | A | Playful male | Entertainment, casual |\n| `Cantonese_CuteGirl` | 可爱女孩 | F | C | Cute girl | Children's, animations |\n| `Cantonese_KindWoman` | 善良女声 | F | A | Kind female | Warm, friendly |\n\n### English\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `English_Trustworthy_Man` | Trustworthy Man | M | A | Reliable, sincere | Business, narration |\n| `English_Graceful_Lady` | Graceful Lady | F | A | Elegant, refined | Formal, professional |\n| `English_Aussie_Bloke` | Aussie Bloke | M | A | Casual Australian | Casual, entertainment |\n| `English_Whispering_girl` | Whispering Girl | F | Y | Soft whisper | Romance, intimate |\n| `English_Diligent_Man` | Diligent Man | M | A | Earnest, hardworking | Motivational, educational |\n| `English_Gentle-voiced_man` | Gentle-voiced Man | M | E | Soft-spoken, kind | Calm, supportive |\n| `English_Sweet_Girl` | Sweet Girl | F | C | Sweet, innocent | Children's, friendly |\n| `Charming_Lady` | Charming Lady | F | A | Elegant, sophisticated | Professional, romance |\n| `Attractive_Girl` | Attractive Girl | F | Y | Engaging female | Entertainment, marketing |\n| `Serene_Woman` | Serene Woman | F | A | Calm, peaceful | Meditation, relaxation |\n| `Santa_Claus` | Santa Claus | M | E | Festive, jolly | Holiday, children's |\n| `Charming_Santa` | Charming Santa | M | E | Smooth, charismatic | Holiday, entertainment |\n| `Grinch` | Grinch | M | A | Whiny, mischievous | Comedy, holiday |\n| `Rudolph` | Rudolph | N | C | Cute, nasal reindeer | Children's, holiday |\n| `Arnold` | Arnold | M | A | Deep, robotic | Sci-fi, action |\n| `Cute_Elf` | Cute Elf | N | C | Playful, tiny elf | Fantasy, children's |\n\n### Japanese (日本語)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `Japanese_IntellectualSenior` | Intellectual Senior | M | E | Wise, knowledgeable | Narration, educational |\n| `Japanese_DecisivePrincess` | Decisive Princess | F | A | Confident, royal | Animation, games |\n| `Japanese_LoyalKnight` | Loyal Knight | M | A | Brave, faithful | Fantasy, games |\n| `Japanese_DominantMan` | Dominant Man | M | A | Powerful, commanding | Action, leadership |\n| `Japanese_SeriousCommander` | Serious Commander | M | A | Stern, authoritative | Military, games |\n| `Japanese_ColdQueen` | Cold Queen | F | A | Distant, majestic | Drama, fantasy |\n| `Japanese_DependableWoman` | Dependable Woman | F | A | Reliable, supportive | Guidance |\n| `Japanese_GentleButler` | Gentle Butler | M | A | Polite, refined | Comedy, animation |\n| `Japanese_KindLady` | Kind Lady | F | A | Warm, gentle | Comforting |\n| `Japanese_CalmLady` | Calm Lady | F | A | Composed, serene | Meditation, relaxation |\n| `Japanese_OptimisticYouth` | Optimistic Youth | M | Y | Cheerful, positive | Youth, motivation |\n| `Japanese_GenerousIzakayaOwner` | Generous Izakaya Owner | M | A | Friendly, welcoming | Casual, comedy |\n| `Japanese_SportyStudent` | Sporty Student | M | Y | Energetic, athletic | Sports, youth |\n| `Japanese_InnocentBoy` | Innocent Boy | M | C | Pure, naive | Children's |\n| `Japanese_GracefulMaiden` | Graceful Maiden | F | Y | Elegant, gentle | Romance, drama |\n\n### Korean (한국어)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `Korean_SweetGirl` | Sweet Girl | F | C | Sweet, adorable | Children's, romance |\n| `Korean_CheerfulBoyfriend` | Cheerful Boyfriend | M | Y | Energetic, loving | Romance, dating |\n| `Korean_EnchantingSister` | Enchanting Sister | F | A | Charming, captivating | Family, drama |\n| `Korean_ShyGirl` | Shy Girl | F | Y | Timid, reserved | Comedy, romance |\n| `Korean_ReliableSister` | Reliable Sister | F | A | Trustworthy, dependable | Guidance |\n| `Korean_StrictBoss` | Strict Boss | M | A | Authoritative, demanding | Business, drama |\n| `Korean_SassyGirl` | Sassy Girl | F | Y | Bold, witty | Comedy, entertainment |\n| `Korean_ChildhoodFriendGirl` | Childhood Friend Girl | F | Y | Familiar, friendly | Romance, nostalgia |\n| `Korean_PlayboyCharmer` | Playboy Charmer | M | A | Smooth, flirtatious | Romance, entertainment |\n| `Korean_ElegantPrincess` | Elegant Princess | F | A | Graceful, royal | Animation, fantasy |\n| `Korean_BraveFemaleWarrior` | Brave Female Warrior | F | A | Courageous | Action, fantasy |\n| `Korean_BraveYouth` | Brave Youth | M | Y | Heroic | Action, youth |\n| `Korean_CalmLady` | Calm Lady | F | A | Composed, serene | Meditation, relaxation |\n| `Korean_EnthusiasticTeen` | Enthusiastic Teen | M | Y | Excited, energetic | Youth |\n| `Korean_SoothingLady` | Soothing Lady | F | A | Calming, comforting | Relaxation |\n| `Korean_IntellectualSenior` | Intellectual Senior | M | E | Wise, knowledgeable | Educational, narration |\n| `Korean_LonelyWarrior` | Lonely Warrior | M | A | Solitary, melancholic | Drama, fantasy |\n| `Korean_MatureLady` | Mature Lady | F | A | Sophisticated | Professional, drama |\n| `Korean_InnocentBoy` | Innocent Boy | M | C | Pure, naive | Children's |\n| `Korean_CharmingSister` | Charming Sister | F | A | Attractive, delightful | Family, romance |\n| `Korean_AthleticStudent` | Athletic Student | M | Y | Sporty, energetic | Sports, youth |\n| `Korean_BraveAdventurer` | Brave Adventurer | M | A | Courageous explorer | Adventure, fantasy |\n| `Korean_CalmGentleman` | Calm Gentleman | M | A | Composed, refined | Formal, professional |\n| `Korean_WiseElf` | Wise Elf | M | E | Ancient, mystical | Fantasy, narration |\n| `Korean_CheerfulCoolJunior` | Cheerful Cool Junior | M | Y | Popular, friendly | Youth, entertainment |\n| `Korean_DecisiveQueen` | Decisive Queen | F | A | Commanding | Drama, fantasy |\n| `Korean_ColdYoungMan` | Cold Young Man | M | Y | Distant, aloof | Drama, romance |\n| `Korean_MysteriousGirl` | Mysterious Girl | F | Y | Enigmatic, secretive | Mystery, drama |\n| `Korean_QuirkyGirl` | Quirky Girl | F | Y | Eccentric, unique | Comedy |\n| `Korean_ConsiderateSenior` | Considerate Senior | M | E | Thoughtful, caring | Warm, supportive |\n| `Korean_CheerfulLittleSister` | Cheerful Little Sister | F | C | Playful, adorable | Family, comedy |\n| `Korean_DominantMan` | Dominant Man | M | A | Powerful, commanding | Leadership, action |\n| `Korean_AirheadedGirl` | Airheaded Girl | F | Y | Bubbly, spacey | Comedy |\n| `Korean_ReliableYouth` | Reliable Youth | M | Y | Trustworthy, dependable | Supportive |\n| `Korean_FriendlyBigSister` | Friendly Big Sister | F | A | Warm, protective | Family, support |\n| `Korean_GentleBoss` | Gentle Boss | M | A | Kind, understanding | Business |\n| `Korean_ColdGirl` | Cold Girl | F | Y | Aloof, distant | Drama, romance |\n| `Korean_HaughtyLady` | Haughty Lady | F | A | Arrogant, proud | Drama, comedy |\n| `Korean_CharmingElderSister` | Charming Elder Sister | F | A | Graceful | Romance, family |\n| `Korean_IntellectualMan` | Intellectual Man | M | A | Smart, knowledgeable | Educational |\n| `Korean_CaringWoman` | Caring Woman | F | A | Nurturing | Supportive, warm |\n| `Korean_WiseTeacher` | Wise Teacher | M | E | Experienced | Educational |\n| `Korean_ConfidentBoss` | Confident Boss | M | A | Self-assured, capable | Business, leadership |\n| `Korean_AthleticGirl` | Athletic Girl | F | Y | Sporty, energetic | Sports, fitness |\n| `Korean_PossessiveMan` | Possessive Man | M | A | Intense, protective | Romance, drama |\n| `Korean_GentleWoman` | Gentle Woman | F | A | Soft-spoken, kind | Calm |\n| `Korean_CockyGuy` | Cocky Guy | M | Y | Confident, arrogant | Comedy |\n| `Korean_ThoughtfulWoman` | Thoughtful Woman | F | A | Reflective, caring | Drama |\n| `Korean_OptimisticYouth` | Optimistic Youth | M | Y | Positive, hopeful | Motivation |\n\n### Spanish (Español)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `Spanish_Narrator` | Narrator | M | A | Professional narrator | Documentaries |\n| `Spanish_CaptivatingStoryteller` | Captivating Storyteller | M | A | Engaging narrator | Audiobooks |\n| `Spanish_WiseScholar` | Wise Scholar | M | A | Knowledgeable | Educational |\n| `Spanish_SereneWoman` | Serene Woman | F | A | Calm, peaceful | Relaxation |\n| `Spanish_MaturePartner` | Mature Partner | M | A | Sophisticated | Romance, drama |\n| `Spanish_ConfidentWoman` | Confident Woman | F | A | Self-assured | Professional |\n| `Spanish_DeterminedManager` | Determined Manager | M | A | Ambitious, driven | Business |\n| `Spanish_BossyLeader` | Bossy Leader | M | A | Commanding | Leadership |\n| `Spanish_ReservedYoungMan` | Reserved Young Man | M | Y | Quiet, introverted | Drama |\n| `Spanish_ThoughtfulMan` | Thoughtful Man | M | A | Reflective | Educational |\n| `Spanish_RationalMan` | Rational Man | M | A | Logical, analytical | Business |\n| `Spanish_Deep-tonedMan` | Deep-toned Man | M | A | Deep, resonant | Commanding |\n| `Spanish_Jovialman` | Jovial Man | M | A | Cheerful, friendly | Entertainment |\n| `Spanish_Steadymentor` | Steady Mentor | M | A | Reliable mentor | Guidance |\n| `Spanish_ReliableMan` | Reliable Man | M | A | Trustworthy | Professional |\n| `Spanish_RomanticHusband` | Romantic Husband | M | A | Loving, romantic | Romance |\n| `Spanish_Comedian` | Comedian | M | A | Humorous | Comedy |\n| `Spanish_Debator` | Debator | M | A | Persuasive | Debate |\n| `Spanish_ToughBoss` | Tough Boss | M | A | Harsh, demanding | Business, drama |\n| `Spanish_AngryMan` | Angry Man | M | A | Frustrated | Drama, comedy |\n| `Spanish_PowerfulSoldier` | Powerful Soldier | M | A | Strong, brave | Action, military |\n| `Spanish_PassionateWarrior` | Passionate Warrior | M | A | Fierce, dedicated | Action, fantasy |\n| `Spanish_PowerfulVeteran` | Powerful Veteran | M | A | Experienced | Military |\n| `Spanish_SensibleManager` | Sensible Manager | M | A | Practical | Business |\n| `Spanish_Kind-heartedGirl` | Kind-hearted Girl | F | C | Warm, compassionate | Children's |\n| `Spanish_SophisticatedLady` | Sophisticated Lady | F | A | Elegant, refined | Formal |\n| `Spanish_FrankLady` | Frank Lady | F | A | Direct, honest | Comedy |\n| `Spanish_Fussyhostess` | Fussy Hostess | F | A | Demanding | Comedy, drama |\n| `Spanish_Wiselady` | Wise Lady | F | E | Experienced, wise | Guidance |\n| `Spanish_ThoughtfulLady` | Thoughtful Lady | F | A | Considerate | Advice |\n| `Spanish_AssertiveQueen` | Assertive Queen | F | A | Commanding | Drama, fantasy |\n| `Spanish_CaringGirlfriend` | Caring Girlfriend | F | Y | Nurturing | Romance |\n| `Spanish_ChattyGirl` | Chatty Girl | F | Y | Talkative, sociable | Comedy |\n| `Spanish_CompellingGirl` | Compelling Girl | F | Y | Persuasive | Marketing |\n| `Spanish_WhimsicalGirl` | Whimsical Girl | F | C | Playful, imaginative | Children's |\n| `Spanish_Intonategirl` | Intonate Girl | F | Y | Musical, melodic | Singing |\n| `Spanish_SincereTeen` | Sincere Teen | M | Y | Honest, genuine | Youth |\n| `Spanish_Strong-WilledBoy` | Strong-willed Boy | M | Y | Determined | Youth, motivation |\n| `Spanish_EnergeticBoy` | Energetic Boy | M | C | Active, lively | Youth, sports |\n| `Spanish_StrictBoss` | Strict Boss | M | A | Strict | Business |\n| `Spanish_HumorousElder` | Humorous Elder | M | E | Funny | Comedy |\n| `Spanish_SereneElder` | Serene Elder | M | E | Calm, peaceful | Meditation |\n| `Spanish_SantaClaus` | Santa Claus | M | E | Festive | Holiday |\n| `Spanish_Rudolph` | Rudolph | N | C | Reindeer | Holiday |\n| `Spanish_Arnold` | Arnold | M | A | Robotic | Sci-fi |\n| `Spanish_Ghost` | Ghost | N | A | Spooky | Horror |\n| `Spanish_AnimeCharacter` | Anime Character | N | Y | Anime-style | Animation |\n\n### Portuguese (Português)\n\n| voice_id | Name | G | Age | Description | Best For |\n|----------|------|---|-----|-------------|----------|\n| `Portuguese_Narrator` | Narrator | M | A | Professional narrator | Documentaries |\n| `Portuguese_CaptivatingStoryteller` | Captivating Storyteller | M | A | Engaging narrator | Audiobooks |\n| `Portuguese_WiseScholar` | Wise Scholar | M | A | Knowledgeable | Educational |\n| `Portuguese_Deep-VoicedGentleman` | Deep-voiced Gentleman | M | A | Deep, rich | Commanding |\n| `Portuguese_ReservedYoungMan` | Reserved Young Man | M | Y | Quiet, introverted | Drama |\n| `Portuguese_ThoughtfulMan` | Thoughtful Man | M | A | Reflective | Educational |\n| `Portuguese_RationalMan` | Rational Man | M | A | Logical | Business |\n| `Portuguese_Jovialman` | Jovial Man | M | A | Cheerful | Entertainment |\n| `Portuguese_Steadymentor` | Steady Mentor | M | A | Reliable mentor | Guidance |\n| `Portuguese_ReliableMan` | Reliable Man | M | A | Trustworthy | Professional |\n| `Portuguese_RomanticHusband` | Romantic Husband | M | A | Loving | Romance |\n| `Portuguese_Comedian` | Comedian | M | A | Humorous | Comedy |\n| `Portuguese_Debator` | Debator | M | A | Persuasive | Debate |\n| `Portuguese_ToughBoss` | Tough Boss | M | A | Demanding | Business |\n| `Portuguese_StrictBoss` | Strict Boss | M | A | Strict | Business |\n| `Portuguese_AngryMan` | Angry Man | M | A | Frustrated | Drama |\n| `Portuguese_Godfather` | Godfather | M | A | Authoritative | Drama |\n| `Portuguese_PowerfulSoldier` | Powerful Soldier | M | A | Strong, brave | Action |\n| `Portuguese_PowerfulVeteran` | Powerful Veteran | M | A | Experienced |","readmeExcerpt":"Skill: Minimax-Multimodal-Toolkit Owner: minimax-ai-dev Summary: Use mmx to generate text, images, video, speech, and music via the MiniMax AI platform. Use when the user wants to create media content, chat with MiniMax mo... Tags: latest:1.0.2 Version history: v1.0.2 | 2026-04-08T22:01:58.739Z | user **Switch to mmx-cli: major simplification and platform change** - Replaces all separate bash scripts and API referenc","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"# Install\nnpm install -g mmx-cli\n\n# Auth (persisted to ~/.mmx/credentials.json)\nmmx auth login --api-key sk-xxxxx\n\n# Or pass per-call\nmmx text chat --api-key sk-xxxxx --message \"Hello\""},{"language":"bash","snippet":"mmx text chat --message <text> [flags]"},{"language":"bash","snippet":"# Single message\nmmx text chat --message \"user:What is MiniMax?\" --output json --quiet\n\n# Multi-turn\nmmx text chat \\\n  --system \"You are a coding assistant.\" \\\n  --message \"user:Write fizzbuzz in Python\" \\\n  --output json\n\n# From file\ncat conversation.json | mmx text chat --messages-file - --output json"},{"language":"bash","snippet":"mmx image generate --prompt <text> [flags]"},{"language":"bash","snippet":"mmx image generate --prompt \"A cat in a spacesuit\" --output json --quiet\n# stdout: image URLs (one per line in quiet mode)\n\nmmx image generate --prompt \"Logo\" --n 3 --out-dir ./gen/ --quiet\n# stdout: saved file paths (one per line)"},{"language":"bash","snippet":"mmx video generate --prompt <text> [flags]"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: mmx-cli\ndescription: Use mmx to generate text, images, video, speech, and music via the MiniMax AI platform. Use when the user wants to create media content, chat with MiniMax models, perform web search, or manage MiniMax API resources from the terminal.\n---\n\n# MiniMax CLI — Agent Skill Guide\n\nUse `mmx` to generate text, images, video, speech, music, and perform web search via the MiniMax AI platform.\n\n## Prerequisites\n\n```bash\n# Install\nnpm install -g mmx-cli\n\n# Auth (persisted to ~/.mmx/credentials.json)\nmmx auth login --api-key sk-xxxxx\n\n# Or pass per-call\nmmx text chat --api-key sk-xxxxx --message \"Hello\"\n```\n\nRegion is auto-detected. Override with `--region global` or `--region cn`.\n\n---\n\n## Agent Flags\n\nAlways use these flags in non-interactive (agent/CI) contexts:\n\n| Flag | Purpose |\n|---|---|\n| `--non-interactive` | Fail fast on missing args instead of prompting |\n| `--quiet` | Suppress spinners/progress; stdout is pure data |\n| `--output json` | Machine-readable JSON output |\n| `--async` | Return task ID immediately (video generation) |\n| `--dry-run` | Preview the API request without executing |\n| `--yes` | Skip confirmation prompts |\n\n---\n\n## Commands\n\n### text chat\n\nChat completion. Default model: `MiniMax-M2.7`.\n\n```bash\nmmx text chat --message <text> [flags]\n```\n\n| Flag | Type | Description |\n|---|---|---|\n| `--message <text>` | string, **required**, repeatable | Message text. Prefix with `role:` to set role (e.g. `\"system:You are helpful\"`, `\"user:Hello\"`) |\n| `--messages-file <path>` | string | JSON file with messages array. Use `-` for stdin |\n| `--system <text>` | string | System prompt |\n| `--model <model>` | string | Model ID (default: `MiniMax-M2.7`) |\n| `--max-tokens <n>` | number | Max tokens (default: 4096) |\n| `--temperature <n>` | number | Sampling temperature (0.0, 1.0] |\n| `--top-p <n>` | number | Nucleus sampling threshold |\n| `--stream` | boolean | Stream tokens (default: on in TTY) |\n| `--tool <json-or-path>` | string, repeatable | Tool definition JSON or file path |\n\n```bash\n# Single message\nmmx text chat --message \"user:What is MiniMax?\" --output json --quiet\n\n# Multi-turn\nmmx text chat \\\n  --system \"You are a coding assistant.\" \\\n  --message \"user:Write fizzbuzz in Python\" \\\n  --output json\n\n# From file\ncat conversation.json | mmx text chat --messages-file - --output json\n```\n\n**stdout**: response text (text mode) or full response object (json mode).\n\n---\n\n### image generate\n\nGenerate images. Model: `image-01`.\n\n```bash\nmmx image generate --prompt <text> [flags]\n```\n\n| Flag | Type | Description |\n|---|---|---|\n| `--prompt <text>` | string, **required** | Image description |\n| `--aspect-ratio <ratio>` | string | e.g. `16:9`, `1:1` |\n| `--n <count>` | number | Number of images (default: 1) |\n| `--subject-ref <params>` | string | Subject reference: `type=character,image=path-or-url` |\n| `--out-dir <dir>` | string | Download images to directory |\n| `--out-prefix <prefix>` | string | Filename prefix (default:"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn70p6rdfg6k3598at7gm6r5fd82h5rs\",\n  \"slug\": \"minimax-multimodal\",\n  \"version\": \"1.0.2\",\n  \"publishedAt\": 1775685718739\n}"},{"path":"skill-card.md","content":"## Description:\n\nUse mmx to generate text, images, video, speech, and music via the MiniMax AI platform.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[minimax-ai-dev](https://clawhub.ai/user/minimax-ai-dev)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and agents use this skill to call the MiniMax CLI for multimodal generation, image understanding, web search, quota checks, and related API-resource workflows from the terminal.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill relies on a global npm install for mmx-cli, and the release evidence notes that the package is unpinned.\n\nMitigation: Verify the npm package and version before installation, and install only in environments where that package is trusted.\n\nRisk: The skill shows API-key usage patterns that can expose MiniMax credentials if copied directly into shell history or logs.\n\nMitigation: Prefer environment variables or secret-manager workflows, and avoid placing real API keys directly in commands.\n\nRisk: The --yes flag can bypass confirmations before quota-consuming operations.\n\nMitigation: Use --dry-run or review command parameters before using --yes in automated workflows.\n\n## Reference(s):\n\n\n## Skill Output:\n\n**Output Type(s):** [Guidance, Shell commands, Configuration, Text, JSON, Files]\n\n**Output Format:** [Markdown guidance with inline shell commands and expected CLI output formats]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [The skill guides mmx CLI calls that may return text, JSON, task IDs, file paths, media URLs, or generated media files depending on the command.]\n\n## Skill Version(s):\n\n1.0.2 (source: release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Use mmx to generate text, images, video, speech, and music via the MiniMax AI platform. Use when the user wants to create media content, chat with MiniMax mo... Skill: Minimax-Multimodal-Toolkit Owner: minimax-ai-dev Summary: Use mmx to generate text, images, video, speech, and music via the MiniMax AI platform. Use when the user wants to create media content, chat with MiniMax mo... Tags: latest:1.0.2 Version history: v1.0.2 | 2026-04-08T22:01:58.739Z | user **Switch to mmx-cli: major simplification and platform change** - Replaces all separate bash scripts and API referenc","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1239,"uniquenessScore":48,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T03:48:47.536Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T03:48:47.536Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T23:14:58.337Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}