{"id":"f1fd3615-cb81-47da-989e-d5d5cd509327","entityType":"agent","slug":"clawhub-luischarro-music-craft-minimax","name":"Music Craft — MiniMax","canonicalUrl":"https://www.xpersona.co/agent/clawhub-luischarro-music-craft-minimax","canonicalPath":"/agent/clawhub-luischarro-music-craft-minimax","generatedAt":"2026-10-11T14:16:54.981Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T11:16:31.985Z","emptyReason":null},"description":"MiniMax-native music generation for OpenClaw — cover and style transfer that preserves melody, two-song mashups, AI lyrics generation and edit, emotion-driven prompt engineering, and per-flag mmx CLI control over BPM, key, structure, and avoid lists. Extends music-craft with MiniMax Music 2.6 features.","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.1K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s171padhas9w52sdjc3110vhss83j8jp:music-craft-minimax","sourceUrl":"https://clawhub.ai/luischarro/music-craft-minimax","homepage":"https://clawhub.ai/luischarro/skills/music-craft-minimax","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/luischarro/music-craft-minimax","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/luischarro/skills/music-craft-minimax","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":61,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Music Craft — MiniMax technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T11:16:31.985Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T11:16:31.985Z","emptyReason":null},"stars":null,"forks":null,"downloads":1079,"packageName":null,"latestVersion":"1.6.0","tractionLabel":"1.1K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T11:16:31.970Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T11:16:31.985Z","lastCrawledAt":"2026-10-11T11:16:31.970Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T11:16:31.970Z","lastVerifiedAt":null,"highlights":[{"version":"1.6.0","createdAt":"2026-07-30T11:39:14.548Z","changelog":"v1.6.0: discoverability reorder — hero sections (What is Music Craft — MiniMax? / MiniMax-specific features / Why use this instead of music-craft? / Quick Start) at top of SKILL.md, licensing and commercial-use gate moved to bottom. No behaviour, no env vars, no bins.","fileCount":49,"zipByteSize":276256},{"version":"1.5.1","createdAt":"2026-07-28T13:17:07.814Z","changelog":"v1.5.1: document ClawHub MIT-0 scope, require own MiniMax credentials, and distinguish API terms from consumer app terms","fileCount":44,"zipByteSize":230429},{"version":"1.5.0","createdAt":"2026-06-14T05:45:29.171Z","changelog":"v1.5.0: BREAKING — local audio files only, URL/image/LRCLib flags removed, cloud transmission consent sharpened","fileCount":44,"zipByteSize":229506},{"version":"1.4.1","createdAt":"2026-06-13T15:20:41.914Z","changelog":"v1.4.1: add lyrics tag safety checks, semantic lyrics verification, and require mmx --out with wrapper output-path","fileCount":50,"zipByteSize":268107},{"version":"1.4.0","createdAt":"2026-06-13T11:46:59.371Z","changelog":"v1.4.0: add arranger helper release notes, clarify MiniMax-vs-ACE-Step routing, and remove maintainer-specific/private details from published docs","fileCount":48,"zipByteSize":261202},{"version":"1.3.0","createdAt":"2026-06-13T05:05:53.827Z","changelog":"v1.3.0: harden MiniMax retry and output verification flows, add cloud reliability checks, and expand field-run guidance and prompt recipes","fileCount":45,"zipByteSize":252340},{"version":"1.1.0","createdAt":"2026-06-11T16:44:36.579Z","changelog":"v1.1.0: add MiniMax execution caveats, truncation warnings, and safer wrapper output handling","fileCount":43,"zipByteSize":240461},{"version":"1.0.1","createdAt":"2026-06-09T22:51:42.448Z","changelog":"v1.0.1: Rename display name from 'OpenClaw Music Workflow — MiniMax' to 'Music Craft — MiniMax' for slug consistency. Bundle body unchanged from v1.0.0 (999-line SKILL.md, 10 reference docs, 21 scripts, 34 files, 587 KB).","fileCount":35,"zipByteSize":203961}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s171padhas9w52sdjc3110vhss83j8jp:music-craft-minimax","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s171padhas9w52sdjc3110vhss83j8jp:music-craft-minimax` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/luischarro/music-craft-minimax before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-luischarro-music-craft-minimax/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-luischarro-music-craft-minimax/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-luischarro-music-craft-minimax/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-luischarro-music-craft-minimax/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-luischarro-music-craft-minimax/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-luischarro-music-craft-minimax/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T14:16:54.976Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-luischarro-music-craft-minimax/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-luischarro-music-craft-minimax/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-luischarro-music-craft-minimax/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-luischarro-music-craft-minimax/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T11:16:31.985Z","emptyReason":null},"readme":"Skill: Music Craft — MiniMax\n\nOwner: luischarro\n\nSummary: MiniMax-native music generation for OpenClaw — cover and style transfer that preserves melody, two-song mashups, AI lyrics generation and edit, emotion-driven prompt engineering, and per-flag mmx CLI control over BPM, key, structure, and avoid lists. Extends music-craft with MiniMax Music 2.6 features.\n\nTags: latest:1.6.0\n\nVersion history:\n\nv1.6.0 | 2026-07-30T11:39:14.548Z | user\n\nv1.6.0: discoverability reorder — hero sections (What is Music Craft — MiniMax? / MiniMax-specific features / Why use this instead of music-craft? / Quick Start) at top of SKILL.md, licensing and commercial-use gate moved to bottom. No behaviour, no env vars, no bins.\n\nv1.5.1 | 2026-07-28T13:17:07.814Z | user\n\nv1.5.1: document ClawHub MIT-0 scope, require own MiniMax credentials, and distinguish API terms from consumer app terms\n\nv1.5.0 | 2026-06-14T05:45:29.171Z | user\n\nv1.5.0: BREAKING — local audio files only, URL/image/LRCLib flags removed, cloud transmission consent sharpened\n\nv1.4.1 | 2026-06-13T15:20:41.914Z | user\n\nv1.4.1: add lyrics tag safety checks, semantic lyrics verification, and require mmx --out with wrapper output-path\n\nv1.4.0 | 2026-06-13T11:46:59.371Z | user\n\nv1.4.0: add arranger helper release notes, clarify MiniMax-vs-ACE-Step routing, and remove maintainer-specific/private details from published docs\n\nv1.3.0 | 2026-06-13T05:05:53.827Z | user\n\nv1.3.0: harden MiniMax retry and output verification flows, add cloud reliability checks, and expand field-run guidance and prompt recipes\n\nv1.1.0 | 2026-06-11T16:44:36.579Z | user\n\nv1.1.0: add MiniMax execution caveats, truncation warnings, and safer wrapper output handling\n\nv1.0.1 | 2026-06-09T22:51:42.448Z | user\n\nv1.0.1: Rename display name from 'OpenClaw Music Workflow — MiniMax' to 'Music Craft — MiniMax' for slug consistency. Bundle body unchanged from v1.0.0 (999-line SKILL.md, 10 reference docs, 21 scripts, 34 files, 587 KB).\n\nv1.0.0 | 2026-06-09T22:41:11.365Z | user\n\nFirst stable release. MiniMax Music 2.6 power-user upgrade: cover/style transfer, two-song mashup, lyrics generation API, emotion-driven prompt engineering, fine mmx flag control, 26-test smoke suite. Inherits shared content (output file layout, rate limits, anti-sparse rules) from music-craft via cross-references.\n\nArchive index:\n\nArchive v1.6.0: 49 files, 276256 bytes\n\nFiles: README.md (6468b), references/advanced-audio-analysis.md (26306b), references/changelog.md (13392b), references/cover-workflow.md (11095b), references/emotion-analysis.md (29021b), references/emotion-delivery.md (38670b), references/error-handling.md (19280b), references/examples.md (10851b), references/free-tool-inputs.md (21747b), references/lyrics-generation.md (8546b), references/mashup-workflow.md (14829b), references/minimax-generation-caveats.md (5833b), references/mmx-flags-reference.md (16959b), references/mmx-recipe-pattern.md (33896b), references/music-3-migration.md (3972b), references/orchestrator-quickstart.md (5867b), references/quota-checking.md (23848b), references/setup-and-preflight.md (10500b), references/short-prompt-recipes.md (2056b), references/vocal-workflow.md (55261b), scripts/_analysis_cache.py (3466b), scripts/_audio_features.py (5329b), scripts/_key_compat.py (9604b), scripts/analysis_orchestrator.py (11901b), scripts/analyze_audio.py (6686b), scripts/analyze_two_songs.py (13152b), scripts/analyze_vocal_emotion.py (61198b), scripts/batch_cover.py (6439b), scripts/check_environment.py (4116b), scripts/check-quota.py (6684b), scripts/classify_instruments.py (8078b), scripts/compute_audio_embedding.py (11217b), scripts/emotion_to_prompt.py (61593b), scripts/extract_lyrics_whisper.py (8755b), scripts/extract_melody.py (11778b), scripts/extract_stems.py (9483b), scripts/finalize_track.sh (590b), scripts/generate_with_retry.py (23025b), scripts/hybrid_remix.py (6055b), scripts/lint_lyrics.py (4578b), scripts/lint_music_request.py (38221b), scripts/per_stem_analysis.py (4643b), scripts/smoke_test.py (83862b), scripts/track_beats.py (7240b), scripts/verify_cloud_output.sh (3974b), scripts/verify_lyrics_alignment.py (3768b), skill-card.md (3669b), SKILL.md (43725b), _meta.json (138b)\n\nFile v1.6.0:SKILL.md\n\n---\nname: music-craft-minimax\nversion: 1.6.0\ndescription: MiniMax-native music generation for OpenClaw — cover and style transfer that preserves melody, two-song mashups, AI lyrics generation and edit, emotion-driven prompt engineering, and per-flag mmx CLI control over BPM, key, structure, and avoid lists. Extends music-craft with MiniMax Music 2.6 features.\nmetadata: '{\"openclaw\":{\"requires\":{\"env\":[\"MINIMAX_API_KEY\"],\"bins\":[\"python3\",\"ffmpeg\",\"mmx\"]},\"primaryEnv\":\"MINIMAX_API_KEY\",\"emoji\":\"\\ud83c\\udfb6\",\"homepage\":\"https://github.com/LuisCharro/skills/tree/main/publish/music-craft-minimax\",\"envVars\":[{\"name\":\"MINIMAX_API_KEY\",\"required\":true,\"description\":\"API key for the MiniMax Music 2.6 token plan. Required for cover, mashup, lyrics generation, and mmx flag control.\"}]}}'\n---\n# Music Craft — MiniMax\n\n## What is Music Craft — MiniMax?\n\nMusic Craft — MiniMax is the **MiniMax-native power-user upgrade** of [`music-craft`](../music-craft/). It unlocks the features that need the MiniMax Music 2.6 token plan: cover and style transfer that preserves melody, two-song mashups, AI lyrics generation with edit-iter, emotion analysis on input audio, and precise per-flag control over BPM, key, structure, and avoid lists via the `mmx` CLI.\n\n## MiniMax-specific features\n\n- **Cover and style transfer** — preserve the melody of Song X, apply the style of Song Y.\n- **Two-song mashup** — Song A's lyrics and emotion + Song B's style, in one output.\n- **Lyrics generation API** — `write_full_song` for blank-page or `edit` for revisions, with structure tags.\n- **Emotion analysis** — intensity, vocal speed, pitch bends, and 25+ emotion classes drive your prompt.\n- **`mmx` per-flag control** — `--bpm`, `--key`, `--structure`, `--vocals`, `--genre`, `--mood`, `--instruments`, `--avoid` as separate flags; prompt and flags lint-checked before generation.\n- **Quota-aware batch runs** — 5-hour rolling window check before any cloud batch.\n\n## Why use this instead of music-craft?\n\nUse **music-craft** when you want a structured default that picks the backend for you, or when exact-duration vocal tracks matter. Use **music-craft-minimax** when you need a MiniMax-only feature — cover, mashup, lyrics API, emotion-driven prompting — or when you specifically want `mmx` flag-level control over BPM, key, structure, or an avoid list.\n\n## Quick Start\n\n1. Decide which MiniMax-only feature you need (cover, mashup, lyrics API, emotion analysis).\n2. Provide one or two source audio files / lyrics / emotion inputs as the skill requests.\n3. The skill picks the right `mmx` flag combination and calls the API.\n4. It lints flags + prompt for conflicts, then runs sequentially (one job at a time, never parallel).\n5. It verifies duration, file size, and quota headroom before delivering.\n\n## When To Use\n\nUse this skill when the task involves:\n\n- generating a cover of an existing song with a different style (chanson version of a rock track, reggaeton version of a pop hit, and so on). Source must be a local file.\n- style transfer from a local audio file to a target genre\n- two-song mashup where Song A's lyrics and emotional arc are kept, but Song B's style is applied\n- emotion analysis on input audio to extract intensity curves, vocal speed, pitch bends, and emotion classifications\n- generating lyrics in a specific language and theme via the MiniMax `lyrics_generation` API\n- editing existing lyrics to match a target style or emotional arc (MiniMax `lyrics_generation` edit mode)\n- using `mmx` CLI directly for fine control over `--avoid`, `--bpm`, `--key`, `--structure`, `--vocals`, `--instruments` as separate flags\n- accessing MiniMax's `music-cover` or `music-cover-free` models for melody preservation\n\n## When NOT To Use\n\nDo not use this skill when:\n\n- the user only needs standard song generation without cover, mashup, or analysis — use `music-craft` instead (lighter, no MiniMax dependency)\n- the runtime does not expose a `music_generate` tool and there is no `MINIMAX_API_KEY` configured — both skills need the runtime\n- the user wants deterministic, single-shot generation with no iteration — overkill\n- the user wants to mutate a specific existing audio file (pitch shift, time stretch, stem split) — that is post-production, not generation\n- the user is not on a MiniMax Token Plan — the advanced features (cover, mmx per-flag control, lyrics API, emotion-driven prompts) require the plan\n- the user needs a reliable full-length 3:00+ song with exact duration — prefer `music-craft` with ACE-Step instead; MiniMax is the right tool when speed, convenience, cover workflows, mashups, or mmx flag control matter more than exact output length\n\n## Decision Tree\n\nUse the base skill unless one of these MiniMax-specific needs is present:\n\n- melody-preserving cover or style transfer from a local audio file\n- two-song mashup\n- lyrics API preview/edit flow\n- emotion analysis that feeds the prompt\n- exact `mmx` control for BPM, key, structure, or avoid lists\n\nIf the user wants a new song that only borrows a style, stay in `music-craft` unless they also need exact flag control or lyrics API iteration.\n\nIf the user provides a URL, do not attempt to download it. Tell them this skill accepts only local files, and suggest `music-source-fetch` if they need to fetch audio by title.\n\n## Required MiniMax Workflow\n\n> **Operator rules for MiniMax generations:**\n>\n> - Run MiniMax generations sequentially, not in parallel.\n> - Do not assume the CLI will honor a requested output path — pass `--out` to `mmx` and verify the file exists after each run.\n> - Treat requested duration as a target, not a guarantee. Verified range: 57-135% of requested duration.\n> - If `mmx` exits with SIGTERM/SIGKILL after saving, verify the file before rerunning; file existence is the source of truth.\n> - Before running multi-output MiniMax generations, load [`references/minimax-generation-caveats.md`](references/minimax-generation-caveats.md).\n\nFor every cover, style-transfer, mashup, or precision `mmx` generation, follow this checklist in order:\n\n1. **Analyze source audio** with `scripts/analysis_orchestrator.py` when audio is available.\n2. **Build M1 + M2 prompts** from the analysis: M1 is the primary style, M2 is a strong contrast style.\n3. **Lint both prompts and lyrics** with `scripts/lint_music_request.py` before generation. Use `--lyrics-file` when lyrics exist so invalid tags and duration density are caught. Stop on blockers.\n4. **Generate with retry** via `scripts/generate_with_retry.py --output-path <final.mp3> -- music generate ... --out <final.mp3>` or `-- music cover ... --out <final.mp3>`. `--output-path` does not replace `mmx --out`.\n5. **Verify outputs**: duration, LUFS/peak, file size, audible completeness, and lyrics alignment. When lyrics matter, listen or transcribe and use `scripts/verify_lyrics_alignment.py` before delivery.\n6. **Finalize delivery copy** with `scripts/finalize_track.sh input.mp3 output.mp3` when the user wants production-ready loudness.\n7. **Deliver both versions** with a short analysis summary and any caveats.\n\nPreserve the anti-sparse guard in prompts: fully arranged, instruments keep playing, no a cappella dropouts unless explicitly requested.\n\n## Reliability Caveats\n\nLoad [`references/minimax-generation-caveats.md`](references/minimax-generation-caveats.md)\nbefore any multi-output, long-duration, or quota-sensitive cloud run.\n\n- `--length` is a hint, not a contract. Cloud output can undershoot or overshoot\n  the request; local ACE-Step is the exact-duration fallback.\n- A shell SIGTERM/SIGKILL after save can still be a successful generation. Check\n  file existence, file size, and `ffprobe` duration before retrying.\n- Keep standard cloud generation prompts under about 500 characters when\n  possible. Put detailed production sheets in the local ACE-Step route.\n- Treat `--references` as optional and potentially flaky. Inline concise\n  references in the prompt when reliability matters.\n- Expected cloud output is MP3, stereo, 44.1 kHz, about 256 kbps. There are no\n  documented flags for FLAC, 48 kHz, or bitrate selection.\n- There is no documented MiniMax batch API or preview/draft mode. For repeated\n  covers, use `scripts/batch_cover.py` so runs remain sequential and verified.\n- Quota usage may not be visible from the CLI. Check the MiniMax dashboard when\n  planning large batches.\n\nShort cloud prompt recipes: [`references/short-prompt-recipes.md`](references/short-prompt-recipes.md).\n\n## Cloud Workflows\n\nFor sequential, reproducible, quota-aware `mmx` runs use the typed `mmx_recipe` wrapper pattern: a frozen `MMXReceipt` (argv, output, stdout/stderr, rc, elapsed, optional quota snapshot) plus a `--dry` knob and `check_quota=True` pre-spend snapshot. The skill's [`scripts/generate_with_retry.py`](scripts/generate_with_retry.py) already covers the mmx-music operational contract (transient retry, `--timeout 600`, signal recovery, file move); roadmap v1.1.5 item 17 plans to fold the typed-receipt shape and `--dry` into that wrapper. Reference implementation, contract, and composition recipe: [`references/mmx-recipe-pattern.md`](references/mmx-recipe-pattern.md).\n\n## Quota\n\nThe 5-hour rolling window on the MiniMax Token Plan is the real ceiling on this skill, not the documented 120 RPM API limit. Check before any multi-output, quota-sensitive, or batch cloud run with `mmx quota show --output json` (or the HTTP `/v1/token_plan/remains` endpoint), and short-circuit the batch when headroom is low. **For pre-flight gates, use the shipped [`scripts/check-quota.py`](scripts/check-quota.py)** (verified against `mmx` CLI v1.0.16; supports `--quiet` / `--budget N` / `--json`). Coverage, cost-per-call estimates, and the live-check commands: [`references/quota-checking.md`](references/quota-checking.md).\n\n## Routing and Blocker Checks\n\nClassify the request before analysis or generation:\n\n- **Text-only style reference** means the user gave a song name, artist, era, or genre cue without source audio. Treat it as style inference, not cover analysis.\n- **Reference audio** means the user provided a local file that should be analyzed. URLs are not accepted; ask for a local file path.\n- **Cover** preserves melody and usually needs a source file plus a target style decision.\n- **Style transfer** uses a reference track or analyzed audio as style input, then changes the production direction.\n- **Mashup** needs Song A and Song B, plus a decision about which one contributes content and which one contributes style.\n- **Emotion prompt** means the user wants analysis turned into descriptive prompt language, not a full cover.\n\nThe [`scripts/lint_music_request.py`](scripts/lint_music_request.py) helper emits one of these routes:\n\n| Route | When |\n| --- | --- |\n| `base_prompt` | Standard generation, no MiniMax-specific feature needed. |\n| `minimax_cover` | Melody-preserving cover from a local audio file. |\n| `minimax_mashup` | Two-song mashup (A + B, both identified). |\n| `minimax_style_transfer` | Style transfer that does not preserve the source melody. |\n| `minimax_emotion_prompt` | Emotion analysis, or precision `mmx` flag usage. |\n| `needs_clarification` | At least one blocker is unresolved; ask the user first. |\n\nSurface blockers before analysis:\n\n- no local source file (URLs are not accepted)\n- unclear which track is Song A versus Song B\n- missing target style\n- missing lyrics decision, such as original, translated, rewritten, or instrumental\n- conflicting cover/style-transfer intent: the user asked for both \"cover\" (preserve melody) and \"style transfer\" (reproduce style) at once. These are mutually exclusive. Ask the user to pick one.\n\nAfter you have prompt text and `mmx` flags, lint them together before generation:\n\n- compare prompt BPM with `--bpm`\n- compare prompt key with `--key`\n- compare prompt structure line with `--structure`\n- compare prompt duration with `--duration` (or implicit length expectation)\n- compare prompt vocal mode with `--vocals`\n- compare prompt language with `--language`\n- compare prompt avoid language with `--avoid`\n- stop when the prompt says one thing and the flags say another\n- warn when prompt text exceeds 1800 UTF-8 bytes\n- stop when prompt text exceeds 2000 UTF-8 bytes (observed API rejection at 2079 bytes)\n\nIf the user only has a text reference, route to the free-tool path in `references/free-tool-inputs.md` first. If the user has audio, analyze first and only then build the prompt. The linter returns a `retry_guidance` array with one hint per conflict so the operator can re-align prompt and flags on the next attempt.\n\n## Audio Source (Local Only)\n\nThe source for cover, mashup, or style transfer must be a local audio file\npath. URL inputs trigger a soft warning and the agent asks for a local\nfile. If you have a URL, fetch it locally first with the private\n`music-source-fetch` skill.\n\n## First Response Defaults\n\nUse these defaults on the first pass:\n\n- **Cover from a local audio file**: start with the one-step cover path. Switch to two-step only if the user wants translated lyrics, edited ASR lyrics, or custom lyrics.\n- **Style transfer only**: do not use cover unless melody preservation matters. Use standard generation plus `mmx` flags if exact BPM/key/structure matter.\n- **Two-song mashup**: anchor on Song A. If Song A has audio, default to the cover two-step workflow; if Song B is only named, ask for a short style description or fetch more context if free tools are available.\n- **Lyrics API generation or edit**: use `write_full_song` for blank-page generation and `edit` for revisions.\n- **Emotion-analysis-to-prompt**: run analysis first, then convert to a prompt; only ask whether the output should be cover, mashup, or standard generation, plus the target language if missing.\n- **Exact BPM/key/structure control**: make `mmx` flags the source of truth and keep the prompt descriptive but non-conflicting.\n\n## Ambiguity Questions\n\nAsk at most 1-3 questions. Separate blockers from quality tweaks:\n\n- Required blockers first: local source file, which song is A vs B, whether lyrics already exist, whether the output must preserve melody.\n- Optional quality after blockers: target language, target style, BPM, key, structure, instruments, vocal color, avoid list.\n\nUse these exact patterns when clarification is needed:\n\n- **Cover**: \"Which source should I use?\" \"Do you want the original lyrics, translated lyrics, or new lyrics?\" \"Any target style, or should I derive it from the source?\"\n- **Mashup**: \"Which song is A and which is B?\" \"Do you have audio for Song B, or only the name?\" \"Should the lyrics stay the same or be rewritten?\"\n- **Lyrics API**: \"Write from scratch or edit existing lyrics?\" \"What language should I target?\" \"Any hard structure requirements?\"\n- **Emotion prompt**: \"Do you want cover, mashup, or standard generation?\" \"What language should the output use?\" \"Should I prioritize tenderness, energy, or structure?\"\n- **mmx precision**: \"Which values are mandatory: BPM, key, structure, or avoid list?\" \"Any instruments or vocals that must stay in or stay out?\"\n\n## Request Intake (adapted for MiniMax features)\n\nAfter the Routing and Blocker Checks classify the request, run this 2-pass intake to extract the full set of fields the user cares about. Label each field's confidence: **clear** (user said it), **inferred** (sensible default), **missing** (need to ask), or **conflicting** (user said two incompatible things — pause to resolve).\n\n### Fields checklist (MiniMax-specific)\n\n| # | Field | What to look for | MiniMax-specific notes |\n| --- | --- | --- | --- |\n| 1 | Route | Cover / style transfer / mashup / standard / emotion prompt | From the Routing and Blocker Checks section. Determines which MiniMax features to use. |\n| 2 | Source audio | Local file path | Required for cover, mashup, style transfer. For standard, optional (text-only style reference is also fine). |\n| 3 | Song A identity | Name, artist, audio | For mashup: needed. For cover: this is the source. |\n| 4 | Song B identity | Name, artist, audio | For mashup only. |\n| 5 | Target style | Genre / mood / reference | The destination of the cover or style transfer. If user says \"like Rosalía\", that's clear. If user says \"something good\", that's missing. |\n| 6 | Lyrics decision | Original / translated / new / instrumental | For cover, default to original (translated if user requests it). For standard, default to new (or user-provided). |\n| 7 | Vocal mode | Solo / duet / choir / instrumental | Drives `--vocals` and `--language` flags. |\n| 8 | Language | BCP-47 code (en, fr, es, etc.) | For lyrics language AND vocal language. |\n| 9 | Duration | Approximate length (jingle ~30s, standard ~3min, epic ~6min) | `--length` is a hint in milliseconds, not a guarantee. Length is still driven mainly by lyrics + structure. |\n| 10 | BPM, key, structure | Exact values if user wants `--bpm`/`--key`/`--structure` | Optional. If provided, the prompt AND flags must agree (lint them). |\n| 11 | Emotion arc | For emotion-prompt workflows: which emotions to emphasize | Drives the analysis-to-prompt translation. |\n| 12 | **Output location** | Where the audio and analysis files go | Same as the base skill — per-song subfolder in `~/Music mix/<project>/<song-slug>/`. |\n\nConfidence map example: [`references/examples.md`](references/examples.md).\n\nIf any field is **missing** or **conflicting**, that's a question to ask. The `Ambiguity Questions` section below has specific patterns for each route. If everything is **clear** or **inferred**, the request is ready to translate.\n\n## User Preference Flow (message patterns → action)\n\nThe skill does not start with a questionnaire. It starts by reading and inferring from the user's natural-language request.\n\n| User says... | Skill does... |\n| --- | --- |\n| \"Haz un cover de X en Y\" | Route: `minimax_cover`. Ask: local source audio file path, target language for lyrics, vocal register. |\n| \"Make this song sound like Rosalía\" | Route: `minimax_style_transfer`. Ask: source audio, which album/era of Rosalía. |\n| \"I have audio of A, mash with B, keep A's melody\" | Route: `minimax_mashup`. Ask: A vs B confirmation, source audio for A, B can be name or audio. |\n| \"Analyze the emotion curve of this track\" | Route: `minimax_emotion_prompt` (analysis-only). Run `analysis_orchestrator.py --audio` first, then read the JSON. |\n| \"I want the lyrics to be about X, in French, melancholic\" | Route: `base_prompt` (standard). Use the lyrics API to generate, then pass to `mmx music generate --lyrics-file`. Ask: target BPM/key/structure or derive from analysis. |\n| \"Recreate the song but in 90 BPM D minor\" | Route: `base_prompt` with `mmx` flags. Lint prompt vs flags before generation. Verify BPM/key consistency. |\n| \"I don't know, surprise me\" | Pick a coherent default (e.g. upbeat indie pop, EN, ~3min, auto-lyrics, standard generation) and confirm with the user before generating. |\n| \"Same song again but as a reggaeton version\" | Route: `minimax_cover` with the existing song as source. Use the same project/song subfolder, suffix the MP3 (`M1_original.mp3` + `M2_reggaeton.mp3`). |\n\nThis table is the **abstract** of `references/user-preference-flow.md` (which lives in the base skill). If you want a more detailed case, defer to the base skill's table and combine with this skill's route mapping.\n\n## Output File Layout (Per-Song Subfolders)\n\n**MiniMax-specific additions** (drop these into the per-song subfolder alongside the base items):\n\n| File | Source | Notes |\n| --- | --- | --- |\n| `<song-slug>_analysis.json` | `analysis_orchestrator.py --output` | MiniMax-specific analysis results (emotion, BPM, key, segments) |\n| `<song-slug>_lyrics.txt` | `mmx music generate --lyrics-file` | Optional if user provided lyrics inline |\n| `<song-slug>_<style>_prompt.txt` | The exact text passed to `--prompt` | For reproducibility |\n\nThe LLM should aim for the base skill's layout by default. The MiniMax-specific files are added on top when MiniMax features are used (cover workflow, mashup, analysis, etc.).\n\nIf the runtime needs a `MEDIA:` delivery path, use a path without spaces or\ncopy the final file into a workspace media folder first. Keep the archival copy\nin the per-song output folder.\n\n## Relationship to `music-craft`\n\nThis skill **extends** the base skill, it does not replace it. The shared concepts are:\n\n| Concept | Where it lives |\n| --- | --- |\n| Pre-Flight Check (platform detection) | This skill (extended required list) |\n| Anti-sparse rules (canonical text) | Base skill, referenced from here |\n| Prompt formula (production sheet) | Base skill, referenced from here |\n| Structure tags (14 tags) | Base skill, referenced from here |\n| User preference flow (auto-detect + ask) | Base skill, referenced from here |\n| Output file layout (per-song subfolders, slug rules, version prefix) | Base skill, referenced from here; MiniMax adds analysis.json and lyrics.txt |\n| Rate limits (generic) | Base skill |\n| Quality verification checklist | Base skill, extended here for MiniMax |\n| Operating rules (6-step loop) | Base skill, summarized here with MiniMax-specific extensions |\n\nThe MiniMax-specific additions are:\n\n| MiniMax concept | Where it lives |\n| --- | --- |\n| `mmx` CLI quick reference | This skill |\n| `mmx` full flag reference | This skill, [`references/mmx-flags-reference.md`](references/mmx-flags-reference.md) |\n| Cover workflow (one-step, two-step) | This skill, [`references/cover-workflow.md`](references/cover-workflow.md) |\n| Lyrics generation API | This skill, [`references/lyrics-generation.md`](references/lyrics-generation.md) |\n| Mashup workflow (A + B) | This skill, [`references/mashup-workflow.md`](references/mashup-workflow.md) |\n| Emotion analysis (vocal speed, intensity, pitch) | This skill, [`references/emotion-analysis.md`](references/emotion-analysis.md) |\n| MiniMax-specific error handling | This skill, [`references/error-handling.md`](references/error-handling.md) |\n| Audio analysis scripts | This skill, [`scripts/`](scripts/) |\n| Free tool inputs (web, memory; image removed in v1.5.0) | Both skills — base layer in [`music-craft`](../music-craft/), MiniMax layer here in [`references/free-tool-inputs.md`](references/free-tool-inputs.md) |\n\n## Pre-Flight Check\n\nRun the extended pre-flight in\n[`references/setup-and-preflight.md`](references/setup-and-preflight.md)\nbefore the first generation or analysis. Never install anything without\nexplicit user consent; required: the `music_generate` tool, `MINIMAX_API_KEY`, `python3`, `mmx` —\nif one is missing, ask, do not degrade silently.\n\n## Free Tool Augmentation (Input Enrichment)\n\nThe OpenClaw runtime exposes several free tools (web_fetch, web_search, memory, browser) that enrich the music generation workflow. The base layer is documented in [`music-craft` → Free Tool Augmentation](../music-craft/SKILL.md#free-tool-augmentation) and [`references/free-tool-inputs.md`](../music-craft/references/free-tool-inputs.md). This section shows how they compose with MiniMax-specific features.\n\n> **v1.5.0+**: The `image` tool flow was removed (no album art / OCR / face / VLM analysis in this skill). The skill is audio-only. Web tools (`web_fetch`, `web_search`) are still available for metadata-only enrichment (artist info, genre descriptions) but never for downloading audio.\n\nFree-tool routing, blocker checks, prompt/flag lint, and MiniMax combos: [`references/free-tool-inputs.md`](references/free-tool-inputs.md).\n\n## Operating Rules\n\nSame 6-step loop as `music-craft`, with MiniMax-specific extensions:\n\n1. **Read and auto-detect** — same\n2. **Ask only the ambiguous parts** — same, plus ask if the user wants cover / mashup / standard\n3. **Translate to a production-sheet prompt** — same, but consider whether to use `mmx` flags (see [`references/mmx-flags-reference.md`](references/mmx-flags-reference.md)) instead of packing everything into the prompt\n4. **Structure the lyrics** — same, plus consider lyrics API for generation or edit (see [`references/lyrics-generation.md`](references/lyrics-generation.md))\n5. **Generate and verify** — same, plus the `music-cover` model for melody preservation\n6. **Iterate** — same, plus emotion analysis to inform the next prompt adjustment\n\nFor the full 6-step detail, see `music-craft` → Operating Rules.\n\n## Song length (`--length` is a hint, not a guarantee)\n\n`mmx music generate --length` accepts milliseconds as a **duration hint**. It is useful, but it is not precise. **Don't expect mmx to hit 3:30 exactly.** In the 2026-06-12 field run, cloud outputs ranged from 57-135% of requested duration. If you need precise length, ACE-Step is the right tool (it has `audio_duration`). If you want MiniMax's speed and the song length is flexible, mmx is fine.\n\nDetails: [`references/mmx-flags-reference.md`](references/mmx-flags-reference.md).\n\n## mmx CLI Quick Reference\n\nThe shape of a complete generation command:\n\n```bash\nmmx music generate \\\n  --prompt \"<production-sheet prompt>\" \\\n  --lyrics-file lyrics.txt \\\n  --model music-2.6 \\\n  --bpm 96 --key \"D major\" --structure \"intro-verse-chorus-verse-chorus-bridge-chorus-outro\" \\\n  --vocals \"<vocal description>\" --genre \"<genre>\" --mood \"<mood>\" --instruments \"<instruments>\" \\\n  --avoid \"<what to avoid>\" --out output.mp3\n```\n\nFull reference with all flags, examples, and model selection guidance: [`references/mmx-flags-reference.md`](references/mmx-flags-reference.md).\n\n## mmx Music Generation — verified patterns (June 2026)\n\nPattern A: full song with detailed prompt + 6 metadata flags (`--vocals`, `--genre`, `--mood`, `--instruments`, `--bpm`, `--key`) — production-grade output. Pattern B: crazy combo experiments (e.g. opera vocals over heavy metal), uses `scripts/generate_with_retry.py` wrapper.\nModel selection guidance covers `music-2.6` vs `music-2.6-free` and when to use each; `--instrumental` and `--lyrics-optimizer` flags bypass the `--lyrics` requirement for BGM and auto-lyrics workflows.\nPrompt length safety: prompts `>2000` UTF-8 bytes fail with `invalid params, prompt length not valid` — run `scripts/lint_music_request.py` before generation. Lyrics tag safety: run `scripts/lint_lyrics.py` or pass `--lyrics-file` to `lint_music_request.py`; non-whitelisted bracket tags may be sung. URL expiration: when using `--output-format url`, the returned URL has a 24h time limit — always use `--out` or download promptly.\n\nFor cloud reliability, keep standard `music generate` prompts under about 500\ncharacters when possible. Inline reference artists in the prompt instead of\nrelying on `--references` for batch runs.\n\nDetails: [`references/mmx-flags-reference.md`](references/mmx-flags-reference.md).\n\n## Cover Workflow\n\nCover workflow preserves the original song's melody while applying a different style. Two paths exist: one-step (`scripts/generate_with_retry.py -- music cover ...`, MiniMax extracts lyrics via ASR and applies the new style) and two-step (preprocess to get a `cover_feature_id`, edit ASR lyrics, then generate — better when lyrics need correction or the user wants different lyrics in the new style).\n\nFull workflow with payloads, error handling, and use cases: [`references/cover-workflow.md`](references/cover-workflow.md).\n\n## Lyrics Generation\n\nMiniMax has a dedicated `lyrics_generation` endpoint that produces structured lyrics (with `[Verse]`, `[Chorus]`, etc. tags) from a theme prompt. Two modes:\n\n- `write_full_song` — create new lyrics from a theme\n- `edit` — modify existing lyrics (e.g., make the chorus stronger, shift to a hopeful ending)\n\nThe output is structured lyrics that can be passed directly to `music_generate` or `mmx music generate`.\n\nFull detail with API examples, parameters, and use cases: [`references/lyrics-generation.md`](references/lyrics-generation.md).\n\n## Lyrics Source\n\nWhisper transcription of the local audio file is the only lyrics source\nin this skill. The LRCLib web lookup moved to the private\n`music-source-fetch` skill in v1.5.0.\n\n## Mashup Workflow\n\nThe signature MiniMax-specific feature: combine Song A (content + emotion) with Song B (style).\n\nWorkflow:\n\n1. Get Song A (local audio file or song name)\n2. Get Song B (local audio file or song name)\n3. Run emotion analysis on Song A (if audio available) to extract the emotional arc\n4. Build a prompt that applies Song B's style to Song A's content and emotion\n5. Generate using the cover workflow (preserves melody) or standard generation (creative reimagining)\n\nThis is the most powerful feature in this skill. The output preserves what makes Song A recognizable (lyrics, melody, emotion) while applying Song B's production style.\n\nFull detail with the emotion-to-prompt conversion and the two-song analysis script: [`references/mashup-workflow.md`](references/mashup-workflow.md) and [`references/emotion-analysis.md`](references/emotion-analysis.md).\n\n## Emotion Analysis\n\nEmotion analysis extracts per-section features from input audio (intensity, pitch, vocal effort, breathiness, spectral centroid, emotion classification, repetitive intensification, emotional shifts, vocal speed, pitch bends). The analysis outputs JSON that the `emotion_to_prompt.py` script converts into a ready-to-use production-sheet prompt. Run analysis first when audio is available; use the local-only path (assemble prompt from JSON without the cloud helper) when MiniMax API access is unavailable.\n\nFull detail with detection cookbook, pipeline, scripts, local-only path, and the 25+ emotion set: [`references/emotion-analysis.md`](references/emotion-analysis.md). For emotion recipes in the OUTPUT, see [`references/emotion-delivery.md`](references/emotion-delivery.md).\n\n## Analysis Quality (Summary Format, Confidence, Fallbacks)\n\nAnalysis scripts in `scripts/` produce different views (emotion, beats, melody, structure, instrumentation). The skill expects them to converge on a single compact summary so downstream code and humans can read the same shape regardless of which scripts ran.\n\n### Compact Analysis Summary\n\nEvery analysis result should include a `summary` object with these keys:\n\n| Key | Type | Meaning |\n| --- | --- | --- |\n| `tempo` | string | BPM value with confidence, e.g. `120 BPM (confidence 0.92)` |\n| `key` | string | Detected key, e.g. `E minor (confidence 0.71)` |\n| `sections` | list | Section labels with timing, e.g. `[{\"label\": \"verse\", \"start\": 0.0, \"end\": 28.5}, ...]` |\n| `instrumentation` | list | Detected instrument palette, e.g. `[\"electric guitar\", \"drums\", \"bass\"]` |\n| `vocal_traits` | dict | Breathiness, intensity, pitch range, e.g. `{\"breathiness\": \"high\", \"intensity\": \"medium\"}` |\n| `energy_curve` | list | Per-section energy values, e.g. `[{\"t\": 0, \"energy\": 0.6}, ...]` |\n| `hook_points` | list | Timestamps of detected hooks, e.g. `[12.4, 48.0]` |\n| `mix_notes` | list | Short strings, e.g. `[\"vocal upfront\", \"wide stereo drums\", \"rolled-off highs\"]` |\n\nScripts may add their own fields, but every script must return at least the keys above (use empty list / unknown string when a key has no data).\n\nConfidence levels and fallback behavior for missing optional dependencies: [`references/advanced-audio-analysis.md`](references/advanced-audio-analysis.md).\n\nFor arranger-style triage, extract stems with `scripts/extract_stems.py`, then run\n`scripts/per_stem_analysis.py <stems.json>`. Use full-mix Whisper for lyrics;\nstems are for timbre, pitch, masking, and mix decisions. If the user asks to\nchange one instrument, `scripts/hybrid_remix.py` is experimental and requires a\nvalidated transformed stem unless `--allow-stem-cover` is explicitly used for a\nsmoke test.\n\n## Rate Limits (MiniMax-specific)\n\nHard limits: 120 RPM, 20 concurrent connections, output URLs expire in 24 hours, cover feature IDs expire in 24 hours. Under Token Plan 3.0 (June 2026+), the actual ceiling is credit-based: **the documented 120 RPM is the API limit, but the Token Plan 3.0 quota is what determines your real ceiling.**\n\nFull detail, Token Plan 3.0 credit-pool mechanics, 429 recovery steps, and the usage-check command: [`references/error-handling.md`](references/error-handling.md#rate-limits-minimax-specific).\n\n## Anti-Sparse (MiniMax-Specific Deep Dive)\n\nThe base anti-sparse rules live in [`../music-craft/SKILL.md`](../music-craft/SKILL.md). MiniMax adds a more severe failure mode: **MiniMax interprets \"sparse\" or \"minimal\" as \"remove all instruments\"**, even more aggressively than other providers — never use those words in a prompt without pairing them with an explicit instrument list.\n\nFull deep-dive with observed failure modes, mitigation steps, and the canonical phrase blocklist: [`references/error-handling.md`](references/error-handling.md#anti-sparse-minimax-specific-deep-dive).\n\n## Quality Verification Checklist\n\nSame 8-point checklist as the base skill, plus 4 MiniMax-specific items:\n\n1. **Cover preserves melody recognisably.** If the user said \"make it sound like Song X\", the new version should be recognisable as Song X's melody with Song Y's style.\n2. **Emotion curve matches Song A** (for mashups). The dynamic arc of the output should follow the original's intensity, not flatten to a single energy.\n3. **`--avoid` flags are respected.** If the user said \"no electronic sounds\", the output should not have synths.\n4. **Per-flag control worked** (BPM, key, structure). If the user asked for 80 BPM in E minor, the output should be in that range, not \"close enough\".\n\n## Output Verification (Covers, Mashups, Style Transfer)\n\nAfter generation, run a post-generation check that is specific to the route. Every cover, mashup, and style-transfer output is verified against its route checklist before delivery.\n\nRoute checklists (cover / mashup / style transfer / emotion prompt), failure-signature table, and revision prompt templates: [`references/error-handling.md`](references/error-handling.md#output-verification-covers-mashups-style-transfer).\n\n## Lyrics Optimizer Behavior\n\nSame as the base skill — when `music_generate` is called without explicit lyrics, MiniMax auto-generates. With this skill, you can also call the `lyrics_generation` API directly to preview the lyrics before generation, or to iterate via the `edit` mode.\n\nIf the user wants specific words, the `lyrics_generation` API's `edit` mode lets you modify auto-generated lyrics to match the user's intent without regenerating the whole song.\n\n## Reference Map\n\n- [`references/setup-and-preflight.md`](references/setup-and-preflight.md) — extended pre-flight: platform notes, required/optional dependencies, ask-the-user pattern, local analysis memory\n- [`references/mmx-flags-reference.md`](references/mmx-flags-reference.md) — full `mmx` CLI flag reference with worked examples\n- [`references/examples.md`](references/examples.md) — practical MiniMax examples with routing, first questions, workflow shapes, and prompt/flag lint catches\n- [`references/cover-workflow.md`](references/cover-workflow.md) — one-step and two-step cover workflow with payloads, error handling, use cases\n- [`references/lyrics-generation.md`](references/lyrics-generation.md) — the `lyrics_generation` API endpoint, both modes, examples\n- [`references/mashup-workflow.md`](references/mashup-workflow.md) — two-song mashup workflow, emotion-to-prompt conversion, decision tree\n- [`references/emotion-analysis.md`](references/emotion-analysis.md) — 25+ emotion classifications + per-emotion detection cookbook + emotion combinations + the analysis pipeline\n- [`references/emotion-delivery.md`](references/emotion-delivery.md) — 21 emotion recipes for the OUTPUT + iteration loop + common mistakes\n- [`references/vocal-workflow.md`](references/vocal-workflow.md) — end-to-end vocal workflow (preflight → generation → post-processing → validation → delivery); Gate 5 lessons from the youtube-studio vocal pilot\n- [`references/orchestrator-quickstart.md`](references/orchestrator-quickstart.md) — per-input orchestrator commands (audio, two-song pairs), extraction guidance, per-song output layout\n- [`references/minimax-generation-caveats.md`](references/minimax-generation-caveats.md) — sequential-run rules, output-file verification, duration-is-a-target caveats, and delivery copy templates\n- [`references/short-prompt-recipes.md`](references/short-prompt-recipes.md) — short prompt recipes for reliable cloud iterations under about 500 characters\n- [`references/advanced-audio-analysis.md`](references/advanced-audio-analysis.md) — advanced free tools (Essentia, Demucs, Basic Pitch, Music21, CREPE) for deeper analysis when basic librosa/parselmouth is not enough\n- [`references/error-handling.md`](references/error-handling.md) — MiniMax-specific error table, recovery patterns, anti-sparse failure recovery\n- [`references/mmx-recipe-pattern.md`](references/mmx-recipe-pattern.md) — typed `MMXReceipt` wrapper pattern around `mmx` (argv, dry-run, quota snapshot, audit trail); the roadmap target for `generate_with_retry.py`\n- [`references/quota-checking.md`](references/quota-checking.md) — Token Plan Plus 5h rolling quota fundamentals, subcommand coverage, cost-per-call estimates, and live-check commands\n- [`references/music-3-migration.md`](references/music-3-migration.md) — Music 3.0 migration path (BLOCKED: `music-3.0` exists in API but `mmx` CLI v1.0.16 does not expose it; checklist for when mmx adds the model)\n- [`references/free-tool-inputs.md`](references/free-tool-inputs.md) — MiniMax layer: free-tool routing, blocker checks, and prompt/flag conflict lint before analysis\n- [`scripts/check_environment.py`](scripts/check_environment.py) — lightweight preflight diagnostic for Python, env vars, CLI tools, and optional packages\n- [`scripts/check-quota.py`](scripts/check-quota.py) — Token Plan Plus quota checker: prints summary, supports `--quiet` / `--budget N` / `--json`; verified against `mmx` CLI v1.0.16\n- [`scripts/lint_music_request.py`](scripts/lint_music_request.py) — standard-library helper for routing, blocker, missing-field, prompt, lyrics-tag, duration-density, and `mmx` flag conflict checks\n- [`scripts/lint_lyrics.py`](scripts/lint_lyrics.py) — standard-library lyrics preflight for section-tag whitelist checks and syllable/BPM duration estimates\n- [`scripts/verify_lyrics_alignment.py`](scripts/verify_lyrics_alignment.py) — standard-library post-generation transcript-vs-lyrics overlap check for semantic delivery\n- [`scripts/verify_cloud_output.sh`](scripts/verify_cloud_output.sh) — shell helper for file existence, size, MP3 probe, and expected-duration range checks after cloud generation\n- [`scripts/batch_cover.py`](scripts/batch_cover.py) — sequential batch wrapper for repeated `music cover` prompts via `generate_with_retry.py`\n- [`scripts/per_stem_analysis.py`](scripts/per_stem_analysis.py) — lightweight arranger triage from `extract_stems.py` `stems.json`\n- [`scripts/hybrid_remix.py`](scripts/hybrid_remix.py) — experimental preview remix planner; stem-to-cover is gated behind explicit smoke-test consent\n- [`scripts/smoke_test.py`](scripts/smoke_test.py) — standard-library smoke tests for pure helper behavior\n- [`scripts/`](scripts/) — Python helpers for audio analysis (download, segment, analyze, convert emotion to prompt)\n- [`music-source-fetch`](../music-source-fetch/) — private skill (NOT published) that holds the YouTube/JioSaavn/mx3.ch/LRCLib download code Luis uses to fetch source audio + lyrics by title\n- [`music-craft`](../music-craft/) — base skill with shared concepts (Pre-Flight, anti-sparse, prompt formula, structure tags, Request Intake, User Preference Flow)\n- [`music-craft` → references/free-tool-inputs.md](../music-craft/references/free-tool-inputs.md) — base layer for free tool inputs (web_fetch, web_search, image, memory)\n- [`references/changelog.md`](references/changelog.md) — release history (v1.4.1, v1.4.0, v1.3.0, v1.1.0, v1.0.0, v0.3.0); operating guidance lives in the topic references\n\n## Data, Consent, and Local Side Effects\n\nThis skill is allowed to process music media, but it should not surprise the user:\n\n- **Cloud generation:** prompts, lyrics, reference audio, and cover/mashup inputs may be sent to MiniMax through `mmx` or MiniMax API calls. Confirm consent before sending sensitive, private, or third-party-owned audio.\n- **Local files only:** audio input must be a local file path. URLs are not accepted in v1.5.0+; use the private `music-source-fetch` skill to fetch audio by title first.\n- **Lyrics:** Whisper transcription of the local audio file is the only lyrics source in this skill. LRCLib web lookup moved to `music-source-fetch`.\n- **Local outputs:** analysis JSON, lyrics, prompts, temporary media, and generated audio are written to user-selected paths or temporary directories.\n- **Overwrites:** helper scripts refuse to replace existing user-visible outputs unless the operator passes an explicit `--overwrite` flag.\n\nThere is no image, face, OCR, or VLM pipeline in this skill. Album art, video frames, and similar visual inputs are not analyzed.\n\n## Licensing and commercial-use gate\n\nClawHub publishes this skill bundle under MIT-0, so these instructions and\nbundled helper code may be used, modified, and redistributed commercially\nwithout attribution. MIT-0 does not grant a MiniMax license or transfer\nrights to any input or output. Each operator must use their own MiniMax account/API key, accept the\ncurrent applicable MiniMax terms, and verify that their specific Token Plan,\nAPI product, and region allow the intended commercial use. Do not share an\nAPI key through the skill or use the maintainer's account.\n\nThe MiniMax Open Platform terms and the consumer MiniMax app/web terms are\ndifferent products and can impose different restrictions. This skill uses the\nAPI/CLI path; do not infer API commercial rights from a web-app subscription\nor from third-party marketing claims. If the applicable product terms are\nunclear, do not release the generated track commercially until MiniMax\nconfirms the rights in writing.\n\nBefore cloud generation, the operator must own or have permission to upload\nreference audio, lyrics, samples, and voices. A paid plan does not legalize a\ncopyrighted cover, unauthorized voice, or unlicensed lyrics. AI-generated\naudio may also lack copyright protection or exclusivity under local law.\n\n**Cloud duration is approximate.** Verified field run (2026-06-12): 18/18\ncloud jobs saved an MP3, but output length ranged from 57-135% of requested\nduration (12 truncated, 3 close, 3 extended). Use `music-craft` with local\nACE-Step when exact duration matters; use this skill when speed, MiniMax cover\nworkflows, lyrics API, emotion analysis, or `mmx` flag control matter more.\n\nThis is the **power-user upgrade** of [`music-craft`](../music-craft/). It does everything that skill does, plus the features that require the MiniMax Music 2.6 token plan:\n\n- **Cover and style transfer** from a reference audio file (preserves melody)\n- **Two-song mashup** (Song A's content and emotion + Song B's style)\n- **Lyrics generation** via the MiniMax API endpoint (with edit mode for iteration)\n- **Emotion analysis** on input audio to drive prompt construction (vocal speed, intensity curve, pitch bends)\n- **Fine control** over generation parameters (BPM, key, structure, avoid list as separate flags via `mmx`)\n\nFor everything else (standard song generation, instrumentation, anti-sparse prompt engineering, structure tags, user preference flow), this skill uses the same workflow as `music-craft`. Read that skill first to understand the base, then come back here for the MiniMax-specific extensions.\n\nFile v1.6.0:README.md\n\n# Music Craft — MiniMax\n\nAdvanced music generation for OpenClaw, using the MiniMax Music 2.6 token plan.\n\nCurrent release: v1.6.0.\n\nThis is the **power-user upgrade** of [`music-craft`](../music-craft/). It does everything that skill does, plus the features that require MiniMax:\n\n- Cover and style transfer (preserves melody)\n- Two-song mashup (content + style)\n- Lyrics generation API\n- Emotion-driven prompt engineering\n- Fine control via `mmx` CLI (BPM, key, structure, avoid list as separate flags)\n- Production helpers: prompt/lyrics linting, retry wrapper, MP3 defaults, lyrics-alignment verification, and loudness finalization\n\nCloud generation is for speed and MiniMax-native workflows, not exact duration.\nThe 2026-06-12 field run saved 18/18 cloud outputs, but durations ranged from\n57-135% of requested length. Use [`music-craft`](../music-craft/) with local\nACE-Step when exact duration matters.\n\n## Data and consent\n\nThis skill may send prompts, lyrics, reference audio, and cover/mashup inputs to MiniMax when the user chooses cloud generation. Audio input must be a **local file path** — URLs are not accepted in v1.5.0+. Use the private `music-source-fetch` skill to fetch audio by title first. There is no image, face, OCR, or VLM pipeline.\n\n## Licensing and commercial use\n\nClawHub publishes this skill bundle under MIT-0, so the skill instructions and\nbundled helper code may be used, modified, and redistributed commercially\nwithout attribution. That does not grant MiniMax rights or rights to the\ninputs or outputs. Each operator must use their own MiniMax account/API key\nand accept the current terms for the exact API/Token Plan product. The API/Open Platform and\nconsumer web/app terms are separate; a subscription or third-party claim of\n“commercial use” is not enough to establish the rights for this CLI/API\nworkflow. Verify the applicable terms before commercial release. Inputs must\nbe owned or properly licensed, and generated audio is not guaranteed to have\nexclusive copyright protection in every jurisdiction.\n\n## When to use\n\nUse this skill when the task involves:\n\n- cover or style transfer from a local reference audio file (URLs not accepted; use `music-source-fetch` to fetch by title)\n- two-song mashup\n- lyrics generation via the MiniMax API\n- emotion analysis on input audio\n- fine parameter control via `mmx` CLI\n- fast standard cloud iterations where approximate duration is acceptable\n\nStay in [`music-craft`](../music-craft/) when the user only needs standard song generation, a text-only style reference without melody preservation, or provider-agnostic prompting. Use MiniMax when the request uses a local reference audio file, needs cover/mashup, lyrics API iteration, emotion-driven prompting, or exact `mmx` flags.\n\n## First Response\n\n- **Cover from local audio**: default to the one-step cover path; ask only if the user wants translated lyrics, edited ASR lyrics, or new lyrics.\n- **Style transfer only**: do not use cover unless the melody must survive. Use standard generation plus `mmx` flags when exact BPM/key/structure matter.\n- **Two-song mashup**: anchor on Song A. If Song A has audio, use the cover two-step path; if Song B is only named, ask for a short style description.\n- **Lyrics API**: use `write_full_song` for blank-page generation and `edit` for revisions.\n- **Emotion analysis to prompt**: analyze first, then convert to prompt; ask only for the output type and target language if missing.\n- **Exact `mmx` control**: let flags win for BPM, key, structure, and avoid lists.\n\nIf the user provides a URL, do not attempt to download it. Tell them this skill accepts only local files, and suggest the private `music-source-fetch` skill if they want auto-download by title.\n\n## Requirements\n\n- `MINIMAX_API_KEY` environment variable (from your MiniMax Token Plan)\n- `mmx` CLI (recommended for fine control)\n- `ffmpeg` and `librosa` / `parselmouth` (recommended for emotion analysis and cover preprocessing)\n- `python3` (required for the analysis scripts)\n\n## Platform support\n\nThis workflow is intended to work best on macOS and Linux, where `python3`, `ffmpeg`, and the optional audio/ML packages are easiest to install and run. Windows should be treated as a partial-support target: the concepts still apply, but command examples may need PowerShell equivalents, paths may need adjustment, and some optional audio dependencies may require extra porting or manual setup.\n\nSee the **Pre-Flight Check** in [`SKILL.md`](SKILL.md) for the full list and the platform-aware install commands.\n\nRun `python3 scripts/check_environment.py` and `python3 scripts/smoke_test.py` on macOS/Linux. On Windows PowerShell, use `python scripts/check_environment.py` or `py -3 scripts/check_environment.py`, and the same interpreter form for `scripts/smoke_test.py`.\n\nFor generated files, verify by file existence and duration, not exit code alone:\n`scripts/verify_cloud_output.sh output.mp3 --expected-duration 180`.\nFor `mmx` generation through the retry wrapper, always pass `--out` inside the\n`mmx` command; `--output-path` is only the wrapper's preservation/verification\ndestination.\n\n## Quickstart\n\nThe skill is auto-loaded only when the request explicitly asks for: cover or style transfer from a local audio file, two-song mashup, lyrics generation via the MiniMax API, emotion-curve analysis on input audio, or direct `mmx` flag control. Generic music generation requests stay in `music-craft`. Addresses audit SQP-1.\n\n> \"Take this Beyoncé track and turn it into a reggaeton version.\"\n\n> \"I want a mashup of 'Bohemian Rhapsody' style + 'La Vie en Rose' lyrics.\"\n\n> \"Generate lyrics in French about lost love, melancholic, 80s chanson style.\"\n\n> \"Analyze the emotion curve of this audio file and write a prompt that captures the arc.\"\n\nThe skill runs the extended pre-flight, detects input type, asks 1–3 questions for the ambiguous parts, and generates.\n\n## Relationship to the base skill\n\nThis skill extends [`music-craft`](../music-craft/). Read that skill first to understand the shared concepts (Pre-Flight, anti-sparse, prompt formula, structure tags, user preference flow). The MiniMax-specific additions live here:\n\n- `mmx` CLI quick reference\n- Cover workflow (one-step, two-step)\n- Lyrics generation API\n- Mashup workflow\n- Emotion analysis\n\n## Reference\n\nFor full details, see [`SKILL.md`](SKILL.md).\n\nFor practical routing examples, see [`references/examples.md`](references/examples.md).\n\nFile v1.6.0:_meta.json\n\n{\n  \"ownerId\": \"kn7fkh362pnj2zckq8pcxfaxr9821zv8\",\n  \"slug\": \"music-craft-minimax\",\n  \"version\": \"1.6.0\",\n  \"publishedAt\": 1785411554548\n}\n\nFile v1.6.0:references/advanced-audio-analysis.md\n\n# Advanced Audio Analysis with Free Tools\n\nThe base emotion analysis in [`emotion-analysis.md`](emotion-analysis.md) uses `librosa` + `parselmouth` (Praat) + `scipy` to extract per-section features. Those cover the core 30 emotions. This file covers **advanced free tools** that go deeper: more accurate pitch, source separation, note transcription, music theory, and pre-trained emotion / genre / mood classifiers.\n\nAll tools listed here are **open source** (BSD, MIT, Apache, or GPL). All can be installed via `pip` (some need system dependencies like `ffmpeg`). They are **optional** — the base skill works without them. Install only what you need.\n\n---\n\n## Currently Implemented Advanced Features (v0.1.0)\n\nThe following advanced analysis features **are now implemented** in the MiniMax layer. Each is available as an optional dependency and is used in the prompt pipeline to enrich generation prompts with precise audio characteristics.\n\n| Feature | Library | What it detects | Pip package |\n|---|---|---|---|\n| **tempogram_ratio** | librosa | Swing / groove feel ( eighth-note swing ratio ) | `librosa` (already in base) |\n| **tonnetz** | librosa | Tonal harmony quality — diatonic/chromatic, consonant/dissonant | `librosa` (already in base) |\n| **spectral_contrast** | librosa | Valley-to-peak energy ratio per frequency band (timbre) | `librosa` (already in base) |\n| **HPSS** | librosa.effects.hpss | Harmonic-percussive source separation | `librosa` (already in base) |\n| **Formant analysis** | parselmouth | Vocal register detection — chest / head / falsetto | `parselmouth` (already in base) |\n| **HNR** | parselmouth | Harmonics-to-noise ratio — breathiness vs clarity | `parselmouth` (already in base) |\n| **Jitter / Shimmer** | parselmouth | Microperturbations in pitch and amplitude — voice stability | `parselmouth` (already in base) |\n| **LUFS / LRA** | pyloudnorm | Perceptual loudness (integrated LUFS) and loudness range (LRA) | `pyloudnorm` |\n| **Chord progression** | autochord | Automatic chord symbol extraction | `autochord` |\n| **Song structure** | allin1 | Neural structure segmentation — intro / verse / chorus / bridge / outro | `allin1` |\n| **CLAP classification** | transformers | Zero-shot genre / mood / instrument / era classification | `transformers` |\n\n### tempogram_ratio (librosa)\n\nDetects swing and groove by measuring the deviation of note onsets from a perfectly regular grid. A ratio near 1.0 is straight; above ~1.2 indicates a noticeable swing feel.\n\n**Pipeline use**: When the user wants \"groovy\" or \"laid-back\" output, `tempogram_ratio` quantifies the swing intensity to pass a precise BPM/swing hint to `mmx`.\n\n```python\nimport librosa\n\ny, sr = librosa.load('/tmp/song.wav')\ntempo, beats = librosa.beat.beat_track(y=y, sr=sr)\noenv = librosa.onset.onset_strength(y=y, sr=sr, feature_mode='spectral')\ntempogram = librosa.feature.tempogram(oenv=oenv, sr=sr)\n# Ratio of beat-strength deviations indicates swing\n```\n\n### tonnetz (librosa)\n\nComputes the tonal centroid features (Tonnetz) representing harmonic relationships in the circle-of-fifths space. High diatonic weight = consonant; high chromatic weight = more tension/dissonance.\n\n**Pipeline use**: Maps directly to \"consonant/dissonant\", \"epic/tense\", or \"warm/dark\" style cues in the generation prompt.\n\n```python\ny, sr = librosa.load('/tmp/song.wav')\nchrom = librosa.feature.chroma_cqt(y=y, sr=sr)\ntonnetz = librosa.feature.tonnetz(chroma=chrom)\n# tonnetz shape: (6, frames) — dims represent 5ths, minor/major\n```\n\n### spectral_contrast (librosa)\n\nMeasures the energy ratio between spectral peaks (high energy frequency bands) and valleys (low energy bands) across octave bands. High contrast = bright/drill; low contrast = dark/mellow.\n\n**Pipeline use**: Feeds the \"bright/dark\" and \"warm/cold\" timbre dimension in prompt construction.\n\n```python\ny, sr = librosa.load('/tmp/song.wav')\ncontrast = librosa.feature.spectral_contrast(y=y, sr=sr, n_bands=6)\n# contrast shape: (7, frames) — 6 bands + overall valley mean\n```\n\n### HPSS — Harmonic-Percussive Separation (librosa.effects.hpss)\n\nDecomposes audio into a harmonic component (melody/harmony) and a percussive component (drums/transients). Useful for isolating the vocal melody from rhythmic masking.\n\n**Pipeline use**: Before running pitch or emotion analysis, HPSS can clean the vocal to reduce rhythmic interference.\n\n```python\ny, sr = librosa.load('/tmp/song.wav')\nharmonic, percussive = librosa.effects.hpss(y=y)\n```\n\n### Formant analysis (parselmouth)\n\nExtracts formant frequencies (F1, F2, F3) and their trajectories to detect vocal register — chest voice (low F1, high F2), head voice (high F1, low F2), falsetto (very high F1, very low F2).\n\n**Pipeline use**: Feeds vocal register cues (\"chesty belt\", \"head voice mix\", \"falsetto float\") to `mmx --vocal-style`.\n\n```python\nimport parselmouth\n\nsound = parselmouth.Sound('/tmp/vocal.wav')\nformants = sound.to_formant_burg()\n# F1/F2 at voiced segments indicate register\n```\n\n### HNR — Harmonics-to-Noise Ratio (parselmouth)\n\nRatio of harmonic energy to noise energy. High HNR = clear/clean voice; low HNR = breathy/aspirate. Range: 0–30 dB typical.\n\n**Pipeline use**: Maps to \"breathy\" vs \"focused\" vocal quality descriptors in the prompt.\n\n```python\nimport parselmouth\n\nsound = parselmouth.Sound('/tmp/vocal.wav')\nhnr = sound.to_harmonicity()\n```\n\n### Jitter and Shimmer (parselmouth)\n\n- **Jitter**: cycle-to-cycle variation in fundamental frequency (pitch instability)\n- **Shimmer**: cycle-to-cycle variation in amplitude (volume instability)\n\nBoth measure voice quality stability. High jitter/shimmer = unstable, tense, or emotional voice; low = steady, controlled.\n\n**Pipeline use**: Contributes to \"controlled/emotional\" and \"steady/wavering\" vocal quality cues.\n\n```python\nimport parselmouth\n\nsound = parselmouth.Sound('/tmp/vocal.wav')\npoint_process = sound.to_point_process()\njitter = point_process.get_jitter()\nshimmer = point_process.get_shimmer()\n```\n\n### LUFS and LRA (pyloudnorm)\n\n- **Integrated LUFS**: overall perceived loudness (EBU R128 standard)\n- **LRA (Loudness Range)**: dynamic range in LUFS — difference between loud and quiet sections\n\n**Pipeline use**: LUFS informs the \"loudness feel\" (quiet/quiet, loud/dynamic); LRA informs the \"dynamic range\" descriptor in the prompt (e.g., \"wide LRA — lots of contrast between verse and chorus\").\n\n```python\nimport pyloudnorm as pyln\n\nmeter = pyln.Meter(44100)  # create BS.1770 meter\nloudness = meter.integrated_loudness('/tmp/song.wav')\n# For LRA, use pyloudnorm's full workflow with anchored loudness\n```\n\n### Chord progression (autochord)\n\nAutomatically detects chord symbols (roman numerals and chord types) from audio using a pre-trained CNN/RNN model.\n\n**Pipeline use**: Extracts \"I-V-vi-IV\" or \"i-bVII-bVI-bVII\" style progressions to include as harmonic context in the prompt.\n\n```python\nimport autochord\n\nchords = autochord.recognize('/tmp/song.wav')\n# Returns list of (timestamp, chord_symbol)\n```\n\n### Song structure (allin1)\n\nNeural network-based structural segmentation. Detects: intro, verse, pre-chorus, chorus, bridge, outro. Also returns segment boundaries with confidence scores.\n\n**Pipeline use**: Feeds the `mmx --structure` flag with accurate segment timing instead of generic defaults.\n\n```python\nfrom allin1 import StructuralAnalysis\n\nanalyzer = StructuralAnalysis()\nsegments = analyzer.analyze('/tmp/song.wav')\n# Returns: [{'label': 'chorus', 'start': 15.2, 'end': 30.1, 'confidence': 0.94}, ...]\n```\n\n### CLAP classification (transformers)\n\nZero-shot audio classification using the CLAP (Contrastive Language-Audio Pretraining) model. Can classify genre, mood, instrument, and era without training data for the specific labels.\n\n**Pipeline use**: Provides genre/mood tags as soft constraints for the prompt builder (\"80s pop\", \"melancholic indie folk\", \"aggressive metal\").\n\n```python\nfrom transformers import pipeline\nimport torch\n\nclap = pipeline(\"audio-classification\", model=\"laion/clap-htsat-unfused\")\n# Zero-shot labels\nresults = clap('/tmp/song.wav', candidate_labels=[\"pop\", \"rock\", \"jazz\", \"classical\", \"hip-hop\"])\n```\n\n---\n\n## Future / Roadmap\n\nThe following tools are **roadmap items** for upcoming phases of the analysis pipeline.\n\n| Tool | License | Phase | What it adds | When to use |\n|---|---|---|---|---|\n| **Demucs** | MIT | Phase 2 | Source separation: vocals / drums / bass / other | When you need to analyze the vocal track WITHOUT the accompaniment masking it (huge for emotion analysis). Opt-in via `--use-demucs` flag on the orchestrator. |\n| **Basic Pitch** | Apache-2.0 | Phase 3.2 | Convert audio to MIDI notes | When you need the melody as MIDI for comparison (e.g., cover similarity) or MIDI-confirmed key/scale detection. |\n| **beat_this** | MIT | Phase 3.1 | ISMIR 2024 SOTA beat + downbeat tracker (transformer-based) | Replaces `librosa.beat.beat_track` for more accurate downbeats. Used for prompt-level timing cues. |\n| **MERT v1-330M** | Apache-2.0 | Phase 3.3 | Music SSL embeddings, complementary to CLAP | Use for \"vibe similarity\" scoring between two songs (mashup). |\n| **PANNs / AST** | MIT | Phase 3.4 | 527-class instrument classification via AudioSet | Second opinion to CLAP for instrument detection (oud, sitar, bouzouki, etc.). |\n| **OpenCLIP** | MIT | Phase 4.1 | Image zero-shot style classification | Album art → \"this is synthwave / indie / metal\" classification. |\n| **RapidOCR** | Apache-2.0 | Phase 4.2 | On-device OCR via ONNX | Album-art text / on-screen video subtitles. |\n| **DeepFace / MediaPipe** | MIT / Apache-2.0 | Phase 4.3 | Face detection + emotion | Album art with a face → \"warm, joyful\" mood. |\n| **Qwen3-VL** | Apache-2.0 | Phase 4.5 / 5.4 | Multi-modal vision-language model | Image / video captioning (free-form description). GPU-only, opt-in. |\n| **InternVideo2** | Apache-2.0 | Phase 5.5 | Video action recognition | Music video → \"concert / dance / performance\" classification. |\n\n### Already implemented (v0.1.0+)\n\n- **Whisper** (MIT, `extract_lyrics_whisper.py`) — lyrics extraction with section tagging. Auto-wired into the orchestrator via `--lyrics` flag.\n\n### Will NOT be added (license / maintenance)\n\n| Tool | Why not |\n|---|---|\n| **Essentia** (AGPL-3.0) | AGPL is not compatible with clean skill distribution. |\n| **YOLO Ultralytics** (AGPL-3.0) | Same. Use OpenCLIP / MediaPipe instead. |\n| **ImageBind** (CC-BY-NC 4.0) | Non-commercial weights; not safe for distributed skills. |\n| **audiocraft / MusicGen weights** (CC-BY-NC) | Non-commercial. EnCodec encoder code (MIT) is fine to use. |\n| **madmom** | Unmaintained since 2019; `beat_this` is the replacement. |\n| **Spleeter** | Unmaintained since 2020; Demucs is the replacement. |\n\n## Essentia\n\nEssentia is the most comprehensive open source audio analysis library. It has pre-trained classifiers for mood, genre, and danceability.\n\n### Install\n\n```bash\n# Ubuntu / Debian (Essentia needs ffmpeg and other system deps)\napt install ffmpeg libfftw3-dev libyaml-dev libsamplerate0-dev libtag1-dev\npip install essentia\n\n# macOS\nbrew install pkg-config ffmpeg libyaml libsamplerate taglib\npip install essentia\n```\n\n### What it gives you\n\nEssentia provides hundreds of audio features, but the most useful for emotion analysis are:\n\n| Feature | What it measures | Use |\n|---|---|---|\n| `danceability` | 0–1, how suitable the track is for dancing | Energy + rhythm regularity |\n| `arousal` | 0–1, intensity of emotion (low = calm, high = excited) | Maps to overall energy |\n| `valence` | 0–1, positivity (low = sad, high = happy) | Maps to joy / melancholy |\n| `mood_classifier` | Pre-trained, outputs `happy` / `sad` / `aggressive` / `relaxed` | Direct emotion label |\n| `genre_classifier` | Pre-trained on 8 genres | Genre inference |\n| `key` and `scale` | Detected key and mode (major / minor) | Maps to mood |\n| `bpm` | Tempo | Direct BPM |\n| `loudness` | Integrated loudness in LUFS | Production quality |\n\n### Usage example\n\n```python\nimport essentia.standard as es\n\n# Load audio\naudio = es.MonoLoader(filename='/tmp/song.wav', sampleRate=44100)()\n\n# Mood classification\nmood_classifier = es.MusicExtractor()(audio)\nprint(f\"Valence: {mood_classifier['valence']}\")\nprint(f\"Arousal: {mood_classifier['arousal']}\")\nprint(f\"Danceability: {mood_classifier['danceability']}\")\nprint(f\"BPM: {mood_classifier['bpm']}\")\nprint(f\"Key: {mood_classifier['key_key']} {mood_classifier['key_scale']}\")\n```\n\nThe pre-trained mood classifier outputs one of: `happy`, `sad`, `aggressive`, `relaxed`. Map these to the 25+ emotion set in [`emotion-analysis.md`](emotion-analysis.md):\n\n| Essentia mood | Maps to |\n|---|---|\n| `happy` | joyful, triumphant, celebratory, confident |\n| `sad` | melancholic, yearning, vulnerable, lonely, tragic |\n| `aggressive` | angry, defiant, desperate, anxious |\n| `relaxed` | serene, tender, nostalgic, calm |\n\n### When to use\n\n- You want a quick automated mood label (saves LLM inference)\n- You're processing many tracks and need a consistent classifier\n- You want a confidence score for the mood (Essentia gives probabilities)\n- You want a valence-arousal 2D space mapping (useful for \"mood is between sad and angry\")\n\n### Limitations\n\n- License is AGPL-3 — if you distribute the skill, you may need to comply. For personal use, fine.\n- The mood classifier is trained on Western pop/rock/electronic. Less accurate on jazz, classical, world music.\n- Valence / arousal is a 2D reduction — misses nuances (e.g., \"angry\" and \"joyful\" can both have high arousal but very different valence).\n\n## Demucs\n\nDemucs (by Facebook Research) separates audio into stems: vocals, drums, bass, and other. This is huge for emotion analysis because the vocals carry the emotion, but the accompaniment masks it.\n\n### Install\n\n```bash\npip install demucs\n# First run downloads the pre-trained model (~2 GB)\n```\n\n### What it gives you\n\nFour stems: `vocals.wav`, `drums.wav`, `bass.wav`, `other.wav` (instruments other than drums/bass, like guitar, keys, synths).\n\nBy separating vocals from accompaniment, you can:\n\n- Run emotion analysis on the isolated vocal track (cleaner features, no masking)\n- Run a separate analysis on the drums (for rhythm / energy features)\n- Compare vocal emotion vs accompaniment energy (a song can have sad vocals over a happy beat = bittersweet)\n\n### Usage example\n\n```bash\n# Separate stems (writes to /tmp/separated/htdemucs/<song_name>/)\npython3 -m demucs --two-stems vocals /tmp/song.wav\n\n# Output:\n# /tmp/separated/htdemucs/song/vocals.wav   ← clean vocal\n# /tmp/separated/htdemucs/song/no_vocals.wav ← accompaniment only\n```\n\nFor four-stem separation:\n\n```bash\npython3 -m demucs /tmp/song.wav\n# /tmp/separated/htdemucs/song/{vocals,drums,bass,other}.wav\n```\n\nThen run emotion analysis on the isolated vocal:\n\n```bash\npython3 scripts/analyze_vocal_emotion.py \\\n  /tmp/separated/htdemucs/song/vocals.wav \\\n  --output /tmp/vocal_emotion.json\n```\n\n### When to use\n\n- The input is a full mix and emotion analysis on the mix is inaccurate\n- You want to detect \"sad vocals over happy beat\" (bittersweet) — analyze each separately\n- You want to extract just the vocal for use as a reference (e.g., to sing-along)\n- You're building a karaoke version of the song\n\n### Performance note\n\nDemucs is slow on CPU (1–5 minutes per song). On GPU, much faster (10–30 seconds). It also downloads a 2GB model on first use.\n\nFor a small VPS, prefer the `--two-stems vocals` mode (faster, smaller model) over the full 4-stem separation.\n\n## Basic Pitch\n\nBasic Pitch (by Spotify) converts audio to MIDI notes. Useful for melody comparison (e.g., how similar is the input's melody to the output).\n\n### Install\n\n```bash\npip install basic-pitch\n# Or for GPU:\npip install basic-pitch[onnx]\n```\n\n### What it gives you\n\n- Note events with onset, offset, pitch, confidence\n- MIDI file output\n- Useful for: melody comparison, cover analysis, harmonic analysis\n\n### Usage example\n\n```python\nfrom basic_pitch import ICASSP_2021_MODEL_PATH\nfrom basic_pitch.inference import predict as bp_predict\nfrom basic_pitch import build_output\n\nmodel_output, midi_data, note_events = bp_predict(\n    '/tmp/song.wav',\n    onset_threshold=0.5,\n    frame_threshold=0.3,\n    minimum_note_length=50,\n    minimum_frequency=80,\n    maximum_frequency=1000\n)\n\n# Save as MIDI\nbuild_output.write_midi(\n    note_events,\n    '/tmp/song.midi'\n)\n\n# note_events is a list of (start_time, end_time, pitch_midi, velocity, [pitch_bend])\n```\n\n### When to use\n\n- You want to compare melodies between the input and the output (cover accuracy)\n- You want to extract the input's melody to use in a new composition\n- You're building music theory analysis (chord progressions, scales)\n- You want a MIDI representation of an audio input for use in a DAW\n\n## Music21\n\nMusic21 is a Python toolkit for music theory analysis. It can detect key, chord progressions, scales, and intervals.\n\n### Install\n\n```bash\npip install music21\n# Music21 uses LilyPond for rendering (optional, only if you want visual output)\n# It also has a corpus of public-domain scores\n```\n\n### What it gives you\n\n- Key detection (more sophisticated than chroma-based)\n- Chord progression extraction\n- Roman numeral analysis (I, IV, V, vi, etc.)\n- Scale / mode detection (major, minor, dorian, phrygian, ...)\n- Interval analysis\n- Score comparison\n\n### Usage example\n\n```python\nfrom music21 import converter, analysis\n\n# Convert audio to MIDI first (via basic_pitch)\n# Then load the MIDI\nscore = converter.parse('/tmp/song.midi')\n\n# Key analysis\nkey = score.analyze('key')\nprint(f\"Key: {key.tonic} {key.mode}\")\n\n# Chord analysis (if the score has chords)\nfor chord in score.chordify():\n    print(f\"Chord at {chord.offset}: {chord.pitches}\")\n```\n\n### When to use\n\n- You want to know the exact chord progression of the input\n- You want to suggest a chord progression for the output (e.g., vi-IV-I-V in C major)\n- You're analyzing a piece in a non-Western scale (music21 supports many)\n- You want to compare two pieces' harmonic content\n\n### Limitations\n\n- Music21 works best with symbolic input (MIDI, MusicXML). Converting audio to MIDI first (via Basic Pitch) loses some fidelity.\n- Chord detection on converted MIDI is approximate — drums and non-pitched instruments confuse it.\n\n## CREPE\n\nCREPE is a monophonic pitch detection model (CNN-based). More accurate than librosa's `pyin` for vocals but requires GPU for fast inference.\n\n### Install\n\n```bash\npip install crepe\n# Or for the latest version:\npip install git+https://github.com/marl/crepe.git\n```\n\n### What it gives you\n\n- Frame-level pitch detection (every 10ms) with high accuracy\n- Confidence per frame\n- Useful for: detailed pitch contour analysis, vibrato detection, pitch bend precision\n\n### Usage example\n\n```python\nimport crepe\nfrom scipy.io import wavfile\n\nsr, audio = wavfile.read('/tmp/song.wav')\ntime, frequency, confidence, activation = crepe.predict(\n    audio, sr, viterbi=True, model_capacity='full'\n)\n```\n\n### When to use\n\n- You need very accurate pitch detection (e.g., for detailed vibrato analysis)\n- The vocal is monophonic (one singer at a time)\n- You're OK with the GPU requirement (CPU is too slow for full songs)\n\nFor most use cases, parselmouth (already in the base skill) is enough. CREPE is the next level when you need frame-level precision.\n\n## Whisper (already in OpenClaw)\n\nWhisper is OpenAI's speech-to-text model. The OpenClaw runtime has it (per TOOLS.md).\n\n### Usage example\n\n```bash\nwhisper /tmp/song.wav --language en --model medium --output_format srt\n# Output: /tmp/song.srt with timestamps\n```\n\nOr use the OpenAI API directly for programmatic access.\n\n### When to use\n\n- You need lyrics extraction from audio (for cover workflow ASR)\n- The cover workflow's two-step path uses ASR-extracted lyrics (correctable in the second step)\n- The user provides audio but no lyrics (extract the lyrics, then use as Input Type 2)\n\nWhisper is already in the runtime, so no install needed.\n\n## Integration Patterns\n\n### Pattern 1: Clean vocal emotion analysis\n\n```bash\n# Step 1: Separate vocals (Demucs)\npython3 -m demucs --two-stems vocals /tmp/song.wav\n\n# Step 2: Analyze clean vocal\npython3 scripts/analyze_vocal_emotion.py \\\n  /tmp/separated/htdemucs/song/vocals.wav \\\n  --output /tmp/vocal_emotion.json\n\n# Step 3: Optional — analyze accompaniment for energy\npython3 scripts/analyze_audio.py \\\n  /tmp/separated/htdemucs/song/no_vocals.wav \\\n  > /tmp/accompaniment.json\n\n# Step 4: Compare vocal emotion vs accompaniment energy\n# (LLM: \"vocal_emotion = melancholic, accompaniment = high energy → bittersweet\")\n```\n\n### Pattern 2: Mood + genre classification\n\n```python\nimport essentia.standard as es\n\naudio = es.MonoLoader(filename='/tmp/song.wav', sampleRate=44100)()\nfeatures, _ = es.MusicExtractor()(audio)\n\nmood = features['highlevel.mood_classifier']['value']\nvalence = features['valence']\narousal = features['arousal']\ngenre = features['highlevel.genre_classifier']['value']\n```\n\n### Pattern 3: MIDI + theory for harmonic context\n\n```bash\n# Step 1: Convert audio to MIDI\npython3 -c \"from basic_pitch.inference import predict; predict('/tmp/song.wav')\"\n\n# Step 2: Analyze the MIDI\npython3 -c \"\nfrom music21 import converter\nscore = converter.parse('/tmp/song.midi')\nkey = score.analyze('key')\nprint(f'Key: {key.tonic} {key.mode}')\n\"\n```\n\n### Pattern 4: Cover accuracy via pitch comparison\n\n```python\n# Step 1: Extract melody from input\nfrom basic_pitch.inference import predict\n_, _, input_notes = predict('/tmp/input.wav')\n\n# Step 2: Extract melody from output\n_, _, output_notes = predict('/tmp/output.wav')\n\n# Step 3: Compare (custom logic — note overlap, timing offset, etc.)\n# (LLM does the high-level comparison)\n```\n\n## When to Use Which Tool\n\n| Goal | Best tool combo |\n|---|---|\n| Quick mood label (without LLM) | Essentia |\n| Clean vocal emotion analysis | Demucs + librosa/parselmouth |\n| Melody comparison (cover accuracy) | Basic Pitch + custom comparison |\n| Chord progression extraction | Basic Pitch + Music21 |\n| Very accurate pitch tracking | CREPE (GPU) or parselmouth (CPU) |\n| Lyrics extraction from audio | Whisper (already in runtime) |\n| Source separation (for karaoke, etc.) | Demucs |\n| Genre classification | Essentia |\n\n## When NOT to Use These\n\nThese tools are enhancements, not requirements. Skip them if:\n\n- The base analysis (librosa + parselmouth) is enough\n- You can't install new dependencies\n- You're working with very short clips (< 5 seconds) — the analysis is unreliable\n- You don't have GPU and the tool is GPU-only (CREPE)\n\n## Installation Best Practices\n\nFor the Pre-Flight Check, list the new tools as OPTIONAL with the right install commands. The full per-platform table:\n\n| Tool | Linux (apt) | macOS (brew) | Windows (winget) | Pip alternative |\n|---|---|---|---|---|\n| Essentia | `apt install ffmpeg libfftw3-dev libyaml-dev libsamplerate0-dev libtag1-dev && pip install essentia` | `brew install ffmpeg libyaml libsamplerate taglib && pip install essentia` | (use conda or WSL) | — |\n| Demucs | (system deps via pip) | (system deps via brew) | (use WSL) | `pip install demucs` |\n| Basic Pitch | (system deps via pip) | (system deps via brew) | (use WSL) | `pip install basic-pitch` |\n| Music21 | `pip install music21` | `pip install music21` | `pip install music21` | `pip install music21` |\n| CREPE | (needs TensorFlow, complex) | (needs TensorFlow) | (needs WSL) | `pip install crepe` |\n\nFor the most common case (Ubuntu/Debian), the install command is:\n\n```bash\n# Install all advanced tools\napt install ffmpeg libfftw3-dev libyaml-dev libsamplerate0-dev libtag1-dev\npip install essentia demucs basic-pitch music21\n```\n\nThis adds ~5 minutes to the install time and ~3GB of disk space (mostly Demucs models).\n\n## License and Distribution Notes\n\n- **Essentia** is AGPL-3. If you distribute the skill with Essentia integrated, the AGPL may apply. For personal use, fine.\n- **Demucs, Basic Pitch, Music21, CREPE** are MIT / Apache / BSD — permissive, no copyleft concerns.\n- **Whisper** is MIT — fine.\n\nFor ClawHub distribution, document the dependencies clearly. If AGPL is a concern, Essentia can be moved to \"optional, off by default\" and the user enables it explicitly.\n\n## Confidence Levels\n\nThe following conventions apply to all analysis scripts, including the optional tools documented above.\n\nEvery numeric or categorical detection in the analysis must carry a confidence value so weak detections do not get treated as facts.\n\n| Confidence | Numeric range | Interpretation |\n|---|---|---|\n| `clear` | n/a | The detection is unambiguous (e.g. user-supplied text, MIDI-confirmed key). |\n| `high` | `>= 0.75` | Strong evidence from multiple sources or models. |\n| `medium` | `0.5 - 0.74` | Reasonable evidence but alternative interpretations exist. |\n| `low` | `< 0.5` | Weak signal; treat as a hint, not a fact. |\n| `inferred` | n/a | Not measured directly; derived from context (e.g. lyrics from a YouTube URL). |\n| `missing` | n/a | Not available; the analysis did not run or did not find evidence. |\n\nWhen feeding analysis into a prompt, prefix any `low` or `medium` detection with a hedge like \"around\" or \"approximately\", and never include `missing` values as if they were facts.\n\n## Fallback Behavior for Missing Optional Dependencies\n\nThe advanced analysis scripts depend on optional packages (librosa, parselmouth, transformers, demucs, beat_this, basic_pitch, etc.). Each script must:\n\n1. Try to import the optional dependency at the top of the function.\n2. On `ImportError`, return a JSON object that includes `{\"error\": \"install with pip install X\", \"summary\": {}}` instead of raising.\n3. Never let a missing optional dependency crash the whole workflow.\n\nThe orchestrator at [`../scripts/analysis_orchestrator.py`](../scripts/analysis_orchestrator.py) collects per-script results and continues even if some scripts failed. The combined `summary` simply omits keys whose underlying analysis could not run. The linter, prompt builder, and generation step all read the summary and skip missing keys without erroring.\n\nThis means a user without `demucs` installed can still get tempo, key, and structure analysis from the base pipeline. The only loss is the per-stem vocal analysis, which is opt-in via `--use-demucs`.\n\nFile v1.6.0:references/changelog.md\n\n# Changelog\n\nRelease history for music-craft-minimax. Operating guidance lives in the\ntopic references; this file is history only.\n# v1.6.0\n\n- Reordered `SKILL.md` for discoverability: added hero sections\n  (\"What is Music Craft — MiniMax?\", \"MiniMax-specific features\",\n  \"Why use this instead of music-craft?\", \"Quick Start\") at the top\n  of the body. MiniMax-only features (cover, mashup, lyrics API,\n  emotion-driven prompting, `mmx` flag control, quota-aware batches)\n  are now the first thing a visitor reads.\n- Moved the licensing and commercial-use gate to the end of `SKILL.md`\n  (preserving all legal content verbatim, including the \"Cloud duration\n  is approximate\" routing warning); data / consent sits just above it.\n- Updated the frontmatter `description:` to anchor-keyword pack\n  (cover and style transfer, two-song mashup, lyrics generation API,\n  emotion-driven prompt engineering, mmx CLI, Extends music-craft)\n  for clearer ClawHub routing.\n- Removed the old \"Quick Start with the Orchestrator\" section; the\n  new hero \"Quick Start\" supersedes it.\n- No behaviour, no env vars, no bins — pure content reordering and\n  versioning.\n\n\n\n# v1.5.1\n\n- Added a licensing and commercial-use gate for MiniMax API/CLI workflows.\n- Clarified the difference between MiniMax Open Platform/API terms and\n  consumer app/web terms.\n- Required each operator to use their own account/API key and verify current\n  product-specific commercial-use terms before release.\n\n## v1.5.0\n\nv1.5.0 is a **breaking change** that isolates all internet-download code\ninto the new private `music-source-fetch` skill and removes the album-art\n/ face / OCR / VLM image pipeline entirely. The published skill is now\naudio-only and accepts only local file paths.\n\n**Removed (moved to `publish/music-source-fetch/`):**\n- `scripts/download_youtube.py`, `scripts/download_mx3.py`,\n  `scripts/fetch_lyrics_web.py`, `scripts/audio_sources.py`\n- `analysis_orchestrator.py` flags: `--youtube`, `--audio-url`,\n  `--lyrics-source {web,auto}`\n- LRCLib web lyrics lookup; Whisper on the local file is the only lyrics\n  source\n\n**Deleted entirely (no replacement):**\n- `scripts/analyze_image.py`, `scripts/extract_video_features.py`\n- `analysis_orchestrator.py` flags: `--image`, `--video`, `--vlm`,\n  `--ocr`, `--faces`\n- Album-art color palette, face detection, OCR, VLM captioning flows\n\n**Changed:**\n- `check_environment.py` no longer recommends `yt-dlp`; drops `cv2`/`PIL`\n  from optional imports\n- `lint_music_request.py` URL detection now emits a `url_not_accepted`\n  warning and routes to `needs_clarification`; only local file paths\n  route to `minimax_cover`\n- SKILL.md \"Audio Source Fallback Order\" replaced with \"Audio Source\n  (Local Only)\"\n- `cover-workflow.md` and `mashup-workflow.md` add explicit cloud\n  transmission consent paragraphs (audit SQP-2)\n- `examples.md` neutralises hardcoded language/locale defaults (audit SQP-3)\n- README narrows the auto-load trigger (audit SQP-1)\n- Frontmatter `metadata.openclaw.requires.bins` drops `yt-dlp`\n\n**Audit findings resolved by this release:**\n- SDI-2 (image OCR/face/VLM pipeline)\n- SDI-2 (LRCLib lyrics retrieval copyright exposure)\n- SDI-2 (VLM captioning external transmission)\n- SQP-2 (LRCLib lookup consent)\n- SQP-2 (YouTube/JioSaavn/mx3 download copyright exposure)\n- AST4 (`mmx vision describe` subprocess)\n- AST4 (VLM keyframe captioning subprocess)\n- SQP-3 (hardcoded English/Spanish/Portuguese defaults in examples)\n- SQP-1 (auto-load trigger too broad)\n- Partial: SQP-2 (cover/mashup cloud transmission — sharpened consent text)\n\n**Audit findings NOT resolved (intentional — core MiniMax cloud feature):**\n- E1 (`api.minimax.io` external transmission)\n- AST4 (`generate_with_retry.py` subprocess for `mmx`)\n- SQP-2 (MiniMax cover preprocess cloud transmission — core feature)\n\n## v1.4.1\n\nv1.4.1 adds lyrics safety checks from field feedback without changing the\noverall routing model.\n\n**Scripts and validation:**\n- New `scripts/lint_lyrics.py` in both music skills validates section tags\n  against the canonical whitelist and estimates lyric pacing from syllables,\n  BPM, and target duration.\n- `scripts/lint_music_request.py` accepts `--lyrics-file`, surfaces invalid\n  lyrics tags as blockers, reports a duration-density estimate, and no longer\n  treats a plain prompt mention of \"lyrics\" as user-provided lyrics.\n- New `scripts/verify_lyrics_alignment.py` compares expected lyrics with a\n  transcript so semantic lyric mismatch can be caught after generation.\n- `scripts/generate_with_retry.py` now rejects `--output-path` for `mmx music\n  generate`/`cover` unless the underlying `mmx` command also includes `--out`.\n\n**Documentation:**\n- Both skills now document the lyrics tag whitelist, pre-generation lyrics\n  linting, post-generation lyrics alignment checks, and the `--out`/`--output-path`\n  contract.\n\n## v1.4.0\n\nv1.4.0 packages the arranger-helper expansion and the public-doc cleanup into\nthe next published release.\n\n**Workflow and routing:**\n- Base skill now documents the local ACE-Step arranger helper path more clearly:\n  `wait_for_acestep.py`, `extract_stems.py`, and `remix_stems.py`\n- MiniMax docs now distinguish fast cloud cover from the slower local ACE-Step\n  cover/repaint path without mixing maintainer-specific environment notes\n\n**Scripts and validation:**\n- New `batch_cover.py` for sequential MiniMax cover batches through\n  `generate_with_retry.py`\n- New `per_stem_analysis.py` for lightweight arranger triage from `stems.json`\n- New `hybrid_remix.py` for gated experimental remix planning with source-stem\n  validation at plan-build time\n- Expanded `scripts/smoke_test.py` to cover dry-run batch output and hybrid stem\n  validation gates\n\n**Documentation hygiene:**\n- Removed maintainer/private machine details from published references\n- Replaced user-specific example paths and placeholders with generic examples\n- Generalized Windows/WSL notes away from personal managed-network context\n\n## v1.3.0\n\nv1.3.0 folds the 2026-06-12 9-song field run feedback into the public\noperator workflow.\n\n**Documentation:**\n- Cloud duration caveat strengthened with verified results: 18/18 files saved,\n  but duration ranged from 57-135% of requested length (12 truncated, 3 close,\n  3 extended)\n- New [`references/short-prompt-recipes.md`](short-prompt-recipes.md) gives\n  short cloud prompt patterns under about 500 characters\n- [`references/minimax-generation-caveats.md`](minimax-generation-caveats.md)\n  now documents SIGTERM/SIGKILL-after-save recovery, MP3 output defaults,\n  prompt budget guidance, `--references` flakiness, and local-vs-cloud duration\n  comparison data\n\n**Scripts:**\n- `generate_with_retry.py` treats a fresh, probeable output file as a\n  successful generation after a signal-style return code\n- New `verify_cloud_output.sh` checks file existence, minimum size, MP3 format,\n  and optional expected-duration range\n- `check_environment.py` prints best-effort `mmx` path/version diagnostics\n\n## v1.1.0\n\nv1.1.0 combines the unpublished JioSaavn audio-source work with the later\nworkflow-hardening improvements into the next real ClawHub release after\nv1.0.1.\n\n**Documentation:**\n- New [`references/minimax-generation-caveats.md`](minimax-generation-caveats.md) documents sequential-run requirements, output-path verification, and duration-is-a-target behavior\n- Vocal-confirm gate and target-length confirmation gate added to the base skill intake workflow\n- Exact-duration routing clarified: `music-craft` with ACE-Step is the right tool when exact 3:00+ duration is required; MiniMax is the right tool when speed, cover workflows, or mmx flag control matter more\n\n**Linter (lint_music_request.py):**\n- New warning when a prompt requests >150 seconds with lyric-heavy density — advises that MiniMax often returns ~120–150s for those prompts and suggests ACE-Step as the exact-length alternative\n\n**Wrapper script (generate_with_retry.py):**\n- New `--output-path` flag moves the generated file to a caller-specified path after success (isolates each run from filename collisions)\n- New `--expected-duration-seconds` flag emits a best-effort warning (via ffprobe when available) if output is materially shorter than expected\n- Each invocation now uses an isolated temporary working directory so concurrent runs do not collide\n- Retry backoff is capped to prevent unbounded delays\n\n### Included scope: JioSaavn audio source additions\n\nv1.1.0 adds **JioSaavn as the third official audio source**, completing the fallback chain for cover, mashup, and style-transfer workflows.\n\n**Audio source fallback order:**\n\n| Priority | Source | When to use |\n|---|---|---|\n| 1 | YouTube | Global music, well-known international tracks |\n| 2 | **JioSaavn** | Bollywood, Hindi, Tamil, Telugu, Malayalam, Bengali, and other Indian regional music |\n| 3 | mx3.ch | Niche or regional sources with direct audio links |\n| 4 | Local file / alternate URL | When all cloud sources fail |\n\nJioSaavn URLs (`jiosaavn.com`, `www.jio.com/jiosaavn`) are handled by `yt-dlp` directly — no special auth required. Before downloading, inform the user: \"I'll download this from JioSaavn to analyze the audio.\"\n\nThe `--youtube` flag on `analysis_orchestrator.py` is reused for JioSaavn auto-detection; both YouTube and JioSaavn URLs are passed the same way.\n\n## v1.0.0\n\nv1.0.0 is the first stable release. It builds on the v0.x series (v0.3.0 / v0.4.0 dev line) with stronger preflight routing, wider prompt/flag consistency, and explicit post-generation verification:\n\n**Preflight routing:**\n- `lint_music_request.py` now emits one of six routes: `base_prompt`, `minimax_cover`, `minimax_mashup`, `minimax_style_transfer`, `minimax_emotion_prompt`, or `needs_clarification`\n- New blockers: missing Song B, missing lyrics decision, and conflicting cover/style-transfer intent\n- A `retry_guidance` array on every conflict so the operator can re-align prompt and flags\n\n**Prompt and flag consistency:**\n- Linter now detects conflicts in BPM, key, structure, duration, vocal mode, language, and avoid list\n- The canonical `mmx` prompt schema is documented in `examples.md`\n\n**Analysis quality:**\n- All analysis scripts converge on a compact `summary` (tempo, key, sections, instrumentation, vocal traits, energy curve, hook points, mix notes)\n- Confidence levels (clear, high, medium, low, inferred, missing) attached to every detection\n- Missing optional dependencies fall back to a JSON error block instead of failing the whole workflow\n\n**Output verification:**\n- Post-generation verification checklists for covers, mashups, style transfer, and emotion prompts\n- Eight failure signatures (copied too closely, lost melody, wrong tempo, wrong key, muddy mix, weak chorus, style mismatch, neutral vocals) with matching fixes\n- Revision prompt templates that preserve source identity while fixing one specific dimension\n\n**Tests and portability:**\n- Smoke tests now cover all new linter routes, the new conflict types, and the stdlib-only import guarantee\n- Windows is documented as partial support; scripts stay POSIX-safe, audio tools may need platform install\n\n## v0.3.0\n\nv0.3.0 builds on v0.2.0 with a substantially richer analysis pipeline:\n\n**New analysis scripts (8):**\n- `extract_stems.py` — Demucs source separation (vocal/drums/bass/other)\n- `track_beats.py` — beat_this beat + downbeat tracking (ISMIR 2024 SOTA)\n- `extract_melody.py` — Spotify Basic Pitch polyphonic AMT → MIDI + key/scale\n- `compute_audio_embedding.py` — MERT v1-330M music embeddings (vibe similarity)\n- `classify_instruments.py` — MIT AST 527-class AudioSet tagging\n- `extract_video_features.py` — extended with camera motion + VLM captioning\n- `analyze_image.py` — extended with OpenCLIP, OCR, face detection, VLM caption\n- `analysis_orchestrator.py` — single entry point, --use-demucs, --vlm, --ocr flags\n\n**New prompt slots (consumed in emotion_to_prompt.py):**\n- `beat grid: 4/4 at 150 BPM (confidence 0.80)` from beat_this\n- `melodic key from MIDI: E minor; interval motion: mostly leaps; modal character: pentatonic, blues` from Basic Pitch\n- `AST-detected sound palette: rock music (0.16), punk rock (0.14), grunge (0.20)` from MIT AST\n- `emotion signature from analysis: intense, passionate, dramatic, triumphant` (expanded to 25-emotion classifier)\n- `vocal texture in verse: breathier / more intimate than average` (per-section aggregation)\n- `tempo: tight, on-beat delivery` (from tempo_consistency)\n- `tonal character: dark warm tone, rolled-off highs` (from brightness)\n- `instruments detected: electronic / synthetic textures` (from instrument_hints)\n- `natural dramatic pauses detected at: 2s (11.7s pause), 20s (3.3s pause)` (from Demucs vocal-stem)\n- `style direction: ...` (from analyze_two_songs mashup_plan)\n\n**Bug fixes:**\n- parselmouth 0.4.x API (get_value_at_time / get_value_at_xy)\n- ffmpeg 8.x image2 muxer workaround (per-frame extraction)\n- pylette 5.1+ capital-P import + Pylette fallback\n- open_clip 3.3 3-tuple return + get_tokenizer() for tokenizer\n- demucs 4.x apply_model API\n\n**Prompting wins (verified end-to-end with a source-audio test case):**\n- Mix: 0 silence gaps, 35 pitch bends\n- Vocal stem: 19 silence gaps, 49 pitch bends, 2.32 syll/sec\n- BPM 150 (4/4) from beat_this, E minor (MIDI-confirmed G# minor)\n- AST: \"Rock music\", \"Punk rock\", \"Heavy metal\", \"Grunge\" — matches the actual band\n\nFile v1.6.0:references/cover-workflow.md\n\n# Cover Workflow\n\nCover workflow preserves the original song's melody while applying a different style. This is the feature that lets you turn a rock track into a French chanson version, a reggaeton track into a ballad, or any other style transfer while keeping the original recognisable.\n\n## When to Use Cover\n\n- The user has reference audio (a local file) and wants to keep the melody\n- The user wants to change style, era, or genre of an existing song\n- The user wants a \"what if\" reimagining (what if Bohemian Rhapsody was bossa nova?)\n\n## When NOT to Use Cover\n\n- The user wants to write a new song inspired by another (use standard generation with references)\n- The user wants to combine two songs into one (use mashup workflow)\n- The user only wants to change tempo or key (use standard generation with those parameters)\n- The user does not have reference audio and just describes a style (use standard generation)\n\n## Two Cover Backends\n\n> **Two cover backends exist — pick by what's available:**\n> - **MiniMax cloud cover** (this skill): run `music cover` through\n>   `scripts/generate_with_retry.py`, melody-preserving via MiniMax's\n>   `music-cover` model. Needs `MINIMAX_API_KEY` and network access to MiniMax.\n> - **Local ACE-Step cover** (in [`music-craft`](../../music-craft/)): `task_type=cover` with the\n>   source audio uploaded (multipart) and `audio_cover_strength` controlling how far to restyle.\n>   Fully local, no cloud, follows the source melody/structure. Caveat: a full-length cover is\n>   slow and VRAM-heavy on a ~12 GB GPU and can hit the server's 600 s generation timeout —\n>   cover a shorter segment or raise `ACESTEP_GENERATION_TIMEOUT`. See music-craft's\n>   \"ACE-Step Audio-Conditioned Generation\" section.\n>\n> So if MiniMax is unavailable (no key, CLI missing, or API access unavailable), you can **still do a\n> melody-aware cover locally** with ACE-Step — it is not cloud-only. Only pure text-prompt\n> generation (no source audio) is a \"reimagining\" rather than a cover.\n\n## Cloud Transmission Consent\n\n> ⚠️ **Both cover backends may transmit your audio to MiniMax.** The\n> cloud `music cover` path uploads the source file; the local ACE-Step\n> path stays on your machine. Before running a cloud cover with\n> sensitive, private, or third-party-owned audio, confirm:\n>\n> 1. You have the right to upload this audio to MiniMax.\n> 2. You accept that MiniMax may retain derived features (`cover_feature_id`,\n>    extracted lyrics, structure metadata) for up to 24 hours.\n> 3. The audio is not subject to a license that prohibits cloud processing.\n>\n> If any of these are unclear, use the local ACE-Step cover path instead.\n\n## Two Paths: One-Step vs Two-Step\n\n### One-Step (Quick)\n\n```bash\npython3 scripts/generate_with_retry.py \\\n  --output-path /tmp/cover.mp3 \\\n  -- \\\n  music cover \\\n  --prompt \"French chanson, accordion, strings, passionate French vocal, 80 BPM\" \\\n  --audio-file /tmp/original.ogg \\\n  --out /tmp/cover.mp3\n```\n\n**What happens:**\n\n1. MiniMax loads the local audio file you point at.\n2. Extracts lyrics via ASR (automatic speech recognition).\n3. Detects the structure (verse / chorus / bridge).\n4. Applies the target style while preserving the melody.\n5. Returns the cover.\n\nRun one cover at a time. If you need multiple variants, use\n`scripts/batch_cover.py --dry-run` first, then execute the sequential batch.\n\n**Limitations:**\n\n- ASR may mis-hear lyrics, especially in noisy or non-English audio.\n- The detected structure may not match the user's intent.\n- No way to edit lyrics before generation.\n\n### Two-Step (More Control)\n\n**Step 1: Preprocess the audio**\n\nThe preprocess step returns a `cover_feature_id` (valid 24 hours) plus the auto-extracted `formatted_lyrics` and a `structure_result` with section timestamps.\n\n```bash\ncurl --request POST \\\n  --url https://api.minimax.io/v1/music_cover_preprocess \\\n  --header \"Authorization: Bearer $MINIMAX_API_KEY\" \\\n  --header \"Content-Type: application/json\" \\\n  --data '{\n    \"model\": \"music-cover\",\n    \"audio_url\": \"file:///tmp/original-song.mp3\"\n  }'\n```\n\nNote: in v1.5.0+ `audio_url` must point to a local file the MiniMax API can read (e.g. `file:///tmp/...`). Public streaming URLs (YouTube, JioSaavn, mx3.ch) are no longer accepted — download the file with the private `music-source-fetch` skill first, then reference it here.\n\nThe response includes:\n\n- `cover_feature_id` — valid 24 hours, use in step 2\n- `formatted_lyrics` — editable, with structure tags\n- `structure_result` — JSON with section timestamps\n\nNote: do NOT use `mmx music cover` here. The mmx cover subcommand is the one-step path. For two-step, you must call `music_cover_preprocess` directly to get the `cover_feature_id` for step 2.\n\n**Step 2: Generate cover with modified lyrics**\n\n```bash\ncurl --request POST \\\n  --url https://api.minimax.io/v1/music_generation \\\n  --header \"Authorization: Bearer $MINIMAX_API_KEY\" \\\n  --header \"Content-Type: application/json\" \\\n  --data '{\n    \"model\": \"music-cover\",\n    \"cover_feature_id\": \"ID_FROM_STEP_1\",\n    \"lyrics\": \"[Verse]\\nModified French lyrics here\\n\\n[Chorus]\\nMore lyrics\",\n    \"prompt\": \"French chanson, accordion, strings, passionate vocal\",\n    \"output_format\": \"url\",\n    \"audio_setting\": { \"sample_rate\": 44100, \"bitrate\": 256000, \"format\": \"mp3\" }\n  }'\n```\n\n**What the two-step path gives you:**\n\n- Edit the ASR'd lyrics before generation (fix errors, change wording, add structure tags)\n- Use a different language (translate the original lyrics to the target language)\n- Use no lyrics at all (just preserve the melody, instrumental cover)\n- Use external lyrics (the user provided their own)\n\n## Lyrics Strategies for Cover\n\n### Same Lyrics, New Style\n\nThe user wants the same words, different style. Workflow:\n\n1. Use one-step or two-step.\n2. Pass the prompt for the new style.\n3. Let MiniMax extract the lyrics (one-step) or use the user's lyrics (two-step with `--lyrics`).\n\n### New Lyrics, Same Melody\n\nThe user wants different words in the same melody. Workflow:\n\n1. Use two-step (you need to provide the new lyrics).\n2. Preprocess to get the `cover_feature_id`.\n3. Provide the new lyrics in the generation call.\n\nExample: Take the melody of \"Yesterday\" by the Beatles and set new lyrics in Spanish about a modern-day break-up.\n\n### Translated Lyrics\n\nThe user wants the original lyrics in a different language. Workflow:\n\n1. Use two-step.\n2. Translate the original lyrics yourself (or with the LLM).\n3. Pass the translated lyrics in the generation call.\n\n### No Lyrics (Instrumental Cover)\n\nThe user wants the melody as an instrumental. Workflow:\n\n1. Use one-step with `--instrumental` flag, or\n2. Use two-step with empty lyrics in the generation call.\n\n## Style Transfer Without Melody Preservation\n\nIf the user wants the *style* of Song A applied to *new* lyrics (not preserving the melody), use standard generation with the style as a reference:\n\n```bash\nmmx music generate \\\n  --prompt \"Style: similar to Bohemian Rhapsody (epic rock, multi-section, choir, dramatic dynamics), but with original lyrics in Spanish about leaving home\" \\\n  --lyrics \"...\" \\\n  --model music-2.6 \\\n  --out /tmp/song.mp3\n```\n\nThis is NOT a cover — the melody is new, the style is borrowed.\n\n## Cover Workflow with a Local File\n\nThe source for a cover must be a local audio file path. URLs are not\naccepted in v1.5.0+.\n\nIf your source is a YouTube/JioSaavn/mx3.ch URL, download it first with\nthe private `music-source-fetch` skill, then pass the resulting local\nfile to this workflow.\n\n## Audio Input Limits\n\n- **Minimum length:** 6 seconds\n- **Maximum length:** 6 minutes\n- **Maximum file size:** 50 MB\n- **Supported formats:** mp3, wav, flac, ogg, m4a, and others\n\nIf the input is too long, trim it. If too large, convert to a lower bitrate.\n\n## Cover Feature ID Lifecycle\n\n`cover_feature_id` is valid for 24 hours from the preprocess call. After that, you need to re-preprocess.\n\nIf you need to regenerate the same cover multiple times (e.g., to iterate on the prompt), cache the `cover_feature_id` and reuse it.\n\n## Quality Verification for Covers\n\nAfter generating a cover, check:\n\n1. **Melody recognisability** — Would a friend say \"That's [Song X]\"?\n2. **Style application** — Would a friend say \"but it sounds like [style Y]\"?\n3. **Lyrics alignment** — Are the lyrics recognisable (either the original or the new ones)?\n4. **Structure preservation** — Does the new version follow the original's structure (verse-chorus-verse-chorus-bridge-chorus)?\n5. **No a cappella or sparse drops** — Same anti-sparse rules as standard generation\n\nIf 3+ of these fail, adjust the prompt or try the two-step path for more control.\n\n## Anti-Sparse for Covers\n\nThe anti-sparse rules apply even more strictly to covers, because the cover is changing style and the model can interpret \"intimate ballad version\" as \"remove all instruments\".\n\nAlways include in the cover prompt:\n\n```\nALL instruments ALWAYS playing throughout, NEVER go a cappella or silent,\nquiet sections: reduced to [specific instruments] only, still fully played\n```\n\nAnd in `--avoid`:\n\n```\nsparse, a cappella, minimal, silence, electronic sounds (unless desired)\n```\n\n## Errors and Recovery\n\n| Error | Cause | Fix |\n|---|---|---|\n| `audio_url unreachable` | Local file path is wrong, file is unreadable, or it was a streaming URL | Confirm the file exists and is readable; if you only had a streaming URL, fetch it first with `music-source-fetch` |\n| `audio too long` | Source > 6 minutes | Trim with `ffmpeg -ss <start> -t <duration>` |\n| `audio too large` | Source > 50 MB | Convert to lower bitrate with `ffmpeg` |\n| ASR extracted wrong lyrics | Noisy audio, accented vocals, non-English | Use two-step with manual lyrics |\n| Output melody is unrecognisable | Style transfer too aggressive | Reduce the prompt's style intensity, or use a less dramatic target style |\n| Output is sparse | Anti-sparse rules not applied | Add explicit instruments and \"ALL instruments ALWAYS playing\" |\n\n## Worked Example: Rock to Chanson\n\nUser request: \"Take this rock anthem I wrote and turn it into a French chanson.\"\n\nWorkflow:\n\n1. Get the audio file (user upload, or a local file fetched with `music-source-fetch` from a URL).\n2. Preprocess:\n   ```bash\n   python3 scripts/generate_with_retry.py \\\n     --output-path /tmp/chanson_cover.mp3 \\\n     -- \\\n     music cover \\\n     --prompt \"French chanson, accordion, upright bass, orchestral strings, piano, light percussion, 80 BPM in E minor, passionate French male vocal, melancholic romantic dramatic\" \\\n     --audio-file /tmp/rock_track.mp3 \\\n     --lyrics \"[Verse]\\nJ'ai trouvé ta lettre\\nMais je n'ai pas vraiment vérifié\\nCe mot unique que tu as écrit\\nL'histoire est déjà terminée\\n\\n[Pre Chorus]\\nNon, je ne veux pas connaître les raisons\\nCar l'amour n'a pas besoin de raison pour exister\\n\\n[Chorus]\\nJe sais que je, je sais que je, je ne peux pas continuer avec toooooou\" \\\n     --out /tmp/chanson_cover.mp3\n   ```\n3. Verify melody is recognisable.\n4. If sparse, retry with explicit anti-sparse text.\n5. Deliver the cover.\n\nFile v1.6.0:references/emotion-analysis.md\n\n# Emotion Analysis\n\nThe signature feature of `music-craft-minimax` is **emotion-driven prompt engineering**: analyzing the emotional arc of input audio and using it to construct a production-sheet prompt that captures the dynamics, not just the surface style.\n\nThis file covers the **analytical side**: what we detect, the pipeline, the scripts. For the **generation side** (how to use the analysis to evoke emotion in the OUTPUT, emotion recipes, iteration loop), see [`references/emotion-delivery.md`](emotion-delivery.md).\n\n## What It Detects\n\nEmotion analysis extracts per-section features from input audio:\n\n| Feature | Description | Music Generation Use |\n|---------|-------------|---------------------|\n| `avg_intensity` | Loudness (RMS energy) | Dynamic range control |\n| `pitch_range_hz` | Pitch variation width | Emotional intensity |\n| `pitch_trend` | rising / falling / steady | Build-up vs release |\n| `vocal_effort` | low / medium / high | Vocal intensity and strain |\n| `pitch_stability` | 0–1 (1 = very stable) | Controlled vs raw delivery |\n| `breathiness` | 0–1 (from spectral features) | Intimate vs full-voice |\n| `spectral_centroid` | Average brightness | Timbre matching |\n| `emotion_classification` | List of emotions | Mood keywords for prompt |\n| `vocal_speed` | Syllables per second | Elongation cues |\n| `pitch_bends` | Slides at phrase endings | Emotional emphasis |\n\n## Intensity Curve Patterns\n\nThe most powerful output of emotion analysis is the **intensity curve**: how the song's energy changes over time. Five patterns are detected:\n\n| Pattern | Shape | Arrangement Strategy |\n|---------|-------|---------------------|\n| `crescendo` | Builds to climax | Start sparse → add layers progressively |\n| `decrescendo` | Starts intense, fades | Full arrangement → strip back |\n| `wave` | Multiple peaks | Vary density per section, no single peak |\n| `climax_late` | Peak near end | Restrained first 2/3, explosive ending |\n| `climax_early` | Peak at start | Powerful opening, reflective rest |\n\nWhen the analysis detects a pattern, the generated prompt includes arrangement instructions that match:\n\n```json\n{\n  \"section_prompts\": [\n    {\"section_label\": \"intro\", \"arrangement_density\": \"sparse\", \"instruction\": \"INTRO: quiet, intimate — sparse arrangement\"},\n    {\"section_label\": \"chorus\", \"arrangement_density\": \"full\", \"instruction\": \"CHORUS: loud, powerful — full arrangement, building tension\"}\n  ]\n}\n```\n\n## Pause and Intensity Patterns\n\nThe user (and music generally) cares a lot about WHEN the song breathes. Two related patterns:\n\n### Dramatic Pauses (Driven by `[Break]` Tags)\n\nDramatic pauses (silence or near-silence) create tension, release, or emotional impact. They are detected by sudden drops in intensity. The analysis tells you WHERE pauses already exist in the input; the generation side adds pauses via `[Break]` tags in the lyrics.\n\n| Pause location | Effect | Use case |\n|---|---|---|\n| Before first chorus | Build anticipation | Standard pop structure |\n| After climactic moment | Let the impact land | Emotional peaks |\n| Before bridge | Reset before contrast | Standard pop structure |\n| Before final chorus | Build to the biggest moment | Standard pop structure |\n| Between short phrases | Intimacy, breath | Slow ballads, R&B |\n\nIn MiniMax generation, use `[Break]` structure tags to indicate pauses. The model creates 1–2s of silence or near-silence at those points.\n\n### Where to Add `[Break]` Tags in Lyrics (Quick Reference)\n\n| Section transition | Should have `[Break]`? | Why |\n|---|---|---|\n| Intro → Verse | Optional | Depends on style |\n| Verse → Pre-Chorus | No | Build should be continuous |\n| Pre-Chorus → Chorus | **Yes** | Creates anticipation |\n| Chorus → Verse | No | Verse starts lower energy |\n| Verse → Bridge | **Yes** | Signals contrast |\n| Bridge → Chorus | **Yes** | Especially before final chorus |\n| Chorus → Outro | Optional | Depends on fade |\n| Final chorus → Outro | **Yes** | Lets the final note land |\n\nIf the user is having pause problems, the most common cause is missing `[Break]` tags before the chorus. Add them and the pauses appear.\n\n### Intensity Dynamics Across the Song\n\nThe analysis tracks intensity at multiple granularities:\n\n- **Per-section average**: a single number per detected section (verse, chorus, etc.)\n- **Intensity curve**: 20 sample points across the song's duration\n- **Per-frame intensity**: every ~10ms (used internally for feature extraction)\n\nWhen translating to a prompt, the section-level granularity is what matters most. Use the intensity curve to inform the arrangement (crescendo, decrescendo, wave).\n\n## Emotion Classifications (25+)\n\nThe classifier outputs a list of emotions detected across the song. The expanded set covers the full emotional spectrum commonly used in music:\n\n### Core 9 (from the original skill)\n\n| Emotion | Triggers (audio signature) | Music Prompt Effect |\n|---------|----------------------------|---------------------|\n| `intense` | High overall intensity | Loud, full arrangement |\n| `calm` | Low overall intensity | Sparse, intimate |\n| `dramatic` | Wide pitch range + dynamic shifts | Theatrical dynamics |\n| `desperate` | High effort + unstable pitch + falling trend | Raw, urgent, stretched delivery |\n| `passionate` | Medium-high effort + stable + warm | Heartfelt, powerful, sustained |\n| `restrained` | Low effort + controlled | Gentle, measured |\n| `building` | Rising pitch trend + increasing intensity | Tension building |\n| `releasing` | Falling pitch trend + decreasing intensity | Relaxation, release |\n| `breathy` | High ZCR (zero crossing rate) + low energy | Intimate, whispered |\n\n### Extended 16+ (added for richer coverage)\n\n| Emotion | Triggers (audio signature) | Music Prompt Effect |\n|---------|----------------------------|---------------------|\n| **`joyful`** | High energy + stable pitch + bright spectral centroid + high BPM | Upbeat, bright, major key, smiling, energetic delivery |\n| **`triumphant`** | High intensity + rising pitch + stable + building | Powerful, building, celebratory, wide dynamics, choir in climax |\n| **`melancholic`** | Low energy + falling pitch + breathy + slow | Sad, minor key, intimate, slow, low register |\n| **`angry`** | High effort + unstable + sharp dynamics + distorted | Aggressive, shouted, distorted guitars, punchy drums |\n| **`yearning`** | High pitch + breathy + unstable + sustained notes | Longing, sustained, high register, \"where are you\" |\n| **`nostalgic`** | Low energy + slow + falling + vintage timbre | Wistful, gentle, soft, vintage feel |\n| **`anxious`** | Unstable + fast + breathy + irregular rhythm | Tense, urgent, sharp, rapid, irregular |\n| **`confident`** | High energy + stable + falling trend | Strong, controlled, clear, direct |\n| **`vulnerable`** | High breathiness + low stability + low register | Fragile, intimate, exposed, soft |\n| **`defiant`** | High effort + high intensity + dramatic + falling | Rebellious, powerful, confrontational |\n| **`hopeful`** | Rising pitch trend + building intensity + stable + mid-bright spectral centroid | Uplifting, ascending, building, future-focused, brightening |\n| **`tragic`** | Falling pitch + decreasing intensity + low effort + monotone energy | Doomed, fated, resigned, low register, no climax |\n| **`heroic`** | Wide pitch range + high intensity + building + march-like rhythm | Powerful, declarative, marching, brass, building, honor |\n| **`tender`** | Warm spectral centroid + low vocal effort + low-mid intensity + slow + breathy + stable | Warm, gentle, intimate, slow, affectionate, \"you/we/hold\" |\n| **`sensual`** | Low register + high breathiness + slow + warm spectral centroid + sustained | Seductive, breathy, low register, sustained, R&B feel, \"touch/whisper\" |\n| **`lonely`** | Low energy + high breathiness + sparse (single voice vs choir) + falling + low-mid register | Isolated, sparse, single voice, low register, empty, \"alone/silence\" |\n| **`playful`** | High energy + irregular rhythm + mid-uptempo + bright + varied pitch | Bouncy, syncopated, fun, light, \"let's/hey/ha\" |\n| **`haunting`** | Low energy + sparse + dissonant (minor 2nds, tritones) + falling + dark spectral centroid + breathy | Eerie, spectral, low pads, dissonant, sparse, \"ghost/shadow/silence\" |\n| **`serene`** | Very low intensity + no peaks + very slow + low-mid pitch + stable + very low dynamic range | Tranquil, ambient, pad-based, no percussion, nature, \"still/calm/breath\" |\n| **`celebratory`** | High energy + bright + climax_late + major key + dense + full band | Festive, jubilant, full band, dense, horns, choir, exclamations |\n| **`bittersweet`** | MIXED: some sections bright + stable (joy), other sections dim + breathy + falling (melancholy). Same song, different emotions across sections. | Mixed: bright verses, dim chorus (or vice versa). Major key with minor chords. \"Happy melody, sad lyrics\" or vice versa. |\n\nThe detected emotions are translated to mood adjectives in the final prompt. For the full mapping from detected emotion to generated output, see [`references/emotion-delivery.md`](emotion-delivery.md) → \"Emotion Recipes\".\n\n## Per-Emotion Detection Cookbook\n\nThis section tells the LLM (or the classifier) HOW to detect each emotion from the audio features. The base analysis outputs raw features (`vocal_effort`, `breathiness`, `pitch_trend`, `pitch_stability`, `pitch_range_hz`, `spectral_centroid`, `avg_intensity`, `vocal_speed`, `pitch_bends`). The LLM (or the script) maps these to emotions.\n\n### Detection rules per emotion\n\n| Emotion | Detection rule (which features indicate this emotion) |\n|---|---|\n| `intense` | `avg_intensity` above 0.1; high BPM (> 120); dense arrangement |\n| `calm` | `avg_intensity` below 0.05; low BPM (< 90); sparse arrangement |\n| `dramatic` | `pitch_range_hz` above 200; `vocal_effort` varies across sections; wide intensity range |\n| `desperate` | `vocal_effort` = high + `pitch_stability` < 0.5 + `pitch_trend` = falling |\n| `passionate` | `vocal_effort` = high + `pitch_stability` > 0.7 + warm `spectral_centroid` |\n| `restrained` | `vocal_effort` = low + `pitch_stability` > 0.7 + low intensity variation |\n| `building` | `pitch_trend` = rising + intensity curve = `crescendo` |\n| `releasing` | `pitch_trend` = falling + intensity curve = `decrescendo` |\n| `breathy` | `breathiness` > 0.5 + low `vocal_effort` |\n| `joyful` | `avg_intensity` high + `pitch_stability` > 0.7 + bright `spectral_centroid` + BPM > 110 + `vocal_speed` normal-to-fast |\n| `triumphant` | `avg_intensity` high + `pitch_trend` = rising + `pitch_stability` > 0.7 + intensity curve = `climax_late` |\n| `melancholic` | `avg_intensity` low + `pitch_trend` = falling + `breathiness` > 0.4 + BPM < 90 + `vocal_speed` slow |\n| `angry` | `vocal_effort` = high + `pitch_stability` < 0.5 + sharp intensity changes (per-section avg varies > 30%) |\n| `yearning` | high `pitch_range_hz` + `breathiness` > 0.5 + `pitch_stability` < 0.6 + sustained notes (`vocal_speed` < 3 syl/sec in chorus) |\n| `nostalgic` | `avg_intensity` low + `vocal_speed` slow + `pitch_trend` = falling + low spectral centroid |\n| `anxious` | `pitch_stability` < 0.4 + high BPM (> 130) + `breathiness` > 0.3 + irregular rhythm (per-section intensity varies erratically) |\n| `confident` | `avg_intensity` high + `pitch_stability` > 0.8 + `pitch_trend` = falling or steady |\n| `vulnerable` | `breathiness` > 0.7 + `pitch_stability` < 0.5 + low `pitch_range_hz` (low register) |\n| `defiant` | `vocal_effort` = high + `avg_intensity` high + `pitch_range_hz` wide + dramatic dynamics |\n| `hopeful` | `pitch_trend` = rising + intensity curve = `crescendo` + `pitch_stability` > 0.6 + bright `spectral_centroid` |\n| `tragic` | `pitch_trend` = falling + intensity curve = `decrescendo` + monotone energy (low per-section variation) + low `vocal_effort` |\n| `heroic` | `pitch_range_hz` > 250 + `avg_intensity` high + building intensity + march-like rhythm (regular onset intervals) |\n| `tender` | warm `spectral_centroid` (> 1500 Hz but not bright) + `vocal_effort` = low + `breathiness` > 0.3 + slow + `pitch_stability` > 0.7 |\n| `sensual` | low `pitch_range_hz` (< 200) + `breathiness` > 0.6 + slow + warm spectral centroid + sustained notes |\n| `lonely` | `avg_intensity` very low + `breathiness` > 0.5 + `vocal_speed` slow + single voice (vs choir) + falling |\n| `playful` | high `avg_intensity` + irregular rhythm (onset intervals vary) + bright `spectral_centroid` + BPM 100-130 + varied pitch |\n| `haunting` | `avg_intensity` low + dissonant (minor 2nd / tritone intervals) + `breathiness` > 0.4 + low `spectral_centroid` (< 1500) + falling |\n| `serene` | `avg_intensity` very low (< 0.03) + very low dynamic range + `vocal_speed` slow + no peaks in intensity curve + `pitch_stability` very high |\n| `celebratory` | `avg_intensity` high + bright `spectral_centroid` + intensity curve = `climax_late` + dense + full band energy |\n| `bittersweet` | MIXED SIGNALS across sections: some sections have `joyful` features, others have `melancholic` features. Look for the contrast, not any single section. |\n\n### When classification is uncertain\n\n- The audio features are ambiguous (e.g., moderate intensity, stable pitch, mid-range spectral centroid) — could be `calm`, `restrained`, or `serene`. Pick the most specific: if very low intensity, `serene`; if moderate, `calm`; if controlled, `restrained`.\n- The audio has clear primary emotion but ambiguous secondary — report the primary with high confidence, secondary with low.\n- The audio doesn't fit any 25+ emotion well — report as \"neutral\" or describe with custom adjectives.\n\n### Using the cookbook with the LLM\n\nThe cookbook rules are designed to be applied by the LLM (or by the script) AFTER the raw features are extracted. The flow:\n\n1. Run `analyze_vocal_emotion.py` to get raw features (vocal_effort, breathiness, etc.)\n2. Apply the cookbook rules to derive emotion candidates with confidence\n3. Pass the emotion list (with confidence) to `emotion_to_prompt.py` for the final prompt\n4. The LLM can also override the classification based on context (e.g., the lyrics content)\n\nThe LLM is the final arbiter. The cookbook is a guide, not a hard rule.\n\n## Emotion Combinations\n\nSome songs carry TWO primary emotions — a juxtaposition that is itself a creative choice. The skill supports this by detecting combinations.\n\n### Common combinations\n\n| Combination | What it sounds like | Lyrics strategy |\n|---|---|---|\n| `bittersweet` = `joyful` + `melancholic` | Happy melody, sad lyrics (or vice versa) | Major key with minor chords; bright verses, dim chorus (or vice versa) |\n| `tender aggression` = `tender` + `angry` | Soft, gentle vocal over hard, distorted instruments | Quiet delivery with loud backing |\n| `defiant hope` = `defiant` + `hopeful` | Rebellious optimism | \"I/we will fight\" + building intensity + uplifting |\n| `lonely triumph` = `lonely` + `triumphant` | Alone but victorious | \"I did it\" + sparse verses, dense climax |\n| `nostalgic joy` = `nostalgic` + `joyful` | Warm memories of happy times | Past tense + bright delivery |\n| `anxious hope` = `anxious` + `hopeful` | Worried but believing in better | Tense verses, resolving chorus |\n| `vulnerable confidence` = `vulnerable` + `confident` | Exposed but sure | \"I am\" + breathy vocal + strong arrangement |\n| `playful defiant` = `playful` + `defiant` | Fun rebellion | Humorous + rebellious (e.g., punk pop) |\n\n### How to detect a combination\n\nThe LLM should look for the contrast across sections:\n\n- **Verses have emotion A, chorus has emotion B** → combination of A and B\n- **Vocal has emotion A, accompaniment has emotion B** → combination of A and B (analyze vocal and accompaniment separately for this)\n- **Lyrics suggest emotion A, delivery suggests emotion B** → combination of A and B\n\nWhen a combination is detected, the prompt should include BOTH emotion sets in the mood slot, and the arrangement should support both.\n\nExample prompt for `bittersweet`:\n\n```\nBittersweet, joyful verses with melancholic chorus, happy-sad, major key with minor chords,\nbright vocal in verses, falling intonation in chorus,\nstrings + piano, light percussion, building then releasing,\nALL instruments always playing throughout,\n120 BPM in C major\n```\n\nFor the full combinations table, see [`references/emotion-delivery.md`](emotion-delivery.md) → \"Emotion Combinations\".\n\n## Pre-Trained Classifiers (Optional)\n\nFor automated classification (without LLM inference), the skill can use pre-trained models. See [`references/advanced-audio-analysis.md`](advanced-audio-analysis.md) for Essentia's `mood_classifier` and `valence / arousal` predictions.\n\n- **Essentia mood classifier**: outputs `happy` / `sad` / `aggressive` / `relaxed` directly. Map to the 25+ emotion set.\n- **Essentia valence + arousal**: 2D continuous space. Map to emotion quadrants.\n- **Custom CNN models**: train your own on labeled emotion data (out of scope for the skill, but possible).\n\n## Vocal Effort → Prompt Mapping\n\nThe `vocal_effort` feature (low / medium / high) directly maps to specific prompt wording. Use this to evoke the right intensity in the generated output.\n\n| Effort | Audio signature | Prompt wording |\n|---|---|---|\n| `low` | Quiet, breathy, low spectral centroid, low RMS | \"soft vocal\", \"whispered delivery\", \"delicate\", \"intimate\", \"tender\" |\n| `medium` | Balanced, stable, mid-range, mid RMS | \"clear vocal\", \"controlled delivery\", \"expressive but measured\" |\n| `high` | Loud, strained, wide pitch range, high RMS | \"powerful vocal\", \"shouted delivery\", \"raw\", \"strained\", \"intense\" |\n\nFor the same emotion, the effort can vary by section. Example: a desperate song might have `low` effort in verses (whispered) and `high` effort in the final chorus (shouted). The analysis captures this per section.\n\n## Breathiness → Prompt Mapping\n\nThe `breathiness` feature (0.0–1.0) controls how intimate vs projected the voice sounds.\n\n| Breathiness | Audio signature | Prompt wording |\n|---|---|---|\n| `0.0–0.2` (none) | Clear, projected, full voice | \"clear vocal\", \"full voice\", \"projected\" |\n| `0.2–0.5` (low) | Slight breath, controlled | \"warm vocal\", \"natural delivery\" |\n| `0.5–0.8` (medium) | Audible breath, intimate | \"breathy vocal\", \"close-mic\", \"ASMR-like softness\" |\n| `0.8–1.0` (high) | Whispery, fragile, very close | \"whispered\", \"extremely intimate\", \"fragile\", \"delicate\" |\n\nBreathiness interacts with anti-sparse rules: a high-breathiness voice is often sparse (whispered delivery, single instrument). The \"intimate but not a cappella\" anti-sparse phrasing handles this — keep the accompaniment, soften the voice.\n\n## Repetitive Intensification\n\nWhen a phrase repeats with increasing intensity across repetitions (common in choruses), the analysis detects and quantifies it:\n\n```json\n{\n  \"detected\": true,\n  \"confidence\": 0.75,\n  \"increase_ratio\": 1.45,\n  \"description\": \"Detected intensifying repetition — energy increases 1.5x across repetitions\",\n  \"music_generation_note\": \"Increase dynamics and layer density with each repetition\"\n}\n```\n\nThis translates to: \"Chorus repeats with increasing intensity, each version adds layers (bass, then drums, then strings, then choir)\".\n\n## Emotional Shifts\n\nTransitions between sections are classified as sudden or gradual:\n\n```json\n{\n  \"at_seconds\": 145.2,\n  \"type\": \"sudden\",\n  \"from_effort\": \"low\",\n  \"to_effort\": \"high\",\n  \"description\": \"Sudden shift at 145.2s\"\n}\n```\n\nA sudden shift translates to: \"At [section], the arrangement suddenly goes from [X] to [Y], creating dramatic contrast\". A gradual shift translates to: \"Build from [X] to [Y] over [N] seconds\".\n\n## Vocal Speed and Elongation\n\nThe most distinctive feature: **vocal speed / syllable elongation**. Detects word stretching and tempo changes per section.\n\n```json\n{\n  \"pattern\": \"decelerating\",\n  \"description\": \"Vocals progressively slow down, especially at the end — emotional elongation pattern\",\n  \"average_syllables_per_second\": 4.46,\n  \"deceleration_detected\": true,\n  \"final_sections_slowed\": true,\n  \"sections\": [\n    {\"structural_label\": \"intro\", \"syllables_per_second\": 5.01, \"speed_classification\": \"normal\"},\n    {\"structural_label\": \"verse\", \"syllables_per_second\": 6.69, \"speed_classification\": \"accelerated\"},\n    {\"structural_label\": \"outro\", \"syllables_per_second\": 2.21, \"speed_classification\": \"slowed\"}\n  ]\n}\n```\n\n| Pattern | Meaning | Music Generation Effect |\n|---------|---------|------------------------|\n| `decelerating` | Vocals progressively slow down | Stretch syllables per section, elongated ending |\n| `late_elongation` | Final sections are much slower | Slow delivery for final chorus/bridge |\n| `gradual_slowing` | Steady deceleration throughout | Progressive rubato |\n| `accelerating` | Vocals speed up | Urgent, driving delivery |\n| `steady` | Consistent speed | Even pacing |\n\nThe translation to the prompt: instructions to stretch specific syllables (`\"with emotionally elongated delivery in the final chorus\"`) plus lyrics modifications (`\"reduce to ~5 syllables per line in the outro\"`).\n\nFor specific elongation examples by emotion (joy, desperation, melancholy, etc.), see [`references/emotion-delivery.md`](emotion-delivery.md) → \"Emotion Recipes\" and [`../music-craft/references/structure-tags.md`](../../music-craft/references/structure-tags.md) → \"Vocal Effects Through Lyrics\".\n\n## Pitch Bends at Phrase Endings\n\nDetects monotonic pitch slides (rising or falling) at the ends of vocal phrases — a hallmark of emotional delivery:\n\n```json\n{\n  \"start_seconds\": 42.1,\n  \"end_seconds\": 43.8,\n  \"direction\": \"falling\",\n  \"pitch_range_hz\": 85.3,\n  \"duration_seconds\": 1.7\n}\n```\n\nThis is hard to translate directly to a MiniMax prompt (the model does not have a \"pitch bend\" parameter), but it informs the prompt's mood and style choices: \"with emotionally expressive vocal delivery, falling pitch slides at phrase endings\".\n\nFor lyrics that evoke pitch bends, use:\n- Trailing vowels at the end of phrases: `\"...goodbyeee\"` (falling)\n- Rising inflections on questions: `\"Will you stay?^^\"` (upward inflection)\n- Wordless vocalizations: `\"oohhh\"`, `\"aaahhh\"`, `\"mmm\"`\n\n## The Analysis Pipeline\n\n```\nAudio file (WAV)\n  → analyze_vocal_emotion.py\n    → emotion JSON (per-section features, intensity curve, vocal speed, pitch bends)\n      → emotion_to_prompt.py\n        → production-sheet prompt (with arrangement plan, vocal speed cues, mood keywords)\n          → music_generate\n            → output song\n```\n\nThe full Python helpers are in [`../scripts/`](../scripts/):\n\n> **v1.5.0:** the published skill is audio-only. The video, image, and YouTube download scripts were removed — use the private `music-source-fetch` skill to acquire a local audio file first, then run the orchestrator on that path.\n\n- [`../scripts/analysis_orchestrator.py`](../scripts/analysis_orchestrator.py) — single entry point (audio only)\n- [`../scripts/analyze_vocal_emotion.py`](../scripts/analyze_vocal_emotion.py) — the main emotion extractor\n- [`../scripts/analyze_audio.py`](../scripts/analyze_audio.py) — basic features (BPM, key, energy)\n- [`../scripts/emotion_to_prompt.py`](../scripts/emotion_to_prompt.py) — converts emotion JSON to prompt\n- [`../scripts/analyze_two_songs.py`](../scripts/analyze_two_songs.py) — two-song comparison\n- [`../scripts/extract_lyrics_whisper.py`](../scripts/extract_lyrics_whisper.py) — Whisper-based lyrics extraction\n\n## Quick Start\n\n```bash\n# Step 1: Convert to WAV if needed\nffmpeg -i /tmp/song.mp3 -ar 44100 /tmp/song.wav\n\n# Step 2: Run emotion analysis\npython3 scripts/analyze_vocal_emotion.py /tmp/song.wav --output /tmp/emotion.json\n\n# Step 3: Convert to prompt\npython3 scripts/emotion_to_prompt.py \\\n  --emotion /tmp/emotion.json \\\n  --style /tmp/style.json \\\n  --language english \\\n  --output /tmp/mashup_prompt.json\n\n# Step 4: Extract the final prompt\nPROMPT=$(jq -r '.final_prompt' /tmp/mashup_prompt.json)\n\n# Step 5: Generate\nmmx music generate \\\n  --prompt \"$PROMPT\" \\\n  --lyrics \"...\" \\\n  --model music-2.6 \\\n  --out /tmp/output.mp3\n```\n\n## Local-Only Path (When MiniMax Is Unavailable)\n\n`emotion_to_prompt.py` calls the MiniMax cloud, so it fails when MiniMax API access is unavailable or no key is set. In that case build the prompt locally from the analysis JSON without that script: take the extracted BPM and key/scale as explicit metadata fields; turn the energy curve and spectral brightness into texture words; turn the emotion classification and intensity curve into mood words and dynamic section tags; and feed transcribed lyrics (full-mix Whisper) as the lyric body. This is the same data, assembled by the agent instead of the cloud helper, and it feeds any backend (including a local model).\n\n## emotion_to_prompt.py Output\n\nThe conversion script produces a structured JSON with everything needed for generation:\n\n```json\n{\n  \"style_category\": \"french_chanson\",\n  \"target_bpm\": 80,\n  \"target_duration_seconds\": 180,\n  \"final_prompt\": \"french chanson style, 1960s Paris café atmosphere, accordion...\",\n  \"structured_lyrics_template\": {\n    \"template\": \"[Intro]\\n(Instrumental)\\n\\n[Break]\\n(1-2 second dramatic pause)\\n\\n[Build Up]\\n(Tension building)\\n\\n[Chorus]\\n(powerful delivery)\\n{lyrics_here}\",\n    \"sections_count\": 8,\n    \"silence_gaps_used\": 2,\n    \"note\": \"Fill placeholders with actual lyrics, preserve structure tags\"\n  },\n  \"workflow_recommendation\": {\n    \"workflow\": \"cover_two_step\",\n    \"model\": \"music-cover\",\n    \"reasoning\": \"Original audio available → cover workflow for melody preservation\",\n    \"steps\": [\"1. Preprocess audio\", \"2. Edit lyrics\", \"3. Generate cover\"]\n  },\n  \"section_prompts\": [\n    {\"section_label\": \"intro\", \"arrangement_density\": \"sparse\", \"instruction\": \"INTRO: quiet, intimate — sparse arrangement\"},\n    {\"section_label\": \"chorus\", \"arrangement_density\": \"full\", \"instruction\": \"CHORUS: loud, powerful — full arrangement, building tension\"}\n  ],\n  \"arrangement_plan\": {\n    \"intro\": \"sparse: accordion only\",\n    \"chorus\": \"full arrangement: accordion, strings, piano, bass\",\n    \"final_chorus\": \"maximum intensity: all instruments + backing harmonies\"\n  },\n  \"vocal_speed_patterns\": {\n    \"detected\": true,\n    \"pattern\": \"late_elongation\",\n    \"prompt_additions\": [\"final sections feature emotionally stretched syllables\"],\n    \"lyrics_modifications\": [{\"section\": \"outro\", \"instruction\": \"Reduce to ~5 syllables per line\", \"example\": \"I caaan't goooo on with yoooou\"}],\n    \"section_cues\": [{\"section\": \"outro\", \"prompt_cue\": \"outro: slow, emotionally stretched delivery, hold last syllable\"}]\n  }\n}\n```\n\nUse `final_prompt` directly with `mmx music generate` or `music_generate`. Use the `section_prompts` and `arrangement_plan` for guidance when constructing the lyrics with section tags.\n\nThe `vocal_speed_patterns.prompt_additions` and `lyrics_modifications` are particularly important for emotional delivery — they translate the detected vocal speed pattern into concrete instructions.\n\n## When NOT to Use Emotion Analysis\n\n- The user has no audio input (skip the analysis, use standard generation)\n- The input is instrumental only (vocal emotion analysis requires vocals)\n- The input is very short (< 10 seconds) — analysis is unreliable\n- The input is heavily processed (synth, electronic) — pitch detection struggles\n- The user wants a quick draft — emotion analysis takes 10–60 seconds\n\n## Anti-Sparse from Emotion Analysis\n\nThe arrangement plan from emotion analysis includes anti-sparse by default — quiet sections are still \"fully played, NOT silent\". If you see sparse output despite this, the analysis may have been over-confident. Run it again with `--no-parselmouth` to use librosa's pitch detection (less accurate but more conservative).\n\nFor the deeper anti-sparse treatment specific to emotional delivery (which is inherently lower-intensity), see [`references/emotion-delivery.md`](emotion-delivery.md) → \"Common Mistakes in Emotional Delivery\".\n\n## Limitations\n\n- **Emotion detection is approximate** — based on audio features, not semantic understanding\n- **Pitch detection can struggle** with heavily processed or distorted vocals\n- **Style inference is a best guess** — manual override is always better\n- **Repetitive detection requires ≥ 3 repetitions** with consistent increase\n- **Vocal speed detection is ±20% accurate** — syllable count is estimated, not true phoneme alignment\n- **MiniMax has no \"vocal speed\" parameter** — elongation is achieved via lyrics formatting (fewer syllables, repeated vowels) and prompt cues\n- **MiniMax has no \"pitch bend\" parameter** — pitch slides are evoked via trailing vowels and wordless vocalizations\n- **MiniMax has no \"vocal effort\" parameter** — effort is evoked via prompt wording (\"shouted\", \"whispered\", \"raw\")\n- **The model can flatten emotional cues** if the prompt is vague. Always use the emotion recipes in [`references/emotion-delivery.md`](emotion-delivery.md).\n\nFile v1.6.0:references/emotion-delivery.md\n\n# Emotional Delivery in Generation\n\nThe analysis side ([`emotion-analysis.md`](emotion-analysis.md)) tells us what the input audio sounds like. The challenge is making the **output** sound emotional. This file covers the generation side: how to construct prompts, lyrics, and arrangements that actually evoke the target emotion.\n\nIf you have run emotion analysis on input audio, use the detected features (vocal effort, breathiness, intensity curve, vocal speed) to drive the generation choices below. If you have no input, use the LLM's knowledge plus these recipes to evoke emotion from scratch.\n\n> **v1.5.0 (audio-only):** the orchestrator is audio-only. The `image.*` and `video.*` rows in the v0.3.0 table below, the `--vlm` / `--ocr` / `--faces` flags, the `extract_video_features.py` and `analyze_image.py` scripts, and YouTube URL downloads have all been removed. See `references/changelog.md` for the full removal list. Current operating guidance starts at \"The 'Emotion Recipe' Pattern\" below.\n\n## What's New in v0.3.0 (May 2026) — historical, pre-v1.5.0\n\nv0.3.0 represents a major expansion of the analysis pipeline. The emotion\ndetection, beat tracking, melody analysis, instrument tagging, image\ncaptioning, and source separation are all production-grade.\n\n### New Scripts (8 new in v0.2-v0.3) — historical\n\n| Script | Tool | What it does |\n|---|---|---|\n| `extract_stems.py` | Demucs (MIT, opt-in via `--use-demucs`) | Source separation: vocals / drums / bass / other. Dramatically improves vocal emotion on busy mixes. |\n| `track_beats.py` | beat_this (MIT, ISMIR 2024 SOTA) | Beat + downbeat positions, BPM with confidence, time signature estimate |\n| `extract_melody.py` | Spotify Basic Pitch (Apache-2.0) | Polyphonic audio → MIDI; MIDI-confirmed key + scale modes |\n| `compute_audio_embedding.py` | MERT v1-330M (Apache-2.0) | 1024-dim music embedding; cosine similarity for \"vibe\" matching |\n| `classify_instruments.py` | MIT AST (AudioSet 527-class) | Fine-grained instrument / genre tagging (rock, grunge, punk, ...) |\n| `analysis_orchestrator.py` | (built in v0.1.0) | Single entry point; in v1.5.0 reduced to `--audio`, `--lyrics`, `--use-demucs` flags only |\n\n### New Analysis Outputs (v0.3.0 additions) — historical\n\n| Field | Source | Prompt effect |\n|---|---|---|\n| `beat_tracking.bpm_estimated` | beat_this | `\"beat grid: 4/4 at 150 BPM (confidence 0.80)\"` — overrides target_bpm if confidence > 0.8 |\n| `beat_tracking.time_signature_estimate` | beat_this | \"4/4\" or \"3/4\" in the beat grid line |\n| `melody_analysis.key_estimate_from_midi` | Basic Pitch | \"melodic key from MIDI: E minor\" — more reliable than chroma |\n| `melody_analysis.scale_modes` | Basic Pitch | \"modal character: pentatonic, blues\" |\n| `melody_analysis.interval_pattern` | Basic Pitch | \"interval motion: mostly leaps\" |\n| `melody_analysis.monophonic_fraction` | Basic Pitch | proxy for vocal vs polyphonic character |\n| `ast_classification.top_instruments` | MIT AST | \"AST-detected sound palette: rock music (0.16), punk rock (0.14)\" |\n| `vocal_emotion.silence_gaps` (with `--use-demucs`) | Demucs vocal stem | \"natural dramatic pauses detected at: 2s (11.7s pause), 20s (3.3s pause)...\" |\n| `vocal_emotion._demucs` | Demucs | Model + stem path metadata |\n| 25-emotion `emotion_classification` (was 9) | per-section rules | \"emotion signature: intense, triumphant, defiant, building\" — more granular |\n| Per-section `breathier_in_verse / strained_in_chorus` | phase-1 aggregation | \"vocal texture in verse: breathier / more intimate than average\" |\n| `tempo_consistency` → \"tight\" / \"loose\" | new consumption | \"rhythm: tight, on-beat delivery\" |\n| `onset_density` → \"busy\" / \"spacious\" | new consumption | \"high note density — busy, intricate\" |\n| `brightness` → \"dark\" / \"bright\" | new consumption | \"tonal character: dark warm tone, rolled-off highs\" |\n| `instrument_hints` → likely_* | new consumption | \"instruments detected: electronic / synthetic textures\" |\n| `music_generation_hints[]` | now injected | \"Vary arrangement density — fuller for peaks, reduced for valleys — but always keep at least 2 instruments active\" |\n\n> The `image.caption` / `image.text_in_image` / `image.faces` / `video.camera_motion` / `video.vlm_captions[]` rows from v0.3.0 are omitted here because the image and video pipelines were removed in v1.5.0.\n\n### Mashup Improvements (cumulative)\n\n- Key compatibility scoring (circle of fifths) — automatic\n- BPM range scoring — identifies mashup candidates\n- Transposition suggestions when keys clash\n- `mashup compatibility:` line appended to final prompt\n- `mashup_plan.style_notes` + `instrument_prompt_additions` now consumed\n\n### Performance / Bug Fixes\n\n- parselmouth 0.4.x: get_value_at_time / get_value_at_xy (was get_value_in_frame)\n- ffmpeg 8.x image2 muxer: per-frame extraction workaround\n- pylette 5.1+ capital-P package import\n- open_clip 3.3 3-tuple return + get_tokenizer()\n- demucs 4.x apply_model() entry point\n- ffmpeg 8.x `--version` flag dropped: check_tool tries `-version` fallback\n\n### What's New in v0.1.0 — historical, pre-v1.5.0\n\nThe emotion pipeline has been significantly extended. If you ran analysis on a previous version, re-run with the latest scripts to get the new fields.\n\n### New Analysis Outputs (v0.1.0 baseline) — historical\n\n| Field | What it means | Prompt effect |\n|---|---|---|\n| `chord_progression` | Auto-detected chord symbols (Am, F, C, G) | `\"chord progression: Am - F - C - G\"` |\n| `loudness_profile.LRA` | Perceptual dynamic range in LU | \"wide dynamic range\" or \"compressed wall-of-sound\" |\n| `song_structure` | Neural segment labels (intro/verse/chorus) | More accurate `[Verse]` / `[Chorus]` placement |\n| `harmonic_percussive.classification` | Smooth vs percussive texture | \"smooth melodic\" or \"driving rhythmic\" |\n| `vocal_quality.voice_quality` | Smooth vs rough vs pressed | \"clean polished\" or \"raw gritty\" |\n| `emotion_sections[].vocal_register` | Chest / head / falsetto | \"airy falsetto\" / \"full chest voice\" per section |\n| `emotion_sections[].rhythm_feel` | Swing / straight | \"swing feel\" / \"straight eighth-note\" per section |\n| `emotion_sections[].harmony_quality` | Consonant / tense / rich | \"consonant\" / \"dissonant\" / \"rich\" per section |\n| `clap_classification` | Zero-shot genre / mood / instruments | APPENDED to template defaults (not replaced) |\n| `emotion_classification` (per section) | Top-3 detected emotions | \"emotion signature from analysis: ...\" |\n\n### New Scripts (v0.1.0 baseline) — historical\n\n- `extract_lyrics_whisper.py` — Whisper-based ASR with section tagging\n- `analysis_orchestrator.py` — Single entry point; in v1.5.0 audio-only\n\n> `extract_video_features.py` and `analyze_image.py` from v0.1.0 were removed in v1.5.0.\n\n### Mashup Improvements (v0.1.0 baseline)\n\n- Key compatibility scoring (circle of fifths) is now automatic\n- BPM range scoring identifies mashup candidates\n- Transposition suggestions when keys clash\n- `mashup compatibility:` line appended to final prompt\n\n## The \"Emotion Recipe\" Pattern\n\nFor each target emotion, the generation prompt and lyrics need four coordinated elements:\n\n1. **Vocal descriptors**: 2-3 adjectives that evoke the emotion in the prompt's `prompt` field\n2. **Lyrics formatting**: structural choices (elongation, repetition, pause placement) in the `lyrics` body\n3. **Arrangement**: instrumentation density, dynamics, key, BPM\n4. **Section-level instructions**: where the emotion peaks, where it releases\n\nA complete recipe specifies all four. Using only one or two is what produces \"emotionally flat\" output.\n\n## Quick Reference: Emotion → Descriptor Set\n\n| Target emotion | Vocal descriptors | Key/BPM | Arrangement feel |\n|---|---|---|---|\n| Joy | bright, energetic, smiling, celebratory | major / 110-130 | dense, hand claps, bright synths |\n| Desperation | strained, raw, falling, urgent | minor / 70-90 | slow build, dramatic strings, sparse→dense |\n| Melancholy | breathy, low, slow, wistful | minor / 60-80 | sparse, piano+strings, low BPM |\n| Triumph | powerful, building, declarative | major / 100-130 | full orchestra, choir in climax, building |\n| Yearning | breathy, high, sustained, longing | major or minor / 70-90 | high register, sustained pads |\n| Anger | aggressive, shouted, sharp, raw | minor / 110-150 | distorted, punchy drums, dense |\n| Vulnerability | whispered, fragile, intimate, hesitant | minor / 60-80 | very sparse, single instrument |\n| Confidence | strong, clear, stable, direct | major / 110-130 | punchy, driving, full band |\n| Nostalgia | warm, gentle, distant, wistful | major / 70-90 | vintage feel, soft, faded |\n| Anxious | tense, sharp, rapid, unstable | minor / 130-160 | irregular rhythm, dissonant |\n\n## Full Recipes\n\n### Recipe: Joy\n\n**Vocal descriptors:** bright, energetic, smiling, upbeat, celebratory\n\n**Lyrics formatting:**\n- Short syllables, lots of repeated words\n- Exclamations: `\"oh!\"`, `\"hey!\"`, `\"yeah!\"`\n- Rising inflections on questions\n- Examples:\n  ```\n  [Verse]\n  Sun is shiiiiining\n  And I'm feeliiiiing fine\n\n  [Chorus]\n  Oh oh oh! Yeah yeah yeah!\n  This is my dayyy!\n  ```\n\n**Arrangement:**\n- Key: major\n- BPM: 110-130\n- Instrumentation: bright synths or acoustic guitar, hand claps, tambourine\n- All instruments always playing\n- Section intensity: chorus is the highest, verses stay bright\n\n**Section structure:**\n- Intro: brief, bright hook\n- Verses: bright but lower energy\n- Pre-Chorus: build energy\n- Chorus: maximum energy, hook\n- Bridge: contrast (slight key change, drop)\n- Final chorus: biggest energy\n\n**Anti-sparse:** all instruments always playing, dense pop production, hand claps layer in.\n\n---\n\n### Recipe: Desperation\n\n**Vocal descriptors:** strained, raw, falling intonation, urgent, with emotional weight\n\n**Lyrics formatting:**\n- Long stretched vowels in emotional words\n- Repetition with increasing intensity\n- Trailing downward vowels\n- Examples:\n  ```\n  [Verse]\n  I caaaan't goooo on with yoooou\n  Nooooo, pleaseeee, don't leeeave\n\n  [Pre Chorus]\n  Don't youuuu, don't youuuu, don't youuuu goooo\n\n  [Chorus]\n  I'm beggiiiiiing, I'm beggiiiiiing\n  Stay with meeeee\n  ```\n\n**Arrangement:**\n- Key: minor\n- BPM: 70-90\n- Instrumentation: dramatic strings, sparse piano, building to dense\n- Wide dynamic range\n- Section intensity: climax at bridge or final chorus\n\n**\n\nArchive v1.5.1: 44 files, 230429 bytes\n\nFiles: README.md (6468b), references/advanced-audio-analysis.md (26306b), references/changelog.md (12368b), references/cover-workflow.md (11095b), references/emotion-analysis.md (29021b), references/emotion-delivery.md (38670b), references/error-handling.md (19280b), references/examples.md (10851b), references/free-tool-inputs.md (21747b), references/lyrics-generation.md (8546b), references/mashup-workflow.md (14829b), references/minimax-generation-caveats.md (5833b), references/mmx-flags-reference.md (16959b), references/orchestrator-quickstart.md (5867b), references/setup-and-preflight.md (10500b), references/short-prompt-recipes.md (2056b), scripts/_analysis_cache.py (3466b), scripts/_audio_features.py (5329b), scripts/_key_compat.py (9604b), scripts/analysis_orchestrator.py (11901b), scripts/analyze_audio.py (6686b), scripts/analyze_two_songs.py (13152b), scripts/analyze_vocal_emotion.py (61198b), scripts/batch_cover.py (6439b), scripts/check_environment.py (4116b), scripts/classify_instruments.py (8078b), scripts/compute_audio_embedding.py (11217b), scripts/emotion_to_prompt.py (61593b), scripts/extract_lyrics_whisper.py (8755b), scripts/extract_melody.py (11778b), scripts/extract_stems.py (9483b), scripts/finalize_track.sh (590b), scripts/generate_with_retry.py (23025b), scripts/hybrid_remix.py (6055b), scripts/lint_lyrics.py (4578b), scripts/lint_music_request.py (38221b), scripts/per_stem_analysis.py (4643b), scripts/smoke_test.py (83862b), scripts/track_beats.py (7240b), scripts/verify_cloud_output.sh (3974b), scripts/verify_lyrics_alignment.py (3768b), skill-card.md (3193b), SKILL.md (39773b), _meta.json (138b)\n\nArchive v1.5.0: 44 files, 229506 bytes\n\nFiles: README.md (5679b), references/advanced-audio-analysis.md (26306b), references/changelog.md (12052b), references/cover-workflow.md (11095b), references/emotion-analysis.md (29021b), references/emotion-delivery.md (38670b), references/error-handling.md (19280b), references/examples.md (10851b), references/free-tool-inputs.md (21747b), references/lyrics-generation.md (8546b), references/mashup-workflow.md (14829b), references/minimax-generation-caveats.md (5833b), references/mmx-flags-reference.md (16959b), references/orchestrator-quickstart.md (5867b), references/setup-and-preflight.md (10500b), references/short-prompt-recipes.md (2056b), scripts/_analysis_cache.py (3466b), scripts/_audio_features.py (5329b), scripts/_key_compat.py (9604b), scripts/analysis_orchestrator.py (11901b), scripts/analyze_audio.py (6686b), scripts/analyze_two_songs.py (13152b), scripts/analyze_vocal_emotion.py (61198b), scripts/batch_cover.py (6439b), scripts/check_environment.py (4116b), scripts/classify_instruments.py (8078b), scripts/compute_audio_embedding.py (11217b), scripts/emotion_to_prompt.py (61593b), scripts/extract_lyrics_whisper.py (8755b), scripts/extract_melody.py (11778b), scripts/extract_stems.py (9483b), scripts/finalize_track.sh (590b), scripts/generate_with_retry.py (23025b), scripts/hybrid_remix.py (6055b), scripts/lint_lyrics.py (4578b), scripts/lint_music_request.py (38221b), scripts/per_stem_analysis.py (4643b), scripts/smoke_test.py (83862b), scripts/track_beats.py (7240b), scripts/verify_cloud_output.sh (3974b), scripts/verify_lyrics_alignment.py (3768b), skill-card.md (3411b), SKILL.md (38492b), _meta.json (138b)\n\nArchive v1.4.1: 50 files, 268107 bytes\n\nFiles: README.md (5532b), references/advanced-audio-analysis.md (26306b), references/changelog.md (9694b), references/cover-workflow.md (10456b), references/emotion-analysis.md (29082b), references/emotion-delivery.md (38741b), references/error-handling.md (19284b), references/examples.md (10048b), references/free-tool-inputs.md (26047b), references/lyrics-generation.md (10202b), references/mashup-workflow.md (13653b), references/minimax-generation-caveats.md (5833b), references/mmx-flags-reference.md (16959b), references/orchestrator-quickstart.md (6459b), references/setup-and-preflight.md (9998b), references/short-prompt-recipes.md (2056b), scripts/_analysis_cache.py (3466b), scripts/_audio_features.py (5329b), scripts/_key_compat.py (9604b), scripts/analysis_orchestrator.py (33967b), scripts/analyze_audio.py (6686b), scripts/analyze_image.py (17736b), scripts/analyze_two_songs.py (13152b), scripts/analyze_vocal_emotion.py (61198b), scripts/audio_sources.py (5836b), scripts/batch_cover.py (6439b), scripts/check_environment.py (4278b), scripts/classify_instruments.py (7083b), scripts/compute_audio_embedding.py (11217b), scripts/download_mx3.py (2747b), scripts/download_youtube.py (10486b), scripts/emotion_to_prompt.py (61593b), scripts/extract_lyrics_whisper.py (8755b), scripts/extract_melody.py (11778b), scripts/extract_stems.py (9483b), scripts/extract_video_features.py (20134b), scripts/fetch_lyrics_web.py (11888b), scripts/finalize_track.sh (590b), scripts/generate_with_retry.py (23025b), scripts/hybrid_remix.py (6055b), scripts/lint_lyrics.py (4578b), scripts/lint_music_request.py (39093b), scripts/per_stem_analysis.py (4643b), scripts/smoke_test.py (118772b), scripts/track_beats.py (7240b), scripts/verify_cloud_output.sh (3974b), scripts/verify_lyrics_alignment.py (3768b), skill-card.md (3477b), SKILL.md (39911b), _meta.json (138b)\n\nArchive v1.4.0: 48 files, 261202 bytes\n\nFiles: README.md (5330b), references/advanced-audio-analysis.md (26306b), references/changelog.md (8658b), references/cover-workflow.md (10456b), references/emotion-analysis.md (29082b), references/emotion-delivery.md (38741b), references/error-handling.md (19284b), references/examples.md (10048b), references/free-tool-inputs.md (26047b), references/lyrics-generation.md (10202b), references/mashup-workflow.md (13653b), references/minimax-generation-caveats.md (5633b), references/mmx-flags-reference.md (16959b), references/orchestrator-quickstart.md (6386b), references/setup-and-preflight.md (9998b), references/short-prompt-recipes.md (2056b), scripts/_analysis_cache.py (3466b), scripts/_audio_features.py (5329b), scripts/_key_compat.py (9604b), scripts/analysis_orchestrator.py (33967b), scripts/analyze_audio.py (6686b), scripts/analyze_image.py (17736b), scripts/analyze_two_songs.py (13152b), scripts/analyze_vocal_emotion.py (61198b), scripts/audio_sources.py (5836b), scripts/batch_cover.py (6439b), scripts/check_environment.py (4278b), scripts/classify_instruments.py (7083b), scripts/compute_audio_embedding.py (11217b), scripts/download_mx3.py (2747b), scripts/download_youtube.py (10486b), scripts/emotion_to_prompt.py (61593b), scripts/extract_lyrics_whisper.py (8755b), scripts/extract_melody.py (11778b), scripts/extract_stems.py (9483b), scripts/extract_video_features.py (20134b), scripts/fetch_lyrics_web.py (11888b), scripts/finalize_track.sh (590b), scripts/generate_with_retry.py (22352b), scripts/hybrid_remix.py (6055b), scripts/lint_music_request.py (34055b), scripts/per_stem_analysis.py (4643b), scripts/smoke_test.py (114185b), scripts/track_beats.py (7240b), scripts/verify_cloud_output.sh (3974b), skill-card.md (3433b), SKILL.md (38887b), _meta.json (138b)\n\nArchive v1.3.0: 45 files, 252340 bytes\n\nFiles: README.md (5330b), references/advanced-audio-analysis.md (26306b), references/changelog.md (7541b), references/cover-workflow.md (10007b), references/emotion-analysis.md (29067b), references/emotion-delivery.md (38741b), references/error-handling.md (19284b), references/examples.md (9960b), references/free-tool-inputs.md (24464b), references/lyrics-generation.md (10202b), references/mashup-workflow.md (13653b), references/minimax-generation-caveats.md (5633b), references/mmx-flags-reference.md (16959b), references/orchestrator-quickstart.md (5749b), references/setup-and-preflight.md (10744b), references/short-prompt-recipes.md (2056b), scripts/_analysis_cache.py (3466b), scripts/_audio_features.py (5329b), scripts/_key_compat.py (9604b), scripts/analysis_orchestrator.py (33967b), scripts/analyze_audio.py (6686b), scripts/analyze_image.py (17736b), scripts/analyze_two_songs.py (13152b), scripts/analyze_vocal_emotion.py (61198b), scripts/audio_sources.py (5836b), scripts/check_environment.py (4278b), scripts/classify_instruments.py (7083b), scripts/compute_audio_embedding.py (11217b), scripts/download_mx3.py (2747b), scripts/download_youtube.py (10486b), scripts/emotion_to_prompt.py (61593b), scripts/extract_lyrics_whisper.py (8755b), scripts/extract_melody.py (11778b), scripts/extract_stems.py (9483b), scripts/extract_video_features.py (20134b), scripts/fetch_lyrics_web.py (11888b), scripts/finalize_track.sh (590b), scripts/generate_with_retry.py (22352b), scripts/lint_music_request.py (34055b), scripts/smoke_test.py (108226b), scripts/track_beats.py (7240b), scripts/verify_cloud_output.sh (3974b), skill-card.md (3786b), SKILL.md (37979b), _meta.json (138b)\n\nArchive v1.1.0: 43 files, 240461 bytes\n\nFiles: README.md (4805b), references/advanced-audio-analysis.md (26306b), references/changelog.md (6541b), references/cover-workflow.md (10007b), references/emotion-analysis.md (29067b), references/emotion-delivery.md (38741b), references/error-handling.md (19284b), references/examples.md (9960b), references/free-tool-inputs.md (24464b), references/lyrics-generation.md (10202b), references/mashup-workflow.md (13653b), references/minimax-generation-caveats.md (2920b), references/mmx-flags-reference.md (16409b), references/orchestrator-quickstart.md (5749b), references/setup-and-preflight.md (10744b), scripts/_analysis_cache.py (3466b), scripts/_audio_features.py (5329b), scripts/_key_compat.py (9604b), scripts/analysis_orchestrator.py (33967b), scripts/analyze_audio.py (6686b), scripts/analyze_image.py (17736b), scripts/analyze_two_songs.py (13152b), scripts/analyze_vocal_emotion.py (61198b), scripts/audio_sources.py (5836b), scripts/check_environment.py (3480b), scripts/classify_instruments.py (7083b), scripts/compute_audio_embedding.py (11217b), scripts/download_mx3.py (2747b), scripts/download_youtube.py (10486b), scripts/emotion_to_prompt.py (61593b), scripts/extract_lyrics_whisper.py (8755b), scripts/extract_melody.py (11778b), scripts/extract_stems.py (9483b), scripts/extract_video_features.py (20134b), scripts/fetch_lyrics_web.py (11888b), scripts/finalize_track.sh (590b), scripts/generate_with_retry.py (9024b), scripts/lint_music_request.py (34055b), scripts/smoke_test.py (74805b), scripts/track_beats.py (7240b), skill-card.md (3568b), SKILL.md (35704b), _meta.json (138b)\n\nArchive v1.0.1: 35 files, 203961 bytes\n\nFiles: README.md (4174b), references/advanced-audio-analysis.md (24336b), references/cover-workflow.md (8971b), references/emotion-analysis.md (28413b), references/emotion-delivery.md (38741b), references/error-handling.md (10648b), references/examples.md (8660b), references/free-tool-inputs.md (22915b), references/lyrics-generation.md (8176b), references/mashup-workflow.md (13653b), references/mmx-flags-reference.md (7812b), scripts/_analysis_cache.py (3466b), scripts/_audio_features.py (5329b), scripts/_key_compat.py (9604b), scripts/analysis_orchestrator.py (25435b), scripts/analyze_audio.py (6686b), scripts/analyze_image.py (17677b), scripts/analyze_two_songs.py (13152b), scripts/analyze_vocal_emotion.py (60679b), scripts/check_environment.py (3480b), scripts/classify_instruments.py (7083b), scripts/compute_audio_embedding.py (9926b), scripts/download_youtube.py (8440b), scripts/emotion_to_prompt.py (61593b), scripts/extract_lyrics_whisper.py (6297b), scripts/extract_melody.py (11778b), scripts/extract_stems.py (9483b), scripts/extract_video_features.py (20143b), scripts/fetch_lyrics_web.py (11888b), scripts/lint_music_request.py (26963b), scripts/smoke_test.py (31400b), scripts/track_beats.py (7240b), skill-card.md (3037b), SKILL.md (66315b), _meta.json (138b)\n\nArchive v1.0.0: 35 files, 203955 bytes\n\nFiles: README.md (4186b), references/advanced-audio-analysis.md (24336b), references/cover-workflow.md (8971b), references/emotion-analysis.md (28413b), references/emotion-delivery.md (38741b), references/error-handling.md (10648b), references/examples.md (8660b), references/free-tool-inputs.md (22915b), references/lyrics-generation.md (8176b), references/mashup-workflow.md (13653b), references/mmx-flags-reference.md (7812b), scripts/_analysis_cache.py (3466b), scripts/_audio_features.py (5329b), scripts/_key_compat.py (9604b), scripts/analysis_orchestrator.py (25435b), scripts/analyze_audio.py (6686b), scripts/analyze_image.py (17677b), scripts/analyze_two_songs.py (13152b), scripts/analyze_vocal_emotion.py (60679b), scripts/check_environment.py (3480b), scripts/classify_instruments.py (7083b), scripts/compute_audio_embedding.py (9926b), scripts/download_youtube.py (8440b), scripts/emotion_to_prompt.py (61593b), scripts/extract_lyrics_whisper.py (6297b), scripts/extract_melody.py (11778b), scripts/extract_stems.py (9483b), scripts/extract_video_features.py (20143b), scripts/fetch_lyrics_web.py (11888b), scripts/lint_music_request.py (26963b), scripts/smoke_test.py (31400b), scripts/track_beats.py (7240b), skill-card.md (3145b), SKILL.md (66327b), _meta.json (138b)","readmeExcerpt":"Skill: Music Craft — MiniMax Owner: luischarro Summary: MiniMax-native music generation for OpenClaw — cover and style transfer that preserves melody, two-song mashups, AI lyrics generation and edit, emotion-driven prompt engineering, and per-flag mmx CLI control over BPM, key, structure, and avoid lists. Extends music-craft with MiniMax Music 2.6 features. Tags: latest:1.6.0 Version history: v1.6.0 | 2026-07-30T11:3","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"mmx music generate \\\n  --prompt \"<production-sheet prompt>\" \\\n  --lyrics-file lyrics.txt \\\n  --model music-2.6 \\\n  --bpm 96 --key \"D major\" --structure \"intro-verse-chorus-verse-chorus-bridge-chorus-outro\" \\\n  --vocals \"<vocal description>\" --genre \"<genre>\" --mood \"<mood>\" --instruments \"<instruments>\" \\\n  --avoid \"<what to avoid>\" --out output.mp3"},{"language":"python","snippet":"import librosa\n\ny, sr = librosa.load('/tmp/song.wav')\ntempo, beats = librosa.beat.beat_track(y=y, sr=sr)\noenv = librosa.onset.onset_strength(y=y, sr=sr, feature_mode='spectral')\ntempogram = librosa.feature.tempogram(oenv=oenv, sr=sr)\n# Ratio of beat-strength deviations indicates swing"},{"language":"python","snippet":"y, sr = librosa.load('/tmp/song.wav')\nchrom = librosa.feature.chroma_cqt(y=y, sr=sr)\ntonnetz = librosa.feature.tonnetz(chroma=chrom)\n# tonnetz shape: (6, frames) — dims represent 5ths, minor/major"},{"language":"python","snippet":"y, sr = librosa.load('/tmp/song.wav')\ncontrast = librosa.feature.spectral_contrast(y=y, sr=sr, n_bands=6)\n# contrast shape: (7, frames) — 6 bands + overall valley mean"},{"language":"python","snippet":"y, sr = librosa.load('/tmp/song.wav')\nharmonic, percussive = librosa.effects.hpss(y=y)"},{"language":"python","snippet":"import parselmouth\n\nsound = parselmouth.Sound('/tmp/vocal.wav')\nformants = sound.to_formant_burg()\n# F1/F2 at voiced segments indicate register"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: music-craft-minimax\nversion: 1.6.0\ndescription: MiniMax-native music generation for OpenClaw — cover and style transfer that preserves melody, two-song mashups, AI lyrics generation and edit, emotion-driven prompt engineering, and per-flag mmx CLI control over BPM, key, structure, and avoid lists. Extends music-craft with MiniMax Music 2.6 features.\nmetadata: '{\"openclaw\":{\"requires\":{\"env\":[\"MINIMAX_API_KEY\"],\"bins\":[\"python3\",\"ffmpeg\",\"mmx\"]},\"primaryEnv\":\"MINIMAX_API_KEY\",\"emoji\":\"\\ud83c\\udfb6\",\"homepage\":\"https://github.com/LuisCharro/skills/tree/main/publish/music-craft-minimax\",\"envVars\":[{\"name\":\"MINIMAX_API_KEY\",\"required\":true,\"description\":\"API key for the MiniMax Music 2.6 token plan. Required for cover, mashup, lyrics generation, and mmx flag control.\"}]}}'\n---\n# Music Craft — MiniMax\n\n## What is Music Craft — MiniMax?\n\nMusic Craft — MiniMax is the **MiniMax-native power-user upgrade** of [`music-craft`](../music-craft/). It unlocks the features that need the MiniMax Music 2.6 token plan: cover and style transfer that preserves melody, two-song mashups, AI lyrics generation with edit-iter, emotion analysis on input audio, and precise per-flag control over BPM, key, structure, and avoid lists via the `mmx` CLI.\n\n## MiniMax-specific features\n\n- **Cover and style transfer** — preserve the melody of Song X, apply the style of Song Y.\n- **Two-song mashup** — Song A's lyrics and emotion + Song B's style, in one output.\n- **Lyrics generation API** — `write_full_song` for blank-page or `edit` for revisions, with structure tags.\n- **Emotion analysis** — intensity, vocal speed, pitch bends, and 25+ emotion classes drive your prompt.\n- **`mmx` per-flag control** — `--bpm`, `--key`, `--structure`, `--vocals`, `--genre`, `--mood`, `--instruments`, `--avoid` as separate flags; prompt and flags lint-checked before generation.\n- **Quota-aware batch runs** — 5-hour rolling window check before any cloud batch.\n\n## Why use this instead of music-craft?\n\nUse **music-craft** when you want a structured default that picks the backend for you, or when exact-duration vocal tracks matter. Use **music-craft-minimax** when you need a MiniMax-only feature — cover, mashup, lyrics API, emotion-driven prompting — or when you specifically want `mmx` flag-level control over BPM, key, structure, or an avoid list.\n\n## Quick Start\n\n1. Decide which MiniMax-only feature you need (cover, mashup, lyrics API, emotion analysis).\n2. Provide one or two source audio files / lyrics / emotion inputs as the skill requests.\n3. The skill picks the right `mmx` flag combination and calls the API.\n4. It lints flags + prompt for conflicts, then runs sequentially (one job at a time, never parallel).\n5. It verifies duration, file size, and quota headroom before delivering.\n\n## When To Use\n\nUse this skill when the task involves:\n\n- generating a cover of an existing song with a different style (chanson version of a rock track, reggaeton version of a pop hit, and so on). Source must b"},{"path":"README.md","content":"# Music Craft — MiniMax\n\nAdvanced music generation for OpenClaw, using the MiniMax Music 2.6 token plan.\n\nCurrent release: v1.6.0.\n\nThis is the **power-user upgrade** of [`music-craft`](../music-craft/). It does everything that skill does, plus the features that require MiniMax:\n\n- Cover and style transfer (preserves melody)\n- Two-song mashup (content + style)\n- Lyrics generation API\n- Emotion-driven prompt engineering\n- Fine control via `mmx` CLI (BPM, key, structure, avoid list as separate flags)\n- Production helpers: prompt/lyrics linting, retry wrapper, MP3 defaults, lyrics-alignment verification, and loudness finalization\n\nCloud generation is for speed and MiniMax-native workflows, not exact duration.\nThe 2026-06-12 field run saved 18/18 cloud outputs, but durations ranged from\n57-135% of requested length. Use [`music-craft`](../music-craft/) with local\nACE-Step when exact duration matters.\n\n## Data and consent\n\nThis skill may send prompts, lyrics, reference audio, and cover/mashup inputs to MiniMax when the user chooses cloud generation. Audio input must be a **local file path** — URLs are not accepted in v1.5.0+. Use the private `music-source-fetch` skill to fetch audio by title first. There is no image, face, OCR, or VLM pipeline.\n\n## Licensing and commercial use\n\nClawHub publishes this skill bundle under MIT-0, so the skill instructions and\nbundled helper code may be used, modified, and redistributed commercially\nwithout attribution. That does not grant MiniMax rights or rights to the\ninputs or outputs. Each operator must use their own MiniMax account/API key\nand accept the current terms for the exact API/Token Plan product. The API/Open Platform and\nconsumer web/app terms are separate; a subscription or third-party claim of\n“commercial use” is not enough to establish the rights for this CLI/API\nworkflow. Verify the applicable terms before commercial release. Inputs must\nbe owned or properly licensed, and generated audio is not guaranteed to have\nexclusive copyright protection in every jurisdiction.\n\n## When to use\n\nUse this skill when the task involves:\n\n- cover or style transfer from a local reference audio file (URLs not accepted; use `music-source-fetch` to fetch by title)\n- two-song mashup\n- lyrics generation via the MiniMax API\n- emotion analysis on input audio\n- fine parameter control via `mmx` CLI\n- fast standard cloud iterations where approximate duration is acceptable\n\nStay in [`music-craft`](../music-craft/) when the user only needs standard song generation, a text-only style reference without melody preservation, or provider-agnostic prompting. Use MiniMax when the request uses a local reference audio file, needs cover/mashup, lyrics API iteration, emotion-driven prompting, or exact `mmx` flags.\n\n## First Response\n\n- **Cover from local audio**: default to the one-step cover path; ask only if the user wants translated lyrics, edited ASR lyrics, or new lyrics.\n- **Style transfer only**: do not use cover unless the melody must s"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7fkh362pnj2zckq8pcxfaxr9821zv8\",\n  \"slug\": \"music-craft-minimax\",\n  \"version\": \"1.6.0\",\n  \"publishedAt\": 1785411554548\n}"},{"path":"references/advanced-audio-analysis.md","content":"# Advanced Audio Analysis with Free Tools\n\nThe base emotion analysis in [`emotion-analysis.md`](emotion-analysis.md) uses `librosa` + `parselmouth` (Praat) + `scipy` to extract per-section features. Those cover the core 30 emotions. This file covers **advanced free tools** that go deeper: more accurate pitch, source separation, note transcription, music theory, and pre-trained emotion / genre / mood classifiers.\n\nAll tools listed here are **open source** (BSD, MIT, Apache, or GPL). All can be installed via `pip` (some need system dependencies like `ffmpeg`). They are **optional** — the base skill works without them. Install only what you need.\n\n---\n\n## Currently Implemented Advanced Features (v0.1.0)\n\nThe following advanced analysis features **are now implemented** in the MiniMax layer. Each is available as an optional dependency and is used in the prompt pipeline to enrich generation prompts with precise audio characteristics.\n\n| Feature | Library | What it detects | Pip package |\n|---|---|---|---|\n| **tempogram_ratio** | librosa | Swing / groove feel ( eighth-note swing ratio ) | `librosa` (already in base) |\n| **tonnetz** | librosa | Tonal harmony quality — diatonic/chromatic, consonant/dissonant | `librosa` (already in base) |\n| **spectral_contrast** | librosa | Valley-to-peak energy ratio per frequency band (timbre) | `librosa` (already in base) |\n| **HPSS** | librosa.effects.hpss | Harmonic-percussive source separation | `librosa` (already in base) |\n| **Formant analysis** | parselmouth | Vocal register detection — chest / head / falsetto | `parselmouth` (already in base) |\n| **HNR** | parselmouth | Harmonics-to-noise ratio — breathiness vs clarity | `parselmouth` (already in base) |\n| **Jitter / Shimmer** | parselmouth | Microperturbations in pitch and amplitude — voice stability | `parselmouth` (already in base) |\n| **LUFS / LRA** | pyloudnorm | Perceptual loudness (integrated LUFS) and loudness range (LRA) | `pyloudnorm` |\n| **Chord progression** | autochord | Automatic chord symbol extraction | `autochord` |\n| **Song structure** | allin1 | Neural structure segmentation — intro / verse / chorus / bridge / outro | `allin1` |\n| **CLAP classification** | transformers | Zero-shot genre / mood / instrument / era classification | `transformers` |\n\n### tempogram_ratio (librosa)\n\nDetects swing and groove by measuring the deviation of note onsets from a perfectly regular grid. A ratio near 1.0 is straight; above ~1.2 indicates a noticeable swing feel.\n\n**Pipeline use**: When the user wants \"groovy\" or \"laid-back\" output, `tempogram_ratio` quantifies the swing intensity to pass a precise BPM/swing hint to `mmx`.\n\n```python\nimport librosa\n\ny, sr = librosa.load('/tmp/song.wav')\ntempo, beats = librosa.beat.beat_track(y=y, sr=sr)\noenv = librosa.onset.onset_strength(y=y, sr=sr, feature_mode='spectral')\ntempogram = librosa.feature.tempogram(oenv=oenv, sr=sr)\n# Ratio of beat-strength deviations indicates swing\n```\n\n### tonnetz (librosa)\n\nComputes the ton"},{"path":"references/changelog.md","content":"# Changelog\n\nRelease history for music-craft-minimax. Operating guidance lives in the\ntopic references; this file is history only.\n# v1.6.0\n\n- Reordered `SKILL.md` for discoverability: added hero sections\n  (\"What is Music Craft — MiniMax?\", \"MiniMax-specific features\",\n  \"Why use this instead of music-craft?\", \"Quick Start\") at the top\n  of the body. MiniMax-only features (cover, mashup, lyrics API,\n  emotion-driven prompting, `mmx` flag control, quota-aware batches)\n  are now the first thing a visitor reads.\n- Moved the licensing and commercial-use gate to the end of `SKILL.md`\n  (preserving all legal content verbatim, including the \"Cloud duration\n  is approximate\" routing warning); data / consent sits just above it.\n- Updated the frontmatter `description:` to anchor-keyword pack\n  (cover and style transfer, two-song mashup, lyrics generation API,\n  emotion-driven prompt engineering, mmx CLI, Extends music-craft)\n  for clearer ClawHub routing.\n- Removed the old \"Quick Start with the Orchestrator\" section; the\n  new hero \"Quick Start\" supersedes it.\n- No behaviour, no env vars, no bins — pure content reordering and\n  versioning.\n\n\n\n# v1.5.1\n\n- Added a licensing and commercial-use gate for MiniMax API/CLI workflows.\n- Clarified the difference between MiniMax Open Platform/API terms and\n  consumer app/web terms.\n- Required each operator to use their own account/API key and verify current\n  product-specific commercial-use terms before release.\n\n## v1.5.0\n\nv1.5.0 is a **breaking change** that isolates all internet-download code\ninto the new private `music-source-fetch` skill and removes the album-art\n/ face / OCR / VLM image pipeline entirely. The published skill is now\naudio-only and accepts only local file paths.\n\n**Removed (moved to `publish/music-source-fetch/`):**\n- `scripts/download_youtube.py`, `scripts/download_mx3.py`,\n  `scripts/fetch_lyrics_web.py`, `scripts/audio_sources.py`\n- `analysis_orchestrator.py` flags: `--youtube`, `--audio-url`,\n  `--lyrics-source {web,auto}`\n- LRCLib web lyrics lookup; Whisper on the local file is the only lyrics\n  source\n\n**Deleted entirely (no replacement):**\n- `scripts/analyze_image.py`, `scripts/extract_video_features.py`\n- `analysis_orchestrator.py` flags: `--image`, `--video`, `--vlm`,\n  `--ocr`, `--faces`\n- Album-art color palette, face detection, OCR, VLM captioning flows\n\n**Changed:**\n- `check_environment.py` no longer recommends `yt-dlp`; drops `cv2`/`PIL`\n  from optional imports\n- `lint_music_request.py` URL detection now emits a `url_not_accepted`\n  warning and routes to `needs_clarification`; only local file paths\n  route to `minimax_cover`\n- SKILL.md \"Audio Source Fallback Order\" replaced with \"Audio Source\n  (Local Only)\"\n- `cover-workflow.md` and `mashup-workflow.md` add explicit cloud\n  transmission consent paragraphs (audit SQP-2)\n- `examples.md` neutralises hardcoded language/locale defaults (audit SQP-3)\n- README narrows the auto-load trigger (audit SQP-1)\n- Frontmatter `metadata.openclaw.r"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":2252,"uniquenessScore":40,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T11:16:31.985Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T11:16:31.985Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T14:16:54.981Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}