{"id":"88082afa-cae0-421e-a164-64ce18640e2a","entityType":"agent","slug":"clawhub-luischarro-music-craft","name":"music-craft","canonicalUrl":"https://www.xpersona.co/agent/clawhub-luischarro-music-craft","canonicalPath":"/agent/clawhub-luischarro-music-craft","generatedAt":"2026-10-11T10:52:12.120Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T08:23:16.095Z","emptyReason":null},"description":"Generate songs, instrumentals, or lyrics-driven tracks through a structured OpenClaw-native workflow with anti-sparse prompt engineering and quality verification. Provider-agnostic — works with any music backend the runtime exposes (ACE-Step, MusicGen, Stable Audio, mmx).","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.1K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s171padhas9w52sdjc3110vhss83j8jp:music-craft","sourceUrl":"https://clawhub.ai/luischarro/music-craft","homepage":"https://clawhub.ai/luischarro/skills/music-craft","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/luischarro/music-craft","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/luischarro/skills/music-craft","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":61,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"music-craft technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T08:23:16.095Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T08:23:16.095Z","emptyReason":null},"stars":null,"forks":null,"downloads":1113,"packageName":null,"latestVersion":"1.6.0","tractionLabel":"1.1K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T08:23:16.082Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T08:23:16.095Z","lastCrawledAt":"2026-10-11T08:23:16.082Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T08:23:16.082Z","lastVerifiedAt":null,"highlights":[{"version":"1.6.0","createdAt":"2026-07-30T11:36:34.536Z","changelog":"v1.6.0: discoverability reorder — hero sections (What is / Key Capabilities / Why use / Quick Start) at top of SKILL.md, licensing/Commercial-use gate moved to bottom. No behaviour, no env vars, no bins.","fileCount":36,"zipByteSize":175864},{"version":"1.5.1","createdAt":"2026-07-28T13:06:18.009Z","changelog":"v1.5.1: document ClawHub MIT-0 scope, mark MusicGen weights non-commercial, and add backend licensing and commercial-use gates","fileCount":28,"zipByteSize":117153},{"version":"1.5.0","createdAt":"2026-06-14T05:44:55.608Z","changelog":"v1.5.0: BREAKING — local audio files only, image pipeline removed, LRCLib moved to music-source-fetch","fileCount":28,"zipByteSize":115624},{"version":"1.4.1","createdAt":"2026-06-13T15:20:13.854Z","changelog":"v1.4.1: add lyrics tag whitelist linting, transcript-vs-lyrics verification, and safer delivery guidance","fileCount":27,"zipByteSize":117697},{"version":"1.4.0","createdAt":"2026-06-13T11:46:35.024Z","changelog":"v1.4.0: clarify local arranger workflows, generalize Windows guidance, and remove maintainer-specific/private details from published docs","fileCount":25,"zipByteSize":113330},{"version":"1.3.0","createdAt":"2026-06-13T05:05:27.759Z","changelog":"v1.3.0: tighten exact-duration ACE-Step guidance, expand local workflow references, and sharpen source-intake and structure-tag instructions","fileCount":21,"zipByteSize":104853},{"version":"1.1.0","createdAt":"2026-06-11T16:44:07.636Z","changelog":"v1.1.0: add vocal-confirm and target-length gates, clarify exact-duration routing, and tighten source-audio intake guidance","fileCount":18,"zipByteSize":98415},{"version":"1.0.1","createdAt":"2026-06-09T22:51:17.768Z","changelog":"v1.0.1: Rename display name from 'OpenClaw Music Workflow' to 'Music Craft' for slug consistency. Bundle body unchanged from v1.0.0 (1649-line SKILL.md, 9 reference docs, 12 files, 209 KB).","fileCount":13,"zipByteSize":89535}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s171padhas9w52sdjc3110vhss83j8jp:music-craft","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s171padhas9w52sdjc3110vhss83j8jp:music-craft` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/luischarro/music-craft before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-luischarro-music-craft/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-luischarro-music-craft/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-luischarro-music-craft/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-luischarro-music-craft/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-luischarro-music-craft/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-luischarro-music-craft/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T10:52:12.116Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-luischarro-music-craft/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-luischarro-music-craft/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-luischarro-music-craft/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-luischarro-music-craft/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T08:23:16.095Z","emptyReason":null},"readme":"Skill: music-craft\n\nOwner: luischarro\n\nSummary: Generate songs, instrumentals, or lyrics-driven tracks through a structured OpenClaw-native workflow with anti-sparse prompt engineering and quality verification. Provider-agnostic — works with any music backend the runtime exposes (ACE-Step, MusicGen, Stable Audio, mmx).\n\nTags: latest:1.6.0\n\nVersion history:\n\nv1.6.0 | 2026-07-30T11:36:34.536Z | user\n\nv1.6.0: discoverability reorder — hero sections (What is / Key Capabilities / Why use / Quick Start) at top of SKILL.md, licensing/Commercial-use gate moved to bottom. No behaviour, no env vars, no bins.\n\nv1.5.1 | 2026-07-28T13:06:18.009Z | user\n\nv1.5.1: document ClawHub MIT-0 scope, mark MusicGen weights non-commercial, and add backend licensing and commercial-use gates\n\nv1.5.0 | 2026-06-14T05:44:55.608Z | user\n\nv1.5.0: BREAKING — local audio files only, image pipeline removed, LRCLib moved to music-source-fetch\n\nv1.4.1 | 2026-06-13T15:20:13.854Z | user\n\nv1.4.1: add lyrics tag whitelist linting, transcript-vs-lyrics verification, and safer delivery guidance\n\nv1.4.0 | 2026-06-13T11:46:35.024Z | user\n\nv1.4.0: clarify local arranger workflows, generalize Windows guidance, and remove maintainer-specific/private details from published docs\n\nv1.3.0 | 2026-06-13T05:05:27.759Z | user\n\nv1.3.0: tighten exact-duration ACE-Step guidance, expand local workflow references, and sharpen source-intake and structure-tag instructions\n\nv1.1.0 | 2026-06-11T16:44:07.636Z | user\n\nv1.1.0: add vocal-confirm and target-length gates, clarify exact-duration routing, and tighten source-audio intake guidance\n\nv1.0.1 | 2026-06-09T22:51:17.768Z | user\n\nv1.0.1: Rename display name from 'OpenClaw Music Workflow' to 'Music Craft' for slug consistency. Bundle body unchanged from v1.0.0 (1649-line SKILL.md, 9 reference docs, 12 files, 209 KB).\n\nv1.0.0 | 2026-06-09T22:40:43.264Z | user\n\nFirst stable release. Provider-agnostic music generation workflow with anti-sparse prompt engineering, request intake checklist, 5 worked examples, 6 retry recipes, ACE-Step 1.5 (3 quality tiers: fast/standard/xl-mixed) with 4 verified output tracks on 24GB M3, Windows/WSL2 + corporate-proxy setup walkthrough, audio-conditioned generation (cover, repaint, reference audio), and a Dependency Consent Protocol for REQUIRED/OPTIONAL installs.\n\nArchive index:\n\nArchive v1.6.0: 36 files, 175864 bytes\n\nFiles: README.md (4331b), references/acestep-generation.md (30458b), references/acestep-shift-schedule.md (29463b), references/acestep-task-types.md (34522b), references/acestep-xl-models.md (18403b), references/advanced-diffusion-controls.md (6982b), references/audio-format-selection.md (6176b), references/audition-rubric.md (9625b), references/changelog.md (2159b), references/error-handling.md (26912b), references/examples.md (10077b), references/free-tool-inputs.md (16582b), references/input-workflows.md (17275b), references/local-ace-step-curl-template.md (2486b), references/lrc-generation.md (30982b), references/lyrics-cleanup.md (2590b), references/mps-bf16-acceleration.md (6252b), references/other-backends.md (9817b), references/prompt-formula.md (13390b), references/quality-and-revision.md (6279b), references/request-intake.md (8792b), references/setup-and-preflight.md (20709b), references/structure-tags.md (14374b), references/style-categories.md (6418b), references/user-preference-flow.md (7172b), references/wait-and-collect.md (3136b), references/windows-wsl-setup.md (11456b), scripts/extract_stems.py (4873b), scripts/lint_lyrics.py (4425b), scripts/remix_stems.py (2417b), scripts/smoke_test.py (5107b), scripts/verify_lyrics_alignment.py (3768b), scripts/wait_for_acestep.py (4564b), skill-card.md (3029b), SKILL.md (40484b), _meta.json (130b)\n\nFile v1.6.0:SKILL.md\n\n---\nname: music-craft\nversion: 1.6.0\ndescription: Generate songs, instrumentals, or lyrics-driven tracks through a structured OpenClaw-native workflow with anti-sparse prompt engineering and quality verification. Provider-agnostic — works with any music backend the runtime exposes (ACE-Step, MusicGen, Stable Audio, mmx).\nmetadata: {\"openclaw\":{\"requires\":{\"anyBins\":[\"python3\",\"python\"]},\"emoji\":\"\\ud83c\\udfb5\",\"homepage\":\"https://github.com/LuisCharro/skills/tree/main/publish/music-craft\",\"envVars\":[{\"name\":\"MUSIC_PROVIDER_API_KEY\",\"required\":false,\"description\":\"Generic API key for any music provider.\"},{\"name\":\"STABILITY_API_KEY\",\"required\":false,\"description\":\"Stability AI API key. Only needed if using Stable Audio as backend.\"}]}}\n---\n# Music Craft\n\n## What is Music Craft?\n\nMusic Craft is the **provider-agnostic entry point** for music generation in OpenClaw. It guides the model end-to-end — from intake, through a structured production-sheet prompt, to quality verification and delivery — so you get songs, instrumentals, or lyrics-driven tracks that sound complete (no sparse dropouts, no silent sections), in your chosen length, with vocals that match your lyrics.\n\n**It picks the right backend for the job.** Local ACE-Step for exact-duration vocal tracks with lyrics. MusicGen for local instrumental experiments. Stable Audio for production-friendly instrumentals. The `mmx` CLI for fast cloud generation. You pick the outcome; the skill routes.\n\n## Key Capabilities\n\n- **Production-sheet prompt engineering** — every prompt is a structured brief with genre, mood, BPM, key, instruments, structure, vocals, and an avoid list.\n- **Anti-sparse guards by default** — instruments stay playing, no a cappella dropouts, no silent mid-sections.\n- **Exact-duration vocal tracks** — local ACE-Step returns requested length to the millisecond (verified 18/18 local jobs).\n- **Structure-tagged lyrics** — canonical `[Verse]`, `[Chorus]`, `[Break]`, etc., so the model sings the song you wrote.\n- **Pre-generation linting** — duration density, tag whitelist, prompt/flag conflicts caught before you spend a generation.\n- **Quality verification on every output** — duration, loudness, file size, audible completeness, lyrics alignment.\n- **Provider-agnostic** — local ACE-Step, MusicGen, Stable Audio, or the MiniMax cloud CLI; same workflow, same verification.\n\n## Why use this skill?\n\nMost music models treat \"sparse\" as \"remove everything\" or \"3 minutes\" as \"however long the model wants\". Music Craft encodes the rules once, so every generation runs through the same loop with the same anti-sparse guard, the same prompt validation, the same post-generation verification. You get coherent, full, audibly correct songs — not first-draft lottery tickets.\n\n## Quick Start\n\n1. Say what you want — e.g. _\"Make a sad love song in Spanish, ~3 minutes.\"_\n2. The skill auto-detects language, genre, mood, duration, and asks only the 1-3 things that aren't recoverable from your message.\n3. It builds a production-sheet prompt and routes to the right backend.\n4. It lints the prompt + lyrics, generates, then verifies duration/loudness/completeness before delivering.\n5. If quality fails, it adjusts the prompt and retries once — never the same payload twice.\n\n## When To Use\n\nUse this skill when the task involves:\n\n- generating a song from a user description (genre, mood, language, theme)\n- producing structured lyrics with section tags\n- turning prose, a poem, or a list of themes into song lyrics\n- making an instrumental track with explicit style and instrument control\n- iterating on a generated song with controlled prompt adjustments\n- verifying that generated music has no sparse, a cappella, or clipped sections\n- **needing a specific song length (e.g. 3:30)** — this skill's ACE-Step backend takes `audio_duration` as a parameter and is the exact-duration route. Verified field run: ACE-Step returned exact requested duration for 18/18 local jobs; MiniMax cloud returned 57-135% of requested duration.\n\n**Routing:**\n\n- Prefer this skill when exact duration matters more than generation speed.\n- Prefer this skill when the user wants a full-length local vocal track.\n- Redirect to `music-craft-minimax` only for MiniMax-native workflows or when the user explicitly wants the MiniMax path.\n\n## When NOT To Use\n\nDo not use this skill when:\n\n- the user only needs lyrics as text with no audio — use a writing skill instead\n- the user wants a fast cloud cover, upload-based style transfer, emotion analysis, or a mashup — use `music-craft-minimax`\n- the user wants a local cover/repaint experiment but cannot accept ACE-Step's hardware, timeout, and queue constraints — use `music-craft-minimax` or a cloud backend instead\n- the user has specific BPM, key, or per-section structure requirements that need separate flags — use `music-craft-minimax`\n- a deterministic, single-shot generation with no iteration is sufficient and the user already has the right prompt\n- the user wants to mutate a specific existing audio file (pitch shift, time stretch, stem split) — that is post-production, not generation\n\n## Decision Tree\n\nUse this skill unless the request explicitly needs a MiniMax-only path:\n\n1. If the user wants a fast cloud cover, advanced audio/emotion analysis, a mashup, or per-flag control for `--avoid`, `--bpm`, `--key`, or `--structure`, switch to `music-craft-minimax`.\n2. If the user wants a local source-audio restyle and accepts slower, experimental ACE-Step behavior, stay here and use the ACE-Step audio-conditioned guide.\n3. If the user wants a standard song, instrumental, jingle, or lyrics-driven track, stay here.\n4. If the request is vague but still about generation, stay here and infer defaults before asking anything.\n5. If the user is asking to edit or mutate an existing audio file, treat it as post-production, not base generation.\n6. **If ACE-Step is detected and no models are downloaded:** run memory safety check → present download options (fast/standard/xl-mixed/skip to cloud) → wait for user consent → NEVER auto-download.\n7. **If ACE-Step is detected with models loaded:** check available RAM → offer appropriate tiers (fast/standard/xl-mixed based on RAM) → default to `standard` unless user requests otherwise.\n8. **If user says \"best quality\" on 24GB machine:** offer `xl-mixed` with the caveat that the 50-step sft model output quality is currently poor on 24 GB M3 (high-frequency noise, unclear vocals). Recommend `standard` tier for known-good output unless the user wants to experiment with the fix list in the next section.\n9. **Before submitting any ACE-Step request:** fill in the 6 metas (BPM, key, time signature, vocal language, duration, genre) explicitly in the request body, even if `thinking=true` is set. The LM will use these as anchors. If the user hasn't provided them, infer sensible defaults (e.g. 96 BPM for dream pop, \"D major\" if the prompt mentions a key) before submitting. For `xl-sft` (xl-mixed tier), detailed metas are essential; for the standard `v15-turbo` they're optional but improve consistency.\n10. **For exact-duration vocal tracks on Apple Silicon:** prefer the local ACE-Step route. It is slower than cloud but verified exact to the millisecond in the 2026-06-12 field run.\n\n## Core Philosophy\n\nThis skill is **provider-agnostic** by design. It works with whatever music backend is available: a native `music_generate` tool exposed by the runtime, or a CLI like `mmx` invoked via bash. It does not assume any specific provider, model, or API.\n\nThree rules drive every generation:\n\n1. **Production-sheet prompts.** Every prompt reads like a mini production brief, not a vague description.\n2. **Anti-sparse guards.** Every prompt includes explicit instruments, the \"always playing\" rule, and an avoid list.\n3. **Structure-tagged lyrics.** Every lyric body uses `[Verse]`, `[Chorus]`, `[Break]`, and similar tags to give the generator a clear shape.\n\n## Runtime Adapters\n\nThis skill is agent-neutral. It uses whatever music backend is available — a native tool or a CLI — in the active runtime.\n\nIt does not require:\n\n- any specific music provider\n- any CLI (`mmx` or other)\n- any external API key beyond what the runtime already needs\n- any audio analysis library (librosa, parselmouth, ffmpeg)\n\nIf a more capable backend is installed, the `music-craft-minimax` skill unlocks fast cloud cover workflow, separate parameter flags, and emotion-driven mashups. This skill is the entry point; that one is the power-user upgrade.\n\n## Free Tool Augmentation\n\nThe OpenClaw runtime exposes several free tools that enrich the music generation workflow. None of these require user-side installation — they are part of the runtime, and the skill can call them directly to gather more context about the user's request before building the prompt.\n\n| Tool | Purpose | When to use |\n| --- | --- | --- |\n| `web_fetch` | Fetch readable content from any URL | Lyrics pages, Wikipedia, artist bios, music blogs |\n| `web_search` | Search the web with a query | Find lyrics when only the title is known, find artist info, find genre descriptions |\n| `memory_search` / `memory_get` | Recall from the user's durable memory | Previous music preferences, prior generation issues, typical genres |\n| `browser` | Drive a real browser | JS-heavy lyrics sites (genius.com dynamic loading) — fallback when `web_fetch` returns only chrome |\n\n### Quick decision: which tool to reach for\n\n- **The user gave a URL** → `web_fetch`\n- **The user gave just a name or vague reference** → `web_search`, then `web_fetch` the top result\n- **The user has prior music preferences in memory** → `memory_search` first\n- **`web_fetch` returns only chrome (no content)** → `browser` as fallback\n\nDo not surface copyrighted lyrics verbatim in the final song unless the user provided them. Use fetched lyrics as inspiration for style and structure, not as the song's body.\n\nWorked examples, privacy rules, and scope: [`references/free-tool-inputs.md`](references/free-tool-inputs.md).\n\n## Pre-Flight Check\n\nBefore starting the workflow loop, verify the runtime can do the work. This\nskill has **zero external dependencies** — the only requirement is a music\ngeneration backend (native tool or CLI). Before the first generation in a\nsession, run the full pre-flight protocol (including the Required check) in\n[`references/setup-and-preflight.md`](references/setup-and-preflight.md).\n\nNon-negotiable rules that always apply:\n\n- Never install anything without explicit user consent (Dependency Consent\n  Protocol). Show the exact command and its rough size/impact before asking.\n- Detect the platform first (POSIX vs PowerShell vs cmd) and use matching\n  command syntax for everything that follows.\n- Ask hardware/setup questions once per session, then remember the answers.\n- When a required dependency is missing, ask the user — do not silently\n  degrade or skip.\n\n### When to redirect to `music-craft-minimax`\n\nIf the user's request implies any of:\n\n- fast cloud cover or style transfer from a reference audio file\n- emotion analysis on input audio\n- two-song mashup\n- separate `--avoid`, `--bpm`, `--key`, or `--structure` flags\n\nStop the pre-flight and tell the user: \"That needs `<feature>`, which is in `music-craft-minimax`. Switch to that skill and I will run the same pre-flight with the extended check list.\" Do not try to fake these features with the tools this skill has.\n\nIf the user specifically wants a local source-audio restyle and can accept a slow experimental path, stay in this skill and load [`references/acestep-generation.md`](references/acestep-generation.md). ACE-Step cover/repaint is local, melody-aware, and queue-bound; it is not the fast/default cloud cover path.\n\n**Audio source:** the user must provide a local audio file path. URLs are not accepted in either music skill in v1.5.0+; if the user wants to fetch audio by title from the internet, point them at the private `music-source-fetch` skill (not published on ClawHub).\n\n## Local Models\n\nThree local music models are supported by this skill. **License must be named and accepted before any model is downloaded** — the skill never auto-downloads (inherited hard rule from `setup-and-preflight.md`). License decides whether the output is fit for monetized channels, paid work, or advertising; pick the model that matches the distribution intent before installing.\n\n| Model | License | Size | Commercial? | Purpose | Install guide |\n| --- | --- | ---: | --- | --- | --- |\n| **ACE-Step** (`acestep-v15-xl-base-diffusers`) | MIT | ~11 GB | ✅ Yes | Vocals + lyrics, local, exact-duration | [`§ 1`][acestep-install] |\n| **MusicGen** (`facebook/musicgen-large`) | CC-BY-NC 4.0 | ~3.3 GB | ❌ No | Instrumental only, non-commercial | [`§ 2`][musicgen-install] |\n| **Stable Audio Open** (`stabilityai/stable-audio-open-1.0`) | Stability AI Community | ~1.2 GB | ⚠️ Only if revenue < $1M/yr | Instrumental only, commercial-OK under cap | [`§ 3`][stable-audio-install] |\n\nHardware floor: Apple Silicon MPS gives 5–10× speed-up (CPU works but ~15 min / 30 s). NVIDIA CUDA is the reference. Apple Silicon requires `ACESTEP_LM_BACKEND=mlx`; on macOS, `PYTORCH_MPS_HIGH_WATERMARK_RATIO=0.0` is mandatory for XL DiTs. Memory probe protocol and per-model ML-budget rules: [`references/setup-and-preflight.md`](references/setup-and-preflight.md).\n\nFull installation protocol (download paths, environment setup, verify commands, license click-through notes): [Local Models Install Guide](/Users/luis/Repos/skills/publish/LOCAL-MODELS-INSTALL-GUIDE.md).\n\n[acestep-install]: /Users/luis/Repos/skills/publish/LOCAL-MODELS-INSTALL-GUIDE.md#1-ace-step-mit--11-gb\n[musicgen-install]: /Users/luis/Repos/skills/publish/LOCAL-MODELS-INSTALL-GUIDE.md#2-musicgen-cc-by-nc-40--33-gb\n[stable-audio-install]: /Users/luis/Repos/skills/publish/LOCAL-MODELS-INSTALL-GUIDE.md#3-stable-audio-open-stability-ai-community-license--12-gb\n\n## Backend Generation\n\nSelect the backend from the need, then load only that backend's reference:\n\n| Need | Backend | Reference |\n| --- | --- | --- |\n| Vocals + lyrics, local, best local quality | ACE-Step 1.5 | [`references/acestep-generation.md`](references/acestep-generation.md) |\n| Instrumental, local, no API key, non-commercial | MusicGen | [`references/other-backends.md`](references/other-backends.md) |\n| Simple cloud generation (API key, no local model) | mmx CLI | [`references/other-backends.md`](references/other-backends.md) |\n| Local source-audio cover/repaint, experimental | ACE-Step 1.5 | [`references/acestep-generation.md`](references/acestep-generation.md) |\n| Fast cloud cover, style transfer, mashup, fine flag control | `music-craft-minimax` skill | switch skills — see **When to redirect to music-craft-minimax** above |\n| Instrumental via REST API | Stable Audio | [`references/other-backends.md`](references/other-backends.md) |\n| Anything else the runtime exposes | Generic CLI | [`references/other-backends.md`](references/other-backends.md) |\n\nRules that always apply regardless of backend:\n\n- Validate the prompt against the backend's format before generating\n  (MusicGen wants 1–2 natural-language sentences; ACE-Step wants a detailed\n  multi-dimensional caption; see the backend reference).\n- Never retry an identical failing payload; change prompt, parameters, or\n  backend between attempts.\n- Verify the output file (duration, loudness, completeness) before delivery.\n\n## ACE-Step Operational Caveats\n\nLoad [`references/acestep-generation.md`](references/acestep-generation.md)\nand the focused local references before submitting a local ACE-Step job:\n\n- The ACE-Step API server processes one job at a time; queue multiple versions\n  sequentially and collect each output before submitting the next.\n- `GET /v1/stats` exposes only top-level queue state. For long jobs, tail the\n  server log (`/tmp/acestep-api.log`) for DiT/VAE progress.\n- The cache directory accumulates generated files. Clean old files deliberately;\n  never assume the API cleans them for you.\n- There is no cancel endpoint. Lint prompt, lyrics, duration, and metas before\n  submitting a long job.\n- `thinking: true` is the quality default and may produce two cache files for a\n  single request; collect and label both when they exist.\n- For source-audio cover/repaint, use multipart upload and the `wait_for_acestep.py`\n  helper when available. The local API can return empty `/query_result` data\n  while work is still running, so cache-file detection is part of the workflow.\n- For audio-conditioned tasks beyond cover/repaint (`extract`, `lego`, `complete`),\n  the BASE-only DiT is required — load [`references/acestep-task-types.md`](references/acestep-task-types.md)\n  before submitting. For XL (4B) DiT checkpoints (`xl-base`, `xl-sft`, `xl-turbo`,\n  `xl-mixed`), the `num_inference_steps=50` (Diffusers path) / `inference_steps=50` (REST path) and `shift=3.0`\n  footguns apply — load [`references/acestep-xl-models.md`](references/acestep-xl-models.md) before\n  touching XL.\n\nCopy-pasteable local commands and collection workflow:\n[`references/local-ace-step-curl-template.md`](references/local-ace-step-curl-template.md)\nand [`references/wait-and-collect.md`](references/wait-and-collect.md).\n\n## Operating Rules\n\n### 1. Read and auto-detect\n\nBefore asking anything, infer language, genre, mood, duration, and theme from the user's message. Default duration is about 3 minutes; only ask if the user is explicit about length.\n\nFull auto-detect cheat sheet and edge cases: [`references/user-preference-flow.md`](references/user-preference-flow.md) and [`references/input-workflows.md`](references/input-workflows.md).\n\n### First response defaults\n\nUse these deterministic first responses before asking follow-up questions:\n\n- **Standard song request** -> infer language, genre, mood, and duration; ask only for the missing lyric source, voice, or reference if it is not already implied.\n- **User-provided lyrics** -> keep the lyrics intact, add section tags, and ask only for any missing voice or length detail.\n- **Instrumental or jingle** -> set instrumental mode immediately; ask for duration only if the length is still unclear.\n- **Vague style reference** -> use the reference as a style cue, infer the closest genre family, and ask only for lyrics source or voice if those are not recoverable from context.\n- **Image or URL input enrichment** -> fetch or analyze the input first, turn the result into style cues, then ask only for anything that still cannot be inferred.\n\n### 2. Analyze source material when available\n\nIf the user provides source audio or an analysis file, extract the reusable facts before writing the prompt. If there is no source material, continue with the request text and inferred defaults.\n\nFull analysis options, tool choices, and the decision tree: [`references/input-workflows.md`](references/input-workflows.md).\n\n#### Vocal confirmation gate\n\nIf source analysis returns `language=unknown`, suggests instrumental, or does not clearly confirm vocals, ask one targeted question before prompt construction:\n\n> Is this instrumental, or does it have vocals? If vocals, what language, and should I use provided lyrics or extract them?\n\n#### Target-length confirmation gate\n\nIf source audio exists and the user did not explicitly set the output length, confirm one of: same as source, standard 3:00, standard 3:30, or a specific length.\n\n### 3. Ask only the ambiguous parts\n\nAfter auto-detect, ask 1–3 questions max. Do not ask about language, genre, mood, or duration if the request already makes them obvious.\n\nQuestion patterns and worked examples: [`references/user-preference-flow.md`](references/user-preference-flow.md).\n\n### 4. Translate to a production-sheet prompt\n\nThe prompt you pass to `music_generate` is not a restatement of the user's words. It is a structured brief with ten required slots: genre/subgenre, mood, voice, instruments, anti-sparse instruction, BPM/key, structure, dynamics, production quality, and avoid list.\n\nFull formula, slot-by-slot guide, and worked examples: [`references/prompt-formula.md`](references/prompt-formula.md).\n\n### 5. Validate the prompt\n\nBefore generating, validate prompt length, structure, duration, lyrics tags, and backend-specific conflicts. Run `scripts/lint_lyrics.py <lyrics.txt> --bpm <bpm> --target-seconds <seconds>` for any provided/generated lyrics before spending a generation. If `music-craft-minimax` is installed, its `scripts/lint_music_request.py` is the canonical guard for mmx prompt size, missing fields, and conflicts.\n\nPer-backend byte limits, length-reduction techniques, and lint rules: [`references/prompt-formula.md`](references/prompt-formula.md).\n\n### 6. Structure the lyrics\n\nIf the user provides lyrics, add canonical section tags (`[Verse]`, `[Chorus]`, and so on) without altering the words. If the skill writes the lyrics, structure them from the start. Do not invent descriptive bracket tags such as `[Guitar Solo - distorted]` or `[Lyrics]`; bracket text can be sung by the model. ASR-extracted lyrics are unverified — cross-check and clean them before building the prompt.\n\nFull tag reference, Whisper verification rules, transcript cleanup, and emotion-specific lyrics patterns: [`references/structure-tags.md`](references/structure-tags.md) and [`references/lyrics-cleanup.md`](references/lyrics-cleanup.md).\n\n### 7. Generate and verify raw output\n\nCall the detected backend with the production-sheet prompt and structured lyrics. Use the backend-specific generation command from the backend's reference file (routed via the **Backend Generation** table). Adapt the prompt format to the backend (e.g., MusicGen needs prompt + lyrics combined into one text block; mmx accepts them separately). After the tool returns, verify that audio is non-empty, has no sparse or a cappella drops, lyrics alignment is plausible, and structure matches the plan. When lyrics matter, listen or transcribe before delivery; `scripts/verify_lyrics_alignment.py` can compare expected lyrics against a transcript.\n\nBackend commands and output verification details: backend reference files via the **Backend Generation** table above; quality checks: [`references/quality-and-revision.md`](references/quality-and-revision.md).\n\n### 8. Finalize delivery copy\n\nNormalize loudness with `ffmpeg loudnorm` (target -16 LUFS, -1 dBTP true peak), then verify duration, loudness, file size, and absence of silence drops or artifacts.\n\nLoudnorm command, verify checklist, and request-fit checks: [`references/quality-and-revision.md`](references/quality-and-revision.md).\n\n### 9. Iterate, do not retry the same payload\n\nIdentify the failure mode, adjust the prompt or lyrics to target it, and try once with a different seed if available. After 2 failed retries, ask the user to clarify or accept the best attempt. Never retry the same prompt plus lyrics combination twice in a row.\n\nIteration loop, adjustment recipes, and retry patterns: [`references/error-handling.md`](references/error-handling.md).\n\n## Request Intake\n\nCollect the required fields before generating: language, genre/subgenre, mood,\ntheme, vocal mode, lyric source, duration, structure, references, output location.\nAsk the output location once, then reuse it for the whole session.\nBuild a confidence map for what was auto-detected vs assumed, and confirm\nonly the low-confidence slots with the user.\n\nFull checklists, confidence-map examples, language-consistency checks,\nambiguous-phrase routing, and the per-song output layout and slug rules:\n[`references/request-intake.md`](references/request-intake.md).\n\n## Anti-Sparse Rules (Critical)\n\nThe single most common failure mode of music generators: interpreting \"sparse\", \"quiet\", or \"minimal\" as \"remove all instruments and vocals\".\n\n### Always include in the prompt\n\n1. **List every instrument by name.** Example: `accordion, upright bass, orchestral strings, piano, light percussion`.\n2. **The always-playing rule.** `ALL instruments ALWAYS playing throughout, NEVER a cappella or silent`.\n3. **The avoid list.** `AVOID sparse minimal arrangements, AVOID a cappella sections`.\n4. **Explicit treatment of quiet sections.** `quiet sections: reduced to accordion and bass only, still fully played`.\n\n### Never use alone\n\n- `sparse arrangement`\n- `minimal instrumentation`\n- `stripped back`\n- `a cappella section`\n- `quiet with no instruments`\n\nIf the user asks for any of these, translate them into the explicit-instrument form.\n\n### Ground every mood word\n\nEvery mood, energy, or emotion word in the prompt must be tied to at least one concrete production detail. A mood word with no grounding will be ignored — the model defaults to a \"neutral pleasant\" register.\n\n| Mood word | Required grounding (pick at least one) |\n| --- | --- |\n| `sad` | minor key, slow BPM, breathy vocal, sparse chord pattern, low strings |\n| `energetic` | fast BPM, driving drums, sharp synth hits, strong rhythm guitar |\n| `romantic` | warm strings, soft vocal register, sustained pads, slow harmonic rhythm |\n| `dark` | minor key, low register, distorted bass, low-pass mix, breathy vocal |\n| `dreamy` | reverb-heavy mix, soft attack, layered pads, sustained vocal |\n| `aggressive` | distorted guitars, fast BPM, shouted vocal, heavy drums |\n| `triumphant` | major key, building dynamic, brass hits, declarative vocal |\n| `intimate` | close-mic vocal, low dynamic range, soft attack, single voice |\n\nIf a mood word cannot be grounded, drop it. A grounded prompt with five moods beats an ungrounded prompt with fifteen. For the full emotion quick reference (21 emotions with prompt + lyrics + arrangement templates), see [`references/prompt-formula.md`](references/prompt-formula.md).\n\n## Rate Limits\n\nRespect backend rate limits; on a limit error, wait at least 60 seconds and reduce request rate rather than hammering. Details and per-backend behavior: [`references/quality-and-revision.md`](references/quality-and-revision.md).\n\n## Quality Verification Checklist\n\nVerification has two levels. **Technical generation success** means the file exists, is non-empty, and has audible content. **User-fit confirmed** means the output actually fits the user's intent (genre, language, length, structure, lyrics alignment). Always check both.\n\nBefore delivering a generated song to the user, walk this list mentally. If 3 or more items fail, the prompt needs adjustment and a regeneration. If 1–2 fail, you can either accept the result and warn the user, or make a targeted fix and regenerate.\n\n1. **Audio is non-empty and plays.** Sample the first 5 seconds and the midpoint. If the file is empty or silent, regenerate.\n2. **No sparse or a cappella drops.** Check the midpoint specifically — sparse drops are most common in quiet sections.\n3. **No clipped vocals or distortion.** Listen for sudden loudness spikes or harshness.\n4. **Lyrics alignment is plausible.** If the user provided lyrics, the output should hit the key phrases recognizably.\n5. **Structure matches the plan.** If you asked for intro-verse-chorus-verse-chorus-bridge-chorus-outro, the song should have 7–8 distinct sections.\n6. **Genre and mood are recognisable.** A \"French chanson ballad\" should sound like French chanson, not generic acoustic.\n7. **Language is correct.** If the user asked for Spanish, the vocals should be in Spanish, not accented English.\n8. **Energy arc is coherent.** The song should build, peak, and resolve. If it stays at the same energy for 3 minutes, the prompt was likely too vague.\n\nFor the request-fit checklist and revision patterns, see [`references/quality-and-revision.md`](references/quality-and-revision.md).\n\n## Revision Prompts\n\nWhen the output is close but not right, do not regenerate from scratch. Keep 80% of the original prompt. Add a single `REVISION:` block at the end that targets the specific failure.\n\nWorked examples and the full retry recipe library: [`references/quality-and-revision.md`](references/quality-and-revision.md).\n\n## Lyrics Optimizer Behavior\n\nWhen `music_generate` is called **without explicit lyrics** and the request implies a vocal track (not instrumental), the runtime may auto-generate lyrics from the prompt.\n\nPer-provider behavior, the web lyrics lookup option, and handling user surprise at AI-written lyrics: [`references/quality-and-revision.md`](references/quality-and-revision.md).\n\n## User Preference Flow\n\nThe skill does not start with a questionnaire. It starts by reading and inferring.\n\n| User says... | Skill does... |\n| --- | --- |\n| \"Make a sad love song in Spanish\" | Auto-detect: ES, romantic, ~3 min. Ask: lyrics source and vocal register. |\n| \"Instrumental lofi for studying\" | Auto-detect: lofi, no vocals, ~3 min. Ask: nothing. Generate. |\n| \"Here are the lyrics, make it pop\" | Auto-detect: pop, user-lyrics. Ask: tempo and energy preference. |\n| \"Something that sounds like Rosalía\" | Auto-detect: modern Latin pop, female vocal. Ask: lyrics source and theme. |\n| \"I don't know, surprise me\" | Pick a coherent default (for example upbeat indie pop, EN, ~3 min, auto-lyrics) and confirm with the user before generating. |\n\nFor the full decision table and edge cases, see [`references/user-preference-flow.md`](references/user-preference-flow.md).\n\n## Output File Layout\n\nOne subfolder per song under the user's chosen output root; analysis JSON,\nprompt file, and versioned audio files (`A_`, `B_`, `C_`, `M1_`/`M2_`,\n`N1_`/`N2_`, `v2_`/`v3_` prefixes) live together in that subfolder. Slug and version rules: [`references/request-intake.md`](references/request-intake.md).\n\nIf the runtime needs a `MEDIA:` delivery path, use a path without spaces or\ncopy the final file into a workspace media folder first. Keep the archival copy\nin the per-song output folder.\n\n## Reference Map\n\n- [`references/setup-and-preflight.md`](references/setup-and-preflight.md) — pre-flight protocol: dependency consent, platform detection, user and hardware setup, required/optional dependencies, install details\n- [`references/windows-wsl-setup.md`](references/windows-wsl-setup.md) — Windows/WSL setup: certificate/proxy handling, WSL distro setup for local generation\n- [`references/acestep-generation.md`](references/acestep-generation.md) — complete ACE-Step 1.5 guide: API workflow, full-song generation, quality tiers and memory-safe selection, audio-conditioned generation (cover, repaint, reference audio)\n- [`references/local-ace-step-curl-template.md`](references/local-ace-step-curl-template.md) — JSON-safe `/release_task` curl template for direct ACE-Step submission\n- [`references/wait-and-collect.md`](references/wait-and-collect.md) — M1 -> wait -> collect -> M2 local sequencing, log watching, cache collection, and duration verification\n- [`references/lyrics-cleanup.md`](references/lyrics-cleanup.md) — cleanup recipe for Whisper/ASR transcripts, canonical section tags, `info.txt`, and `lyrics_whisper_medium.json`\n- [`references/other-backends.md`](references/other-backends.md) — MusicGen, mmx CLI, Stable Audio, and generic CLI backend guides\n- [`references/request-intake.md`](references/request-intake.md) — full intake protocol and per-song output layout (slugs, version prefixes)\n- [`references/prompt-formula.md`](references/prompt-formula.md) — full production-sheet formula, worked examples across genres, prompt lint, and the emotion quick reference\n- [`references/structure-tags.md`](references/structure-tags.md) — all section tags with rules, effects, and timing hints\n- [`references/user-preference-flow.md`](references/user-preference-flow.md) — the auto-detect plus ask decision table and edge cases\n- [`references/examples.md`](references/examples.md) — five worked examples (Spanish pop, English instrumental jingle, user-provided lyrics, image-inspired track, text-only style reference) with intake → prompt → verification for each\n- [`references/style-categories.md`](references/style-categories.md) — 10 style categories with default instruments, BPM range, and mood\n- [`references/input-workflows.md`](references/input-workflows.md) — 10 input types (description, user-lyrics, audio file, YouTube audio, song name, lyrics URL, YouTube metadata, JioSaavn metadata, image, genre/cultural), plus the signal-extraction rubric and confidence levels\n- [`references/quality-and-revision.md`](references/quality-and-revision.md) — rate limits, request-fit checklist, revision prompts, delivery copy, lyrics-optimizer behavior\n- [`references/error-handling.md`](references/error-handling.md) — error table, retry recipes (wrong language, weak chorus, sparse, vocals in instrumental, missing genre, too generic), and recovery patterns\n- [`references/free-tool-inputs.md`](references/free-tool-inputs.md) — web_fetch, web_search, image, and memory tools for enriching inputs without scripts\n- [`music-craft-minimax/scripts/lint_music_request.py`](../music-craft-minimax/scripts/lint_music_request.py) — optional standard-library helper for routing, blockers, missing fields, prompt, and `mmx` flag linting. Run it before generating to catch missing required slots, conflicting language signals, and vague mood words without grounding.\n- [`scripts/lint_lyrics.py`](scripts/lint_lyrics.py) — standard-library lyrics preflight for section-tag whitelist checks and syllable/BPM duration estimates.\n- [`scripts/verify_lyrics_alignment.py`](scripts/verify_lyrics_alignment.py) — standard-library post-generation transcript-vs-lyrics overlap check for semantic delivery.\n- [`scripts/wait_for_acestep.py`](scripts/wait_for_acestep.py) — standard-library helper for ACE-Step task polling plus cache-file completion detection.\n- [`scripts/extract_stems.py`](scripts/extract_stems.py) — optional Demucs wrapper that writes normalized stem paths and `stems.json` for local arranger experiments.\n- [`scripts/remix_stems.py`](scripts/remix_stems.py) — preview-quality `ffmpeg amix` helper for recombining validated stems.\n- [`scripts/smoke_test.py`](scripts/smoke_test.py) — pure-Python smoke tests for local helper behavior.\n- For emotion-driven generation (vocal speed, intensity, pitch bends, emotion recipes, iteration loop), see the quick reference in [`references/prompt-formula.md`](references/prompt-formula.md) under \"Mood\" and the full shared emotion recipes in [`music-craft-minimax/references/emotion-delivery.md`](../music-craft-minimax/references/emotion-delivery.md)\n- [`references/acestep-task-types.md`](references/acestep-task-types.md) — per-task reference for the five ACE-Step audio-conditioned tasks (`cover`, `repaint`, `extract`, `lego`, `complete`): inputs, outputs, parameters, BASE-only footguns, and a routing tree\n- [`references/acestep-xl-models.md`](references/acestep-xl-models.md) — XL (4B) DiT family: `xl-base` vs `xl-mixed` deployment shapes, hardware floor (≥12 GB VRAM, 24 GB M3 reference, 32 GB+ for `xl-mixed`), the two silent-regression footguns (`num_inference_steps=50` MUST be explicit, `shift=3.0` default), and an XL-vs-standard decision tree\n- [`references/mps-bf16-acceleration.md`](references/mps-bf16-acceleration.md) — honest Apple Silicon (MPS) reference: which env vars actually work (`PYTORCH_MPS_HIGH_WATERMARK_RATIO=0.0`, `ACESTEP_LM_BACKEND=mlx`), and the verifiable fact that ACE-Step currently forces fp32 on MPS (no bf16 escape valve — `ACESTEP_USE_BF16` is NOT consumed)\n- [`references/lrc-generation.md`](references/lrc-generation.md) — backend-neutral LRC (synced lyrics) workflow: WhisperX/Whisper transcription → lyrics alignment → LRC format → validator\n- [`references/audition-rubric.md`](references/audition-rubric.md) — 5-dimension quality scoring (Musicality, Production, Prompt Adherence, Vocal, Technical) for accept/revise/regenerate/reject decisions\n- [`references/acestep-shift-schedule.md`](references/acestep-shift-schedule.md) — full `shift` parameter taxonomy (shift=1.0 turbo, shift=3.0 base/sft, continuous shift) with tier-to-shift mapping table and decision tree\n- [`references/audio-format-selection.md`](references/audio-format-selection.md) — when to use wav vs flac vs mp3 vs ogg\n- [`references/advanced-diffusion-controls.md`](references/advanced-diffusion-controls.md) — `guidance_scale`, `infer_method`, `scheduler`, and `eta` parameter reference\n- [Local Models Install Guide](/Users/luis/Repos/skills/publish/LOCAL-MODELS-INSTALL-GUIDE.md) — cross-cutting install protocol shared with `music-craft-minimax`: license-acceptance gate, where to download, exact sizes, install commands, and verify steps for ACE-Step, MusicGen, and Stable Audio Open\n\n## Data, Consent, and Local Side Effects\n\nThis skill may use local or cloud music backends depending on what the user asks for and what is installed. Keep the workflow user-visible:\n\n- **Cloud backends:** prompts, lyrics, reference URLs, and generated/derived music instructions may be sent to the selected provider.\n- **Local backends:** model downloads, local analysis, temporary files, and generated audio may be written on the user's machine.\n- **Reference material:** fetched webpages and lyrics pages are used only to enrich the music prompt unless the user explicitly chooses an audio/cover workflow.\n- **Output files:** ask where to save generated files and avoid overwriting user-visible outputs without explicit confirmation.\n\nBefore uploading user-owned or third-party media to a cloud backend, state what will be sent and why, then wait for confirmation if the user has not already clearly requested that cloud workflow.\n\n## Licensing and commercial-use gate\n\nThe published skill bundle is MIT-0 under ClawHub's publishing rules: it may\nbe used, modified, and redistributed commercially without attribution. That\nlicense applies to these skill instructions and bundled helper code; it does\nnot grant a license to any third-party model, software package, input audio,\nlyrics, voice, or generated output. Every operator must install the selected\nbackend under its own applicable terms and keep the required notices. Terms\nand model cards can change, so check them before commercial release.\n\n- **ACE-Step 1.5 local:** the project and published model card currently state\n  an MIT license and commercial-ready use. Verify the exact model checkpoint\n  and its current license before shipping a commercial track.\n- **MusicGen local:** the code is MIT, but the MusicGen model weights are\n  CC-BY-NC 4.0. Do not use MusicGen-generated audio for commercial work,\n  monetized channels, client deliverables, paid products, or advertising.\n- **Provider-agnostic/runtime backends:** do not assume commercial rights.\n  Identify the actual provider, model, account/product tier, and output terms\n  before generating for commercial use. If these are unknown, stop and ask.\n- **Inputs:** the user must own or have permission for uploaded/reference\n  audio, lyrics, samples, and voices. A model license does not legalize a\n  copyrighted cover, lyric transcription, or unauthorized voice.\n\nGenerated audio may also lack copyright protection or exclusivity under the\nlaw of the user's jurisdiction. Commercial permission from a provider is not\nthe same as a guarantee of exclusive copyright or zero similarity risk.\n\nTreat music generation as a small, controlled iteration loop, not a single \"press button, get song\" call.\n\nThe required generation loop is:\n\n1. Clarify goal and source material.\n2. Analyze source audio if available, or accept user-provided analysis.\n3. Build a production-sheet prompt with genre, mood, BPM, key, instruments, structure, vocals/lyrics, and constraints.\n4. Select backend based on need: exact-duration vocals/lyrics -> ACE-Step; local melody-aware cover/repaint experiments -> ACE-Step if hardware and time budget allow; instrumental/local -> Stable Audio 3 or MusicGen; fast cloud cover, mashup, or MiniMax-specific flags -> `music-craft-minimax` / mmx.\n5. Validate prompt length, structure, backend-specific conflicts, and expected duration.\n6. Generate with the selected backend.\n7. Verify duration, loudness/peak, file size, audible completeness, lyrics alignment, and structure.\n8. Deliver files with a short analysis summary and caveats. If quality fails, adjust the prompt and retry; do not retry the same payload twice.\n\nFor deep prompt engineering, lyrics structure, and the full user-preference decision table, see the linked references at the end.\n\nFile v1.6.0:README.md\n\n# Music Craft\n\nGenerate songs, instrumentals, and lyrics-driven tracks through a disciplined OpenClaw-native workflow.\n\nCurrent release: v1.6.0.\n\nThis skill is **provider-agnostic**. It works with any music backend the OpenClaw runtime exposes via the `music_generate` tool — no special CLI, API, or library required.\n\n## Data and consent\n\nDepending on the chosen backend, prompts, lyrics, reference URLs, or generated/derived music instructions may be sent to a cloud provider. Local backends may download models and write temporary/generated audio files on the user's machine. Ask before installing/downloading large dependencies, uploading user-owned media, or overwriting existing outputs.\n\n## Licensing and commercial use\n\nClawHub publishes this skill bundle under MIT-0, so the skill instructions\nand bundled helper code may be used, modified, and redistributed commercially\nwithout attribution. MIT-0 does not grant rights to third-party models,\nsoftware, source material, or generated music. The operator must accept the\nselected backend's current terms on their own computer and use their own\nprovider account. ACE-Step 1.5\nis currently documented by its project as MIT/commercial-ready. MusicGen's\ncode is MIT, but its model weights are CC-BY-NC 4.0 and must not be used for\ncommercial output. For any other backend, identify and verify its model,\naccount tier, and output terms before commercial use. Source audio, lyrics,\nsamples, and voices must also be owned or properly licensed.\n\n## Platform support\n\nThis base workflow is effectively OS-neutral: it should work on macOS, Linux, and Windows as long as the active OpenClaw runtime exposes the `music_generate` tool. Platform differences only matter if the runtime provider itself needs local setup.\n\n## What it does\n\n- Translates your request into a production-sheet prompt with anti-sparse guards\n- Structures your lyrics (or auto-generated lyrics) with whitelisted section tags\n- Calls the runtime's `music_generate` tool\n- Verifies duration, loudness, structure, lyrics alignment, and audible quality before delivery\n- Documents prompt-length, lyrics-transcription, direct ACE-Step submission, wait-and-collect, and post-generation finalization safeguards\n\n## When to use\n\nUse this skill for any music generation task that does not require:\n\n- cover or style transfer from a reference audio file\n- emotion analysis or two-song mashup\n- separate `--avoid`, `--bpm`, `--key`, or `--structure` flags\n\nFor those, see [`music-craft-minimax`](../music-craft-minimax/) (requires MiniMax Token Plan). Audio input must be a local file path. URLs are not accepted in v1.5.0+; if you want to fetch audio by title from the internet, use the private `music-source-fetch` skill first (not published on ClawHub).\n\nIf the request is a standard song, instrumental, jingle, or lyrics-driven track, stay here and infer defaults first.\n\nFor exact-duration vocal tracks, prefer the local ACE-Step path documented in\n[`references/acestep-generation.md`](references/acestep-generation.md). The\n2026-06-12 field run verified 18/18 local ACE-Step jobs at exact requested\nduration; MiniMax cloud is faster but approximate.\n\n## Quickstart\n\nThis skill should load automatically when the task is clearly music generation. Typical flows:\n\n> \"Make an upbeat summer pop song in English, ~3 minutes, with original lyrics about road trips.\"\n\n> \"Here's a poem I wrote. Turn it into a rock ballad with male vocals in Spanish.\"\n\n> \"I need a 30-second lofi jingle for a YouTube intro, no vocals.\"\n\nThe skill reads, auto-detects what it can, asks at most 1–3 targeted questions for the ambiguous parts, then generates.\n\n## Workflow loop\n\n1. Clarify the goal and source material\n2. Analyze source audio if available, or accept user-provided analysis\n3. Build a production-sheet prompt\n4. Select the backend based on vocals, lyrics, length, local/cloud preference, speed, and hardware\n5. Validate prompt length, structure, conflicts, and expected duration\n6. Generate\n7. Verify duration, loudness/peak, file size, audible quality, lyrics alignment, and structure\n8. Deliver with a short analysis summary, or iterate with a targeted prompt adjustment\n\n## Reference\n\nFor full details, see [`SKILL.md`](SKILL.md).\n\nFor practical routing examples, see [`references/examples.md`](references/examples.md).\n\nFile v1.6.0:_meta.json\n\n{\n  \"ownerId\": \"kn7fkh362pnj2zckq8pcxfaxr9821zv8\",\n  \"slug\": \"music-craft\",\n  \"version\": \"1.6.0\",\n  \"publishedAt\": 1785411394536\n}\n\nFile v1.6.0:references/acestep-generation.md\n\n# ACE-Step Generation\n\nComplete ACE-Step 1.5 operating guide: API workflow (submit, poll, copy),\nfull-song generation, quality tiers with memory-safe selection, and\naudio-conditioned generation (cover, repaint, reference audio). Load this\nwhen the selected backend is ACE-Step.\n\n## ACE-Step 1.5 (local — free, vocals + lyrics, best local quality)\n\n**Best for:** local generation with real vocals, separate lyrics, song structure, up to 10 minutes (600s). No API key, no quota. Runs natively on Apple Silicon via MLX.\n\n**Verified routing note (2026-06-12):** ACE-Step is the exact-duration route.\nIn a 9-song field run, local M1+M2 generations returned the requested\n`audio_duration` exactly for 18/18 jobs. MiniMax cloud was faster but returned\n57-135% of requested duration for the paired cloud jobs.\n\n**Prerequisites:** REST API must be running on `http://127.0.0.1:8001`. Install with:\n```bash\ngit clone https://github.com/ace-step/ACE-Step-1.5.git \"${ACE_STEP_PATH}\"\ncd \"${ACE_STEP_PATH}\" && uv sync\nuv run acestep-api --port 8001   # or: ./start_api_server_macos.sh\n```\n\n(See [`setup-and-preflight.md`](setup-and-preflight.md) for how `ACE_STEP_PATH` is determined.)\n\n**Generation (3-step async):**\n\n```bash\n# 1. Submit task\nTASK_ID=$(curl -s -X POST http://127.0.0.1:8001/release_task \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"prompt\": \"<detailed caption, e.g.: dreamy 80s synthwave, warm analog synths, gated-reverb drums, arpeggiated bass, neon night-drive mood>\",\n    \"lyrics\": \"[Verse]\\n<lyrics here>\\n\\n[Chorus]\\n<lyrics here>\",\n    \"audio_duration\": 210,\n    \"bpm\": 96,\n    \"key_scale\": \"D major\",\n    \"time_signature\": \"4/4\",\n    \"vocal_language\": \"en\",\n    \"thinking\": true,\n    \"inference_steps\": 8,\n    \"guidance_scale\": 7.0\n  }' | python3 -c \"import json,sys; print(json.load(sys.stdin).get('data',{}).get('task_id',''))\")\n\n# 2. Poll for completion. Treat empty `data` as pending, not failed.\n# Wait, then check:\ncurl -s -X POST http://127.0.0.1:8001/query_result \\\n  -H \"Content-Type: application/json\" \\\n  -d \"{\\\"task_ids\\\": [\\\"$TASK_ID\\\"]}\"\n\n# 3. Copy audio from cache dir when done\n# Files saved to: ${ACE_STEP_PATH}/.cache/acestep/tmp/api_audio/\n```\n\n**Polling caveat:** The `/query_result` endpoint may return `{\"data\": [], \"code\": 200}` even while the task is actively running. This is a known server-side quirk. Don't treat empty data as \"task failed\" — instead, check for new MP3 files in the cache directory, or look at the server log (`/tmp/acestep-api.log`) for actual progress markers (e.g. `MLX DiT diffusion: 24/50`). If available, use `scripts/wait_for_acestep.py` because it reconciles `/query_result` with cache-file detection.\n\n**Stats caveat:** `GET /v1/stats` exposes top-level state such as queued,\nrunning, succeeded, failed, `queue_size`, and `avg_job_seconds`. It does not\nshow current sub-stage progress. For long jobs, use `/tmp/acestep-api.log` as\nthe source of truth for LM, DiT, CFG, and VAE progress.\n\n**Cache caveat:** generated audio accumulates under\n`${ACE_STEP_PATH}/.cache/acestep/tmp/api_audio/`. Review old files before\ndeleting them:\n\n```bash\nfind \"${ACE_STEP_PATH:-$HOME/ACE-Step-1.5}/.cache/acestep/tmp/api_audio\" \\\n  -type f -name '*.mp3' -mtime +7 -print\n```\n\nReplace `-print` with `-delete` only after confirming the files are no longer\nneeded.\n\n**Cancel caveat:** there is no documented cancel endpoint for a queued or\nrunning local job. Lint prompt, lyrics, duration, and metas before submitting.\n\nFor a JSON-safe direct submission template, see\n[`local-ace-step-curl-template.md`](local-ace-step-curl-template.md). For the\nM1 -> wait -> collect -> M2 pattern, see\n[`wait-and-collect.md`](wait-and-collect.md).\n\n**Prompt format:** Prefer a **detailed, multi-dimensional caption** — ACE-Step's own docs call the caption \"the most important factor affecting generated music\", and the project's example prompts are rich 1–3 sentence descriptions, not bare tags. Cover, in order: **genre, key instruments, vocal character, mood, and production/texture** words. A short 2–6 word tag still works (and the LM expands it when `thinking=true`), but specificity measurably improves results. The earlier \"keep it to short tags\" advice was wrong for ACE-Step 1.5.\n\nTwo rules that matter:\n- **Resolve style conflicts temporally, don't stack them.** The model handles \"starts soft strings, builds to driving synth-rock, ends ambient\" far better than \"classical + hardcore metal\" jammed into one static caption.\n- **Keep the caption consistent with the lyrics' section tags** (a \"solo violin, classical\" caption fighting a `[Guitar Solo - distorted]` tag degrades output).\n\n## Full-song generation workflow (3:30 default)\n\nThe default `audio_duration` is **210s (3:30)** — the typical user expectation for a \"song\". Use 210s as the reference default only when the user has not specified a target length and no source-duration expectation needs to be matched. If source audio exists, confirm the target output length before submission rather than assuming the 210s default.\n\n**Inputs to prepare (set up once, reuse for every song):**\n1. **Lyrics** — write or obtain the full song lyrics with `[Verse 1]`, `[Pre-Chorus]`, `[Chorus]`, `[Verse 2]`, `[Bridge]`, `[Outro]` tags. ~150-200 words fits a 3:30 song comfortably.\n2. **Detailed caption** — see \"Prompt format\" above. ~500-1000 chars covering genre, instruments, vocal character, mood, production, emotional arc, and avoid list.\n3. **The 6 metas** — fill in `bpm`, `key_scale`, `time_signature`, `vocal_language`, `audio_duration: 210`, plus the prompt and lyrics.\n\n**Request body template for a full song (copy/paste and fill in):**\n\n```bash\nTASK_ID=$(curl -s -X POST http://127.0.0.1:8001/release_task \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"prompt\": \"<your detailed multi-dimensional caption here>\",\n    \"lyrics\": \"[Verse 1]\\n<line>\\n<line>\\n\\n[Pre-Chorus]\\n<line>\\n<line>\\n\\n[Chorus]\\n<line>\\n<line>\\n\\n[Verse 2]\\n<line>\\n<line>\\n\\n[Bridge]\\n<line>\\n<line>\\n\\n[Outro]\\n<line>\\n<line>\\n\",\n    \"audio_duration\": 210,\n    \"bpm\": 96,\n    \"key_scale\": \"D major\",\n    \"time_signature\": \"4/4\",\n    \"vocal_language\": \"en\",\n    \"thinking\": true,\n    \"inference_steps\": 8,\n    \"guidance_scale\": 7.0\n  }' | python3 -c \"import json,sys; print(json.load(sys.stdin).get('data',{}).get('task_id',''))\")\n```\n\n**Expected wall-clock on M3 24GB (standard tier, 2B turbo, 8 steps):**\n- LM thinking: ~6s (1.7B model, ~50 tok/s)\n- DiT diffusion: ~4-6 min (scales linearly with audio length)\n- VAE decode: ~40s\n- **Total: ~5-7 min per 3:30 song**\n\nSave each reference output in a per-song folder such as `~/Music mix/<project>/<song-slug>/M1_<style>_ACE_210s.mp3`. A good reference run uses `audio_duration: 210`, the detailed prompt format, and the standard tier. If your output sounds comparable (or better), the workflow is working.\n\n**If you only have a 60-second subset of lyrics** (e.g. a hook for a jingle, or a single chorus to test the prompt), set `audio_duration: 60` — that's perfectly fine, just not a full song. Use the full-lyrics version for the final generation.\n\n```\n# Good (detailed, multi-dimensional — matches ACE-Step's own examples)\n\"A groovy funk track with slap bass, tight horn stabs, rhythmic guitar scratching, a charismatic male lead with call-and-response backing vocals, and an irresistible pocket groove\"\n\"Dreamy 80s synthwave: warm analog synths, gated-reverb drums, arpeggiated bassline, shimmering pads, nostalgic neon night-drive mood\"\n\n# Also fine (short tag; LM expands it with thinking=true)\n\"dreamy synthwave, 80s retro, atmospheric pads\"\n\n# Avoid: contradictory styles stacked in one static caption (express as evolution instead)\n\"classical chamber strings AND crushing hardcore metal AND lo-fi hip-hop, all at once\"\n```\n\n**Parameters:**\n\n| Parameter | Type | Default | Notes |\n|---|---|---|---|\n| `prompt` | string | required | **Detailed caption preferred** (genre + instruments + vocal character + mood + production), per ACE-Step's docs and example prompts. A short tag also works and is expanded by the LM when `thinking=true`. Resolve style conflicts temporally rather than stacking them. |\n| `lyrics` | string | optional | `[Verse]`/`[Chorus]` tagged lyrics |\n| `audio_duration` | int | **210** | **10–600s.** **Default: 210s (3:30)** for full songs — this is the typical user expectation and matches the reference workflow using `~/Music mix/<project>/<song-slug>/M1_<style>_ACE_210s.mp3`. Set to fit ALL lyrics (see Duration Guide below). Use shorter values (30–60s) only for jingles, hooks, or test drafts. |\n| `thinking` | bool | false | LM rewrites tags → richer caption. **Always use `true`** for best results. Field note: one request may save two cache files (primary + variant), so collect and label both when present. |\n| `use_format` | bool | false | When true, the LM also enhances your caption/lyrics (similar to `thinking` but for prompt enrichment). Try `true` if the LM seems to be missing context from your prompt. |\n| `inference_steps` | int | 8 | Diffusion steps. For **`acestep-v15-turbo` (standard)**: 8 is the documented setting, do not exceed 20. For **`acestep-v15-xl-sft` (xl-mixed)**: 32-64 recommended, default 50. Using 8 with xl-sft produces \"soup\" output (all elements at same level, no dynamics). |\n| `guidance_scale` | float | 7.0 | Higher = stricter prompt adherence. **Only effective for base/sft models**, not turbo. For `xl-sft`, try 4.0-7.0 range. |\n| `shift` | float | 3.0 | Timestep shift factor (1.0-5.0). **Officially documented as \"only effective for base models, not turbo models\"** — but xl-sft is an SFT model, not turbo. Experiment with 1.0 or 5.0 if the default sounds off. |\n| `infer_method` | string | \"ode\" | Diffusion inference method. `\"ode\"` (Euler, faster) or `\"sde\"` (stochastic, sometimes more stable for SFT models). |\n| `seed` | int | -1 | -1 = random. Set for reproducibility |\n| `vocal_language` | string | \"en\" | BCP-47 language code for vocals. Important for non-English songs — the model picks the right phoneme set. |\n| `bpm` | int | none | Optional. When `thinking=true` and missing, the LM infers it. Set explicitly for tighter control. |\n| `key_scale` | string | \"\" | Optional. E.g. \"D major\", \"A minor\". Same as `bpm`. |\n| `time_signature` | string | \"\" | Optional. E.g. \"4/4\", \"3/4\". Same as `bpm`. |\n| `cfg_interval_start` | float | 0.0 | CFG application start ratio (0.0-1.0). Default applies CFG throughout the diffusion. |\n| `cfg_interval_end` | float | 1.0 | CFG application end ratio (0.0-1.0). |\n| `use_adg` | bool | false | Adaptive Dual Guidance. **Base model only.** Not applicable to `xl-sft`. |\n\n**Environment variables** (set when starting the server, not in the request body):\n\n| Env var | Default | Notes |\n|---|---|---|\n| `ACESTEP_CONFIG_PATH` | `acestep-v15-turbo` | DiT model path. Set to `acestep-v15-xl-sft` for xl-mixed. |\n| `ACESTEP_LM_MODEL_PATH` | `acestep-5Hz-lm-0.6B` | LM model path. Use `acestep-5Hz-lm-1.7B` for higher quality. |\n| `ACESTEP_LM_BACKEND` | `vllm` | Backend for the LM. **On Apple Silicon (macOS), set to `mlx`** for native acceleration. vLLM is meant for Linux+CUDA. |\n| `ACESTEP_GENERATION_TIMEOUT` | 600 | Per-generation timeout in seconds. **Set to `3600` (1 hour) when using xl-mixed on 24GB M3** — default 600s fires mid-generation. |\n| `ACESTEP_OFFLOAD_TO_CPU` | false | Set to `true` for low-VRAM environments to support longer audio generation. |\n| `PYTORCH_MPS_HIGH_WATERMARK_RATIO` | ~0.4 | **On macOS, set to `0.0`** to allow XL model to load (MPS otherwise enforces a tight memory cap that fails to load the 4B DiT). |\n| `ACESTEP_CONFIG_PATH2`, `ACESTEP_CONFIG_PATH3` | empty | Optional secondary DiT models selectable via the `model` parameter in requests. |\n\n**Duration Guide (audio_duration):**\n\nThe `audio_duration` parameter controls how much audio ACE-Step generates. If it's too short, lyrics get cut off. Estimate based on lyrics word count:\n\n| Lyrics length | Words | Recommended `audio_duration` | Real-world length |\n|---|---|---|---|\n| Short (jingle, hook) | <50 | 30–60 | 0:30–1:00 |\n| Single verse + chorus | 50–100 | 60–120 | 1:00–2:00 |\n| Full song (2 verses, chorus, bridge) | 100–200 | 180–240 | 3:00–4:00 |\n| Extended (3+ verses, long outro) | 200–350 | 240–360 | 4:00–6:00 |\n| Epic (ballad, progressive) | 350+ | 360–600 | 6:00–10:00 |\n\n**Rule of thumb:** Count lyrics words × 0.8–1.2 seconds per word, then add 20% for instrumental breaks between sections. Always round UP — ACE-Step will fade out naturally if lyrics end before duration.\n\n**Lyrics format:** Use `[Verse 1]`, `[Chorus]`, `[Bridge]`, `[Outro]` tags. ACE-Step follows these to create song structure. Add `[Intro]` and `[Instrumental Break]` tags for non-vocal sections.\n\n**M3 performance (tested, real-world verified June 2026):**\n\nFor 60s audio (standard tier, 2B turbo, 1.7B LM, 8 steps):\n- LM thinking: ~12s (1.7B MLX model, ~50 tok/s)\n- DiT diffusion: ~50s (8 steps × ~6s)\n- VAE decode: ~28s\n- **Total: ~3.5 min per track**\n\nFor 210s audio (3:30 song, same tier):\n- DiT diffusion: ~4-5 min (scales linearly with audio length, ~3x more diffusion work)\n- VAE decode: ~40s\n- **Total: ~5-7 min per track**\n\nAdditional field timings from the 2026-06-12 9-song run on M3 24 GB,\nstandard tier, 8 steps:\n\n| Requested duration | Typical wall time |\n|---|---|\n| 159-200 s | about 9-10 min |\n| 239 s | about 12 min |\n| 302 s | about 14-15 min |\n\nFirst run adds ~90s for model loading. Subsequent runs are faster because the model stays in MPS memory.\n\n**Key advantages over MusicGen:**\n- Real vocals with accurate lyrics\n- Song structure follows `[Verse]`/`[Chorus]` tags\n- Up to 10 minutes (600s) — long enough for full songs\n- MLX native on Apple Silicon (no conda needed)\n- 48kHz stereo output\n\n## ACE-Step Quality Tiers (memory-safe selection)\n\n**TL;DR:** `fast` = quick drafts; `standard` = daily driver (default, ~10 min/210s); `xl-mixed` = best quality but slow on 24GB M3 (~3–4 h/210s, requires extended timeout). Never auto-download; check ML budget first.\n\nACE-Step supports multiple model sizes with different quality/speed/RAM trade-offs. The skill must check available RAM before offering any tier and NEVER auto-download models.\n\n**Tier table:**\n\n| Tier | DiT Model | LM Model | Peak RAM | Disk cost | Speed (210s) | Quality | When to use |\n|---|---|---|---|---|---|---|---|\n| **fast** | `v15-turbo` (2B) | `5Hz-lm-0.6B` (0.6B) | ~8 GB | Included in base ~10 GB | ~5 min | Good | Quick drafts, low-RAM machines |\n| **standard** (default) | `v15-turbo` (2B) | `5Hz-lm-1.7B` (1.7B) | ~11 GB | Included in base ~10 GB | ~10 min | Very Good | Daily driver, most users |\n| **xl-mixed** (24GB M3: viable with extended timeout) | `v15-xl-sft` (4B) | `5Hz-lm-1.7B` (1.7B) | ~25-30 GB | +~20 GB XL DiT download | **~52 min for 60s audio (verified); ~3-4 hours for 210s** | Very High+ | Final production on any RAM tier — just slow on 24GB |\n\n> **Real-world hardware limits (verified June 2026 on M3 24GB unified memory):**\n> - The `best` tier (4B XL + 4B LM, ~22 GB peak) requires ≥32 GB RAM. NOT offered on 24 GB systems.\n> - The `xl-mixed` tier (4B DiT + 1.7B LM) **IS viable on 24 GB M3** if you extend the server timeout:\n>   - Model loads successfully (~10 GB DiT, but MPS pool pressure ~20 GB with cached state)\n>   - 50-step diffusion runs at ~50-100s/step (varies with audio length and memory pressure)\n>   - **The default 600s server timeout fires mid-generation.** Set `ACESTEP_GENERATION_TIMEOUT=3600` to allow up to 1 hour per generation.\n>   - Free RAM goes to 0 GB during generation, but it works\n>   - **Verified June 2026: 60s audio at 50 steps = ~52 min wall-clock on 24GB M3**\n> - **Recommendation for 24 GB M3 (unified memory):**\n>   - For fast iterations: use `standard` tier (10 min for 210s, fast feedback)\n>   - For final production: use `xl-mixed` tier with extended timeout (52 min for 60s, or ~3-4 hours for 210s)\n> - **Recommendation for 32 GB+ M3/M4 (unified memory, more headroom):** `xl-mixed` runs in ~15 min as documented. `best` tier becomes viable.\n> - **For dedicated GPU (NVIDIA/AMD, system RAM separate from VRAM):** A 12 GB GPU + 16 GB system can run xl-mixed in ~15 min. The probe script auto-detects this and uses the smaller pool.\n\n**Memory safety check (run BEFORE any generation):**\n\nThe memory probe must distinguish **unified memory** (Apple Silicon, integrated graphics) from **dedicated memory** (discrete NVIDIA/AMD GPU). On unified memory, the OS, your apps, and the ML model all share the same pool — so 24 GB total might mean only ~18 GB is actually available for ML after macOS and your open apps. On dedicated GPUs, the VRAM is separate from system RAM, so a 12 GB GPU can run a 10 GB model even on a 16 GB system.\n\n```bash\n# 1. Total system RAM and current free\ncase \"$(uname -s)\" in\n  Darwin)\n    TOTAL_RAM_GB=$(sysctl -n hw.memsize | awk '{printf \"%.0f\", $1/1024/1024/1024}')\n    FREE_RAM_KB=$(vm_stat | awk '/free page count/{print $3 * 4}')\n    FREE_RAM_GB=$(awk \"BEGIN {printf \\\"%.1f\\\", $FREE_RAM_KB/1024/1024}\")\n    ;;\n  Linux)\n    TOTAL_RAM_GB=$(awk '/MemTotal/{printf \"%.0f\", $2/1024/1024}' /proc/meminfo)\n    FREE_RAM_GB=$(awk '/MemAvailable/{printf \"%.1f\", $2/1024/1024}' /proc/meminfo)\n    ;;\n  *) echo \"unknown\" ;;\nesac\necho \"Total RAM: ${TOTAL_RAM_GB} GB\"\necho \"Free RAM now: ${FREE_RAM_GB} GB\"\n\n# 2. Memory architecture detection\nif [ \"$(uname -m)\" = \"arm64\" ] && [ \"$(uname -s)\" = \"Darwin\" ]; then\n  MEM_ARCH=\"unified\"\n  echo \"Architecture: unified (Apple Silicon — GPU shares with system)\"\nelif command -v nvidia-smi >/dev/null 2>&1; then\n  GPU_VRAM_GB=$(nvidia-smi --query-gpu=memory.total --format=csv,noheader,nounits | head -1 | awk '{printf \"%.0f\", $1/1024}')\n  GPU_FREE_GB=$(nvidia-smi --query-gpu=memory.free --format=csv,noheader,nounits | head -1 | awk '{printf \"%.0f\", $1/1024}')\n  echo \"GPU: NVIDIA (${GPU_VRAM_GB} GB total VRAM, ${GPU_FREE_GB} GB free)\"\n  echo \"Architecture: dedicated (system RAM and VRAM are separate pools)\"\n  MEM_ARCH=\"dedicated\"\nelif command -v rocm-smi >/dev/null 2>&1; then\n  echo \"Architecture: dedicated (AMD ROCm)\"\n  MEM_ARCH=\"dedicated\"\nelse\n  echo \"Architecture: integrated/CPU-only (system RAM used for everything)\"\n  MEM_ARCH=\"unified\"\nfi\n\n# 3. ML budget calculation\n# Unified: free RAM minus safety margin (OS can reclaim 1-2 GB more on demand)\n# Dedicated: use the SMALLER of free RAM or free VRAM (the bottleneck)\n# Reserve 2 GB safety margin for OS/other apps\nif [ \"$MEM_ARCH\" = \"unified\" ]; then\n  ML_BUDGET_GB=$(awk \"BEGIN {printf \\\"%.0f\\\", $FREE_RAM_GB - 2}\")\n  echo \"ML budget: ~${ML_BUDGET_GB} GB (free RAM minus OS safety margin)\"\nelse\n  # Dedicated GPU: bottleneck is the smaller pool\n  BOTTLENECK_GB=$FREE_RAM_GB\n  if [ -n \"$GPU_FREE_GB\" ] && [ \"$GPU_FREE_GB\" -lt \"$BOTTLENECK_GB\" ]; then\n    BOTTLENECK_GB=$GPU_FREE_GB\n  fi\n  ML_BUDGET_GB=$(awk \"BEGIN {printf \\\"%.0f\\\", $BOTTLENECK_GB - 2}\")\n  echo \"ML budget: ~${ML_BUDGET_GB} GB (smaller of free system RAM or free VRAM, minus safety margin)\"\nfi\n```\n\nBased on ML budget (NOT total RAM):\n\n| ML budget | Available tiers | Use case |\n|---|---|---|\n| < 8 GB | fast only (warn user about tight fit, expect OOM) | Quick drafts |\n| 8–11 GB | fast + standard | Daily driver |\n| 12–20 GB | fast + standard + xl-mixed (with extended timeout) | Final production |\n| ≥ 25 GB | ALL tiers including best (4B LM) | No constraints |\n\n**Why this differs from \"total RAM\" tables:**\n\nA 24 GB Apple Silicon Mac with macOS and 4 GB of open apps has only ~18 GB ML budget. The probe should report 18 GB, and the table should classify it as \"fast + standard\" — NOT \"xl-mixed eligible\" (which the old table would say based on 24 GB total).\n\nConversely, a Windows desktop with 16 GB system RAM and a 24 GB NVIDIA RTX 4090 has 22 GB ML budget (the VRAM is the bottleneck, not the system RAM). That system CAN run xl-mixed in ~15 min as documented.\n\nThe probe and the table together handle both cases correctly. The previous version of this skill used total RAM as the gating value, which was wrong for unified memory — leading to the \"24GB Mac but xl-mixed sounds bad\" surprise we hit in June 2026. The fix is to use ML budget (free memory minus OS safety margin, with unified-vs-dedicated awareness).\n\n**Model download consent flow (NEVER auto-download):**\n\nWhen ACE-Step is running but no models are loaded (fresh install), OR when the user requests a higher tier whose models aren't downloaded yet:\n\n```\nYou: \"ACE-Step is ready but needs audio models before generating. This will\n       download ~10 GB to your disk — I will NOT do this without your explicit\n       approval.\n       \n       Your options:\n       \n       ① Download standard (~10 GB) → good quality, ~10 min/track, fits your {N}GB RAM ✓\n       ② Download xl-mixed (+20 GB extra = ~30 GB total) → best quality your machine can run,\n          ~15 min/track, needs you to close heavy apps during generation\n       ③ Skip local → use a cloud backend instead\n          - MiniMax (if API key set) — fast, paid\n          - Stable Audio (if STABILITY_API_KEY set) — paid\n          - MusicGen (local fallback) — non-commercial weights, instrumental only\n       \n       You currently have {X} GB free disk space.\n       \n       Which option?\"\n```\n\n**Wait for the user to choose before doing anything.** Do not download. Do not auto-load. Do not start generating.\n\nRules:\n- Always show: model size, free disk space, free RAM, expected speed, what the user gets\n- If disk space < model size: offer to free space first, or skip to cloud\n- If RAM < tier requirement: warn clearly, suggest lower tier or cloud\n- User can change tier later without re-downloading (just switch via `/v1/init`)\n- Once models are downloaded, they persist — no need to ask again unless user wants to upgrade tier\n\n**Switching tiers mid-session:**\n\n```bash\n# Switch to xl-mixed (requires XL model already downloaded)\ncurl -s -X POST http://127.0.0.1:8001/v1/init \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"dit_model\": \"acestep-v15-xl-sft\", \"lm_model\": \"acestep-5Hz-lm-1.7B\"}'\n\n# Switch back to standard\ncurl -s -X POST http://127.0.0.1:8001/v1/init \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"dit_model\": \"acestep-v15-turbo\", \"lm_model\": \"acestep-5Hz-lm-1.7B\"}'\n```\n\n**M3 performance by tier (real-world verified, June 2026):**\n\n| Tier | LM time | DiT time (8 steps) | DiT time (50 steps) | VAE decode | First run | Subsequent | Status on 24GB M3 |\n|---|---|---|---|---|---|---|---|\n| fast (2B+0.6B) | ~6s | ~45s | n/a (model is turbo) | ~28s | ~3 min | ~5 min | ✅ Works great |\n| standard (2B+1.7B) | ~12s | ~50s | n/a (model is turbo) | ~28s | ~3 min | ~10 min | ✅ Works great (default) |\n| xl-mixed (4B+1.7B) | ~12s | ~720s (90s/step × 8) | **~1500s for 60s audio (~50s/step × 30+ steps in 30 min) — verified ~52 min wall-clock** | ~65s | ~5 min | **~52 min for 60s; estimated 3-4 hours for 210s** | ✅ Works with `ACESTEP_GENERATION_TIMEOUT=3600` (1 hour). 8-step version produces \"soup\" output — use 50 steps. |\n| best (4B+4B) | n/a | n/a | n/a | n/a | n/a | n/a | ❌ Excluded — needs 32GB+ |\n\n**Critical findings from real-world testing (June 2026):**\n\n1. **xl-mixed CAN RUN on 24 GB M3** (verified 60s audio at 50 steps completes in ~52 min) with the right env vars:\n   ```bash\n   ACESTEP_CONFIG_PATH=acestep-v15-xl-sft \\\n   ACESTEP_LM_MODEL_PATH=acestep-5Hz-lm-1.7B \\\n   ACESTEP_GENERATION_TIMEOUT=3600 \\\n   PYTORCH_MPS_HIGH_WATERMARK_RATIO=0.0 \\\n   uv run acestep-api --port 8001\n   ```\n   - Without `ACESTEP_GENERATION_TIMEOUT=3600`, the default 600s (10 min) timeout fires mid-generation\n   - Without `PYTORCH_MPS_HIGH_WATERMARK_RATIO=0.0`, the model fails to load (MPS OOM)\n   - **⚠️ Generation completes but audio quality is poor** — high-frequency noise, unclear vocals, \"no sense samples\". Use standard tier for now; xl-mixed needs further tuning (see the \"XL 50-step fixes to try\" section below).\n\n2. **XL + 8 steps produces \"soup\" output** — all elements at the same level, no dynamics. Always use 50 steps for XL.\n   - Verified: LRA of XL+8 steps = 1.8-4.8 LU (very compressed)\n   - Verified: LRA of XL+50 steps = 4.0+ LU (more dynamic)\n   - 60s XL+50 step audio on 24GB M3 = ~52 min wall-clock\n\n3. **Audio length affects time per step:**\n   - 60s audio: ~50-100s/step\n   - 210s audio: ~90-130s/step (more diffusion work per step)\n   - For 210s at 50 steps: 90s × 50 = 75 min minimum, up to 3-4 hours with swap pressure\n\n4. **Memory pressure is the bottleneck, not model loading.** Models load fine; generation just runs slow due to swap-thrashing.\n\n**XL 50-step fixes to try** (in order of likelihood, each test takes ~52 min for 60s audio):\n\nIf you want to experiment with xl-mixed anyway, the most likely fixes are:\n\n1. **Detailed prompt with all 6 metas** (BPM, key, time signature, vocal language, duration, genre filled explicitly in the request body) — the LM benefits from explicit anchors rather than just genre tags\n2. **Add `use_format: true`** — lets the LM enhance your prompt\n3. **Try `shift: 1.0` or `shift: 5.0`** (default is 3.0, officially documented for \"base models, not turbo\" — xl-sft is SFT, not turbo, so worth experimenting)\n4. **Try `guidance_scale: 4.0`** (default 7.0 may be too aggressive for sft CFG)\n5. **Try `infer_method: \"sde\"`** (stochastic, sometimes more stable than Euler for SFT models)\n6. **Try `thinking: false`** (DiT-only mode, skips LM) — if this works, the LM is the problem; if it still sounds bad, the DiT is the problem\n7. **Try `xl-turbo` instead of `xl-sft`** (counterintuitive but turbo is designed for fewer steps)\n\nStart with #1 (free, just data change) and work down. If none of these produce acceptable audio, fall back to the standard tier.\n\n**Conclusion:** `standard` is the practical best quality tier for 24 GB M3. Reserve `xl-mixed` for 32GB+ hardware (M3 Max/Ultra, M4 Max).\n\n## ACE-Step Audio-Conditioned Generation (Cover, Repaint, Reference Audio)\n\nBeyond text2music, ACE-Step 1.5 conditions on an input audio file. This is how you do a\n**melody-aware local cover** — no cloud needed. Select the mode with `task_type` in the\n`/release_task` body:\n\n| `task_type` | What it does | Audio field |\n|---|---|---|\n| `text2music` (default) | Generate from caption + lyrics | none |\n| `cover` | Re-style a song while following its melody/structure | `src_audio` |\n| `repaint` | Regenerate only a time window, keep the rest | `src_audio` + `repainting_start/end` |\n| `extract` | Stem separation | `src_audio` |\n\n**Uploading the source audio (important):** the API **rejects absolute file paths**\n(`{\"detail\":\"absolute audio file paths are not allowed\"}`). Upload the file as multipart\nform-data, not JSON. Fields: `src_audio` (source for cover/repaint) or\n`reference_audio`/`ref_audio` (style-transfer reference). Send other params as form fields:\n\n```bash\ncurl -s -X POST http://127.0.0.1:8001/release_task \\\n  -F \"task_type=cover\" \\\n  -F \"src_audio=@/path/to/song.wav\" \\\n  -F \"audio_cover_strength=0.35\" \\\n  -F \"prompt=dreamy 80s synthwave, warm analog synths, gated-reverb drums, arpeggiated bass, neon night-drive mood\" \\\n  -F \"bpm=129\" -F \"key_scale=D major\" -F \"audio_format=wav\"\n```\n\n**Cover behavior (verified):**\n- `audio_cover_strength` (0.0–1.0): **lower = bigger restyle** (~0.2–0.4 for a strong genre jump\n  like rock to synthwave; 0.7–0.9 for a subtle restyle; 1.0 = closest to source).\n- The **LM is skipped** for cover/repaint/extract — `thinking` has no effect; the caption and\n  lyrics you send are used directly, so write a good caption.\n- **Duration auto-locks to the source length** — `audio_duration` is ignored for cover.\n- `reference_audio` (style transfer) conditions global timbre/feel, NOT melody; `src_audio`\n  (cover) follows melody/structure. Melody capture is best on sparse, mid/slow-tempo songs —\n  expect *melodic variation*, not an exact copy.\n\n**VRAM / time caveat (verified on a 12 GB laptop GPU):** a full ~5-minute cover is impractical\non this class of hardware — encoding the source alone took ~13 minutes and the job hit the\nserver's **default 600 s generation timeout** and failed. Mitigations:\n- **Cover a shorter segment** — trim the source first: `ffmpeg -ss <start> -t 60 -i in.wav out.wav`.\n- **Raise the timeout** when starting the server: `ACESTEP_GENERATION_TIMEOUT=3600`.\n- Full-length covers are realistic on ≥20 GB VRAM.\n\n**Routing gate:** use ACE-Step cover only when the user explicitly accepts a local,\nslower, experimental workflow and the source length/hardware budget make sense.\nUse `music-craft-minimax` for fast cloud cover, long-source turnaround, mashups,\nor when the user needs MiniMax-native controls and analysis scripts.\n\n**Repaint** (fix one bad section instead of regenerating the whole track): `task_type=repaint`,\nupload `src_audio`, set `repainting_start`/`repainting_end` (seconds) and `repaint_mode`\n(`conservative`/`balanced`/`aggressive`) with `repaint_strength` (0–1). Use your structural\nanalysis to choose the window.\n\n**Stem/remix helper path:** ACE-Step `extract` is experimental. For local arranger\nexperiments, prefer `scripts/extract_stems.py` when Demucs is installed; it writes\n`stems.json` with normalized `vocals`, `drums`, `bass`, and `other` paths. Then\nuse `scripts/remix_stems.py --stems-json <stems.json> --output <mix.mp3> --dry-run`\nto inspect the preview-quality `ffmpeg amix` command before rendering. This is\nnot DAW-quality mixing.\n\n**Local audio understanding (no cloud):** ACE-Step can extract BPM, key, time-signature, and a\ncaption directly from an input file (the `analysis_only` / `full_analysis_only` request flags;\nthe \"Audio Understanding\" feature). This is a fully-local way to derive metas/caption from a\nsource song — an alternative to the librosa pipeline when ACE-Step is already running.\n\n**Operational note (single-worker API):** the REST server processes one job at a time and may\nnot answer `/query_result` within a short timeout while mid-generation (60 s poll timeouts are\nnormal under load). Use a generous client timeout, tolerate poll timeouts, and detect\ncompletion by watching `.cache/acestep/tmp/api_audio/` for new files.\n\n**Quality loop:** generate a small batch (`batch_size` 2–4) and keep the best; request\n`audio_format: \"wav\"` to avoid a lossy MP3 round-trip; set `seed` with `use_random_seed=false`\nfor reproducibility.\n\nFile v1.6.0:references/acestep-shift-schedule.md\n\n# ACE-Step Shift Schedule Taxonomy\n\nReference for the ACE-Step 1.5 **shift** parameter: what it controls in the\nflow-matching diffusion process, how the three documented schedules\n(`shift=1.0`, `shift=3.0`, and **continuous shift**) map to each model tier\n(turbo, base, sft, xl-base, xl-mixed), and how to pick the right shift for\nthe request at hand. Load this when a request mentions `shift`,\ntimestep-shifting, or any of the four turbo variants (`turbo`,\n`turbo-shift1`, `turbo-shift3`, `turbo-continuous`).\n\n> **Status:** docs-only. Backed by `music-craft_ROADMAP.md` item **13i** and\n> verified upstream against `ACE-Step-1.5/docs/en/INFERENCE.md` +\n> `Tutorial.md`. No runtime changes; if the upstream semantics shift\n> (e.g. a new default), update § 1 first.\n\n## TL;DR\n\n- **`shift` is a timestep-reshaping factor in flow-matching diffusion.**\n  When `shift != 1.0`, the scheduler applies\n  `t' = shift * t / (1 + (shift - 1) * t)` to every timestep `t` before\n  the step. Higher `shift` = the early (high-noise, structure-defining)\n  timesteps get **more compute**, low-noise detail timesteps get **less**.\n- **Three documented schedules:**\n  - `shift=1.0` — **default in the upstream `GenerationParams`**. Flat\n    schedule; the original cosine/linear timesteps are untouched. Pairs\n    with turbo's few-step (8) inference.\n  - `shift=3.0` — **documented default for base / sft / xl-base / xl-sft\n    checkpoints** (per upstream Tutorial + INFERENCE.md). Steep schedule;\n    more effort on structure, less on micro-detail. The \"best quality\"\n    default in upstream docs.\n  - **continuous shift** — a time-varying schedule where the shift\n    factor changes across timesteps instead of staying constant.\n    Available in the `turbo-continuous` checkpoint (experimental) and\n    as a research lever on custom flow-matching pipelines.\n- **Tier-to-shift mapping:**\n\n  | Tier | Default shift | Notes |\n  | --- | --- | --- |\n  | `turbo` (2B, 8 steps) | `1.0` (joint-distilled 1/2/3) | Most flexible; works at any shift in 1.0–5.0 |\n  | `turbo-shift1` | `1.0` | Distilled only at shift=1; richer details, weaker semantics |\n  | `turbo-shift3` | `3.0` | Distilled only at shift=3; clearer, drier, less orchestration |\n  | `turbo-continuous` | continuous (1–5) | Experimental; not thoroughly tested upstream |\n  | `base` / `xl-base` | `3.0` | Recommended; try `1.0` or `5.0` if defaults feel off |\n  | `sft` / `xl-sft` | `3.0` (experiment 1.0–5.0) | SFT is not strictly turbo; shift is **applicable** |\n  | `xl-mixed` (4B DiT + LM) | `3.0` | Inherits XL DiT default; same `1.0–5.0` envelope |\n\n## 1. What `shift` controls\n\n### 1.1 The formula\n\nWhen `shift != 1.0`, the scheduler runs the standard timestep list\nthrough:\n\n```\nt' = shift * t / (1 + (shift - 1) * t)\n```\n\nwith `t` in `[0.0, 1.0]`. The transformation is monotonic; it just\n**stretches or compresses** the noise schedule. The shape of the\ndenoising trajectory changes — early (high-noise) steps get a bigger\nshare of the step budget, or a smaller share, depending on `shift`.\n\n### 1.2 Plain-language effect\n\nUpstream `Tutorial.md` describes it as \"attention allocation\" during\ndenoising. The mental model:\n\n| Shift value | Step allocation | Sound character |\n| --- | --- | --- |\n| **Lower** (`shift ≈ 1.0`) | Evenly distributed. Roughly equal compute per timestep. | \"Draw and fix simultaneously.\" More micro-detail; risk of noisy detail. |\n| **Higher** (`shift ≈ 3.0`) | Front-loaded. Early (high-noise, structure-defining) timesteps get more compute; late (low-noise, fine-detail) timesteps get less. | \"Draw outline first, then fill.\" Stronger semantics, cleaner overall framework, potentially drier arrangement. |\n\n**What `shift` is NOT:**\n\n- **Not a vocal/instrument separation knob.** Stem separation lives in\n  the VAE; tuning `shift` will not change how cleanly the model pulls\n  vocals out of the mix.\n- **Not a prompt-adherence knob.** That's `guidance_scale` (CFG).\n- **Not a speed knob.** `shift` reshapes the schedule but does not add\n  or remove steps. Wall-clock per step is the same; total wall-clock\n  changes only because the model may converge differently.\n\n### 1.3 The upstream documentation nuance\n\nUpstream `INFERENCE.md` says the parameter is \"**only effective for\nbase models, not turbo models**\" — the SDK doc was written before the\nshift-distilled turbo variants (`turbo-shift1`, `turbo-shift3`,\n`turbo-continuous`) shipped. Read that line as \"the **shift-aware\nbehavior** is best understood on base models; turbo models are\nsensitive in different ways\". In practice:\n\n- `turbo` (default, joint-distilled 1/2/3) **accepts any `shift` in\n  `1.0–5.0` without complaint**; the effect is muted because the model\n  was distilled on three different schedules and learned to handle all\n  of them.\n- `turbo-shift1` is **biased toward `shift=1.0`**; values far from 1.0\n  hurt quality.\n- `turbo-shift3` is **biased toward `shift=3.0`**; values far from 3.0\n  hurt quality.\n- `turbo-continuous` **expects a continuous schedule** (1.0 → 5.0\n  across the run), not a constant.\n- `base` / `sft` / `xl-base` / `xl-sft` — `shift` is **fully effective**\n  and is the standard knob to tune.\n\n## 2. Shift variants\n\n### 2.1 `shift=1.0` — flat schedule (turbo default)\n\n```python\nGenerationParams(\n    caption=\"upbeat electronic dance music\",\n    inference_steps=8,\n    shift=1.0,\n    infer_method=\"ode\",\n)\n```\n\n- **What it does:** the noise schedule is **unchanged**. The denoising\n  trajectory is the unmodified upstream default (typically a linear or\n  cosine schedule from `t=1.0` to `t=0.0`).\n- **When to use:** default for any `*turbo` checkpoint at 8 steps.\n  Pairs naturally with `turbo-shift1`'s distillation; works on the\n  joint-distilled `turbo` and `xl-turbo` without complaint.\n- **Pros:** safest, most predictable, no surprise side effects from\n  schedule reshaping.\n- **Cons:** on long-form or detail-heavy prompts the structure may feel\n  under-anchored (the model spreads effort evenly, so the chorus\n  doesn't stand out from the verse).\n\n### 2.2 `shift=3.0` — steep schedule (base/sft default)\n\n```python\nGenerationParams(\n    caption=\"intricate jazz fusion with complex harmonies\",\n    inference_steps=50,\n    guidance_scale=7.0,\n    use_adg=True,\n    shift=3.0,\n    seed=42,\n)\n```\n\n- **What it does:** reshapes the timestep grid so the early, high-noise\n  steps (where structure, melody, and section shape are decided) get\n  more compute. Late, low-noise steps (where micro-detail is rendered)\n  get less.\n- **When to use:** the documented default for `base`, `sft`, `xl-base`,\n  `xl-sft`, and `xl-mixed`. **Must be passed explicitly** in Diffusers\n  and `/release_task` JSON bodies for these checkpoints — see\n  [`acestep-xl-models.md`](acestep-xl-models.md) § \"Required parameters\n  (FOOTGUNS — read first)\".\n- **Pros:** strong semantic coherence, cleaner chorus-vs-verse contrast,\n  better BPM / key adherence on long-form outputs.\n- **Cons:** can sound \"dry\" or under-arranged if the model didn't have\n  enough steps to spend on detail — try `shift=1.0` if 3.0 feels\n  sterile.\n\n### 2.3 Continuous shift (1.0 → 5.0 across the run)\n\n```python\n# Upstream exposes continuous shift via the turbo-continuous checkpoint\n# and through the custom `timesteps` parameter on GenerationParams.\n# The default API surfaces it as a list of timesteps you provide.\n\nGenerationParams(\n    caption=\"ambient cinematic buildup with slow crescendo\",\n    # Custom schedule: front-load shift=1.0 (broad strokes), then\n    # ramp up to shift=5.0 (tightening detail). The list is the actual\n    # timestep grid the model denoises on.\n    timesteps=[0.97, 0.85, 0.70, 0.55, 0.40, 0.28, 0.18, 0.10, 0.05, 0.0],\n)\n```\n\n- **What it does:** instead of a single constant `shift`, the model\n  uses a **time-varying shift** so different timesteps get different\n  compute treatment. Two common patterns:\n  - **Front-load (broad-then-narrow):** low shift early for global\n    structure, high shift late for crisp detail. Useful for ambient /\n    cinematic where macro arc matters.\n  - **Back-load (narrow-then-broad):** high shift early for a strong\n    skeleton, low shift late for richer micro-detail. Useful for\n    detail-rich genres (jazz fusion, orchestral).\n- **When to use:** only on the `turbo-continuous` checkpoint, or when\n  you supply a custom `timesteps` list on `GenerationParams`. Upstream\n  flag: experimental; \"not thoroughly tested\".\n- **Pros:** can outperform both `shift=1.0` and `shift=3.0` on prompts\n  that need both a strong macro arc AND rich micro-detail (the\n  schedule paradox that a constant shift cannot resolve).\n- **Cons:** no documented \"best\" envelope; requires A/B testing per\n  prompt; can produce inconsistent results if the schedule is\n  mis-shaped.\n\n### 2.4 Quick comparison table\n\n| Aspect | `shift=1.0` | `shift=3.0` | continuous shift |\n| --- | --- | --- | --- |\n| Schedule shape | Linear / unmodified upstream | Steep; front-loaded | Variable per timestep |\n| Compute on structure | Even | High | Configurable |\n| Compute on detail | Even | Lower | Configurable |\n| Default tier | turbo (8 steps) | base / sft / xl-* (50 steps) | turbo-continuous only |\n| Sensitivity to choice | Low on joint-distilled turbo | Medium on SFT | High (per-step schedule) |\n| Quality ceiling | \"Good, balanced\" | \"Best quality\" upstream default | \"Highest potential, lowest reproducibility\" |\n| Reproducibility | High | High | Low (depends on schedule shape) |\n\n## 3. Tier-to-shift mapping table\n\nComprehensive map of ACE-Step 1.5 model checkpoints to their recommended\nshift schedule. Sources:\n[`ACE-Step-1.5/README.md`](/Users/luis/Repos/ACE-Step-1.5/README.md) § Model Zoo\n(DiT Models / XL DiT Models), [`docs/en/Tutorial.md`](/Users/luis/Repos/ACE-Step-1.5/docs/en/Tutorial.md) § DiT\nModels, [`docs/en/INFERENCE.md`](/Users/luis/Repos/ACE-Step-1.5/docs/en/INFERENCE.md) § GenerationParams.\n\n| Tier | HF checkpoint | Distillation config | Recommended `shift` | Acceptable range | Notes |\n| --- | --- | --- | --- | --- | --- |\n| **turbo (default)** | `acestep-v15-turbo` | Joint distillation on shift 1, 2, 3 | `1.0` (or omit) | `1.0`–`5.0` | Most flexible. Default for daily-driver and 8-step inference. The skill's `standard` tier uses this with `shift=1.0` (effectively omitted). |\n| **turbo-shift1** | `acestep-v15-turbo-shift1` | Distilled only on shift=1 | `1.0` (strict) | `0.9`–`1.2` | Richer micro-detail, weaker semantic anchor. Best for dense arrangements where detail matters more than structure. |\n| **turbo-shift3** | `acestep-v15-turbo-shift3` | Distilled only on shift=3 | `3.0` (strict) | `2.5`–`3.5` | Clearer, drier, more minimal orchestration. Best for sparse, structurally-driven music (ballads, ambient, classical). |\n| **turbo-continuous** | `acestep-v15-turbo-continuous` | Experimental; supports continuous shift 1–5 | custom schedule | continuous (1.0–5.0) | Most flexible; least tested. Best for prompts that need both macro arc and micro-detail. Not recommended for production without smoke tests. |\n| **xl-turbo** | `acestep-v15-xl-turbo` | Joint distillation (XL variant of default turbo) | `1.0` | `1.0`–`5.0` | Same behavior as `turbo` but on the 4B DiT. The skill's `xl-turbo` reference tier. |\n| **sft** | `acestep-v15-sft` | SFT (50 steps, CFG-able) | `3.0` (try 1.0–5.0) | `1.0`–`5.0` | SFT is not turbo. The upstream \"only effective for base models\" line in `INFERENCE.md` predates this finding; SFT responds to shift tuning. |\n| **xl-sft** | `acestep-v15-xl-sft` | SFT on 4B DiT | `3.0` (try 1.0–5.0) | `1.0`–`5.0` | The skill's `xl-mixed` reference tier. Same caveat as `sft`: shift is effective. |\n| **base** | `acestep-v15-base` | BASE (50 steps, full CFG, ADG-capable) | `3.0` | `1.0`–`5.0` | BASE is the original target for `shift=3.0` upstream. Required for `extract`/`lego`/`complete`. |\n| **xl-base** | `acestep-v15-xl-base` | BASE on 4B DiT | `3.0` ⚠️ MUST pass explicitly | `1.0`–`5.0` | Documented default per `INFERENCE.md` § \"Advanced DiT Parameters\". Pinned in [`acestep-xl-models.md`](acestep-xl-models.md) § 3. |\n| **xl-mixed** | `acestep-v15-xl-base` + 4B LM | BASE + 4B LM | `3.0` | `1.0`–`5.0` | Inherits `xl-base`'s shift default. Only viable on ≥32 GB hardware (M3 Max/Ultra or 24 GB VRAM GPU + 16 GB system). |\n\n## 4. Decision tree — which shift for this request?\n\n```text\nWhat's the loaded DiT checkpoint?\n│\n├── turbo (any variant) at 8 steps\n│   │\n│   ├── Default behavior needed (no specific quality axis to push)?\n│   │   └── shift=1.0   # safest; what the joint-distilled checkpoint\n│   │                   # was trained to handle equally well\n│   │\n│   ├── Need richer micro-detail (dense arrangement, electronic, dense vocal stacks)?\n│   │   └── switch to turbo-shift1 checkpoint + shift=1.0\n│   │       # stronger at \"draw and fix simultaneously\"\n│   │\n│   ├── Need clearer structure / drier arrangement (ballad, ambient, classical)?\n│   │   └── switch to turbo-shift3 checkpoint + shift=3.0\n│   │       # stronger at \"outline first, fill later\"\n│   │\n│   ├── Need both strong macro arc AND rich detail (cinematic, jazz fusion)?\n│   │   └── switch to turbo-continuous + custom timesteps schedule\n│   │       # experimental; smoke-test before production\n│   │\n│   └── Just trying things / on a budget?\n│       └── shift=1.0  # don't pay the schedule-shaping tax\n│\n├── base / xl-base (50 steps, full CFG, BASE-only tasks)\n│   ├── First attempt / unknown prompt shape?\n│   │   └── shift=3.0   # the documented default; matches upstream training\n│   │\n│   ├── Output feels \"dry\" / under-arranged?\n│   │   └── try shift=1.5  # ease the schedule toward even\n│   │       # or shift=1.0 if 1.5 still feels under-structured\n│   │\n│   ├── Output feels \"mushy\" / detail too noisy?\n│   │   └── try shift=4.5 or shift=5.0  # sharpen structure at the\n│   │                                    # cost of detail\n│   │\n│   ├── Doing extract / lego / complete (BASE-only tasks)?\n│   │   └── shift=3.0  # matches the per-task defaults in\n│   │                  # acestep-task-types.md § 3-5\n│   │\n│   └── Fine-tuning / LoRA training?\n│       └── shift=3.0  # match the upstream training distribution\n│\n├── sft / xl-sft (50 steps, CFG-able)\n│   ├── First attempt / unknown prompt shape?\n│   │   └── shift=3.0  # documented default; best starting point\n│   │\n│   ├── Vocals render harsh at shift=3.0?\n│   │   └── try shift=1.0  # smoother detail phase = softer sibilants\n│   │\n│   ├── Output structure feels flat (verse ≈ chorus)?\n│   │   └── try shift=4.0  # push the macro arc harder\n│   │\n│   └── Switching from xl-sft to xl-base (BASE-only task)?\n│       └── keep shift=3.0  # both checkpoints use the same default\n│\n└── xl-mixed (xl-base + 4B LM)\n    ├── First attempt / unknown prompt shape?\n    │   └── shift=3.0  # inherits xl-base default\n    │\n    ├── LM-driven planning is producing good captions but DiT still feels off?\n    │   └── try shift=1.0 or shift=5.0  # isolate whether the schedule\n    │                                   # is the issue or the LM is\n    │\n    └── On 24 GB M3 and the generation is too slow?\n        └── fall back to xl-sft + 1.7B LM at shift=3.0  # the skill's\n                                                       # documented\n                                                       # compromise tier\n```\n\n## 5. Impact on quality\n\n### 5.1 Qualitative differences\n\n| Aspect | `shift=1.0` | `shift=3.0` | Continuous |\n| --- | --- | --- | --- |\n| **Arrangement richness** | Higher (more steps on detail) | Lower (more steps on structure) | Configurable per phase |\n| **Macro arc (verse → chorus lift)** | Flatter contrast | Sharper contrast | Configurable |\n| **Vocal intelligibility** | Higher at low step counts | Comparable at 50 steps | Comparable |\n| **Orchestration density** | Higher | Lower (can sound \"thin\") | Configurable |\n| **Prompt adherence (long prompts)** | Lower (less structure compute) | Higher | Depends on schedule |\n| **Reproducibility across seeds** | High | High | Lower (more variables) |\n| **Vocal harshness / sibilance** | Slightly softer | Slightly harsher at the same `guidance_scale` | Depends |\n\n### 5.2 When `shift=3.0` clearly wins\n\n- Long-form outputs (≥ 120 s) where the macro arc is the point.\n- Genres with strong verse/chorus contrast (pop, rock, EDM build/drop).\n- BASE-only tasks (`extract`, `lego`, `complete`) where upstream docs\n  pin 3.0.\n- When `guidance_scale ≥ 7.0` — the steeper schedule prevents CFG\n  from over-fitting to late-step noise.\n\n### 5.3 When `shift=1.0` clearly wins\n\n- Short outputs (≤ 60 s) where structure compute doesn't matter much.\n- Genres where micro-detail is the point (jazz fusion, complex\n  electronic, orchestral).\n- When you need **softer vocals** at high CFG (xl-sft at\n  `guidance_scale=7.0` is often harsh — `shift=1.0` + `guidance_scale=5.0`\n  is the gentler combo).\n\n### 5.4 When continuous shift is worth the complexity\n\n- Cinematic / trailer cues that need a strong opening arc AND dense\n  detail in the climax.\n- Jazz fusion / progressive where you want section-by-section schedule\n  variation.\n- Custom fine-tunes that target a specific schedule shape.\n\n## 6. Impact on speed\n\n`shift` **does not change the number of steps** and **does not change\nthe wall-clock per step**. It changes:\n\n| Aspect | Effect of `shift` |\n| --- | --- |\n| **Wall-clock per step** | None — same compute per step regardless of shift |\n| **Total wall-clock** | Roughly constant — same step count, same per-step time |\n| **Convergence quality per step** | Higher `shift` = early steps contribute more to the final output. With a well-shaped schedule, you may converge in fewer steps; with a mis-shaped schedule, you may waste steps. |\n| **Effective step count** | Higher `shift` effectively concentrates \"useful\" compute into fewer steps. Lower `shift` spreads it out. Net effect on a fixed `inference_steps` budget: `shift=3.0` at 8 steps may look like `shift=1.0` at 12 steps in quality. |\n\n**Practical implication for the skill:**\n\n- The standard-tier (`turbo`, 8 steps) wall-clock is **the same**\n  regardless of `shift` setting. Choose `shift` based on desired\n  character, not speed.\n- The xl-mixed tier (4B DiT, 50 steps) is **dominated by per-step\n  time**; `shift` does not materially change wall-clock. The 52-min /\n  60 s figure from [`acestep-generation.md`](acestep-generation.md) §\n  \"M3 performance by tier\" is `shift`-agnostic.\n\n## 7. Examples\n\n### 7.1 — `turbo` (standard tier, 8 steps, `shift=1.0`)\n\n```bash\nTASK_ID=$(curl -s -X POST http://127.0.0.1:8001/release_task \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"prompt\": \"groovy funk track with slap bass, tight horn stabs, rhythmic guitar scratching\",\n    \"audio_duration\": 60,\n    \"thinking\": true,\n    \"inference_steps\": 8,\n    \"shift\": 1.0,\n    \"infer_method\": \"ode\"\n  }' | python3 -c \"import json,sys; print(json.load(sys.stdin).get('data',{}).get('task_id',''))\")\n```\n\n### 7.2 — `xl-base` (BASE, 50 steps, `shift=3.0` — the documented default)\n\n```bash\nTASK_ID=$(curl -s -X POST http://127.0.0.1:8001/release_task \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"prompt\": \"dreamy 80s synthwave, warm analog synths, gated-reverb drums, arpeggiated bass\",\n    \"lyrics\": \"[Verse]\\nneon lights on a vacant street\\n\\n[Chorus]\\nwe are the night\",\n    \"audio_duration\": 210,\n    \"bpm\": 96,\n    \"key_scale\": \"D major\",\n    \"time_signature\": \"4/4\",\n    \"vocal_language\": \"en\",\n    \"thinking\": true,\n    \"inference_steps\": 50,\n    \"guidance_scale\": 7.0,\n    \"shift\": 3.0,\n    \"infer_method\": \"ode\",\n    \"use_adg\": true,\n    \"audio_format\": \"wav\"\n  }' | python3 -c \"import json,sys; print(json.load(sys.stdin).get('data',{}).get('task_id',''))\")\n```\n\n### 7.3 — `xl-sft` with `shift=1.0` (softer vocals experiment)\n\n```bash\nTASK_ID=$(curl -s -X POST http://127.0.0.1:8001/release_task \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"prompt\": \"warm acoustic ballad with breathy female vocal\",\n    \"lyrics\": \"[Verse]\\nwalking through the morning light\",\n    \"audio_duration\": 60,\n    \"thinking\": true,\n    \"inference_steps\": 50,\n    \"guidance_scale\": 5.0,\n    \"shift\": 1.0,\n    \"infer_method\": \"ode\"\n  }' | python3 -c \"import json,sys; print(json.load(sys.stdin).get('data',{}).get('task_id',''))\")\n```\n\n### 7.4 — `xl-sft` with `shift=5.0` (push the macro arc)\n\n```bash\nTASK_ID=$(curl -s -X POST http://127.0.0.1:8001/release_task \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"prompt\": \"anthemic pop with explosive chorus, verse quiet and intimate, chorus huge and cathartic\",\n    \"lyrics\": \"[Verse]\\nquietly holding on\\n\\n[Chorus]\\nwe rise together\",\n    \"audio_duration\": 60,\n    \"thinking\": true,\n    \"inference_steps\": 50,\n    \"guidance_scale\": 7.0,\n    \"shift\": 5.0,\n    \"infer_method\": \"ode\"\n  }' | python3 -c \"import json,sys; print(json.load(sys.stdin).get('data',{}).get('task_id',''))\")\n```\n\n### 7.5 — Continuous shift via custom `timesteps`\n\n```python\nfrom acestep.inference import GenerationParams, GenerationConfig, generate_music\n\n# Continuous schedule: front-load shift=1.0 (broad strokes),\n# then ramp up to shift=5.0 (tightening detail) by providing\n# the actual timestep grid the model will denoise on.\nparams = GenerationParams(\n    caption=\"ambient cinematic buildup with slow crescendo\",\n    timesteps=[0.97, 0.85, 0.70, 0.55, 0.40, 0.28, 0.18, 0.10, 0.05, 0.0],\n    # When `timesteps` is provided, it overrides `inference_steps` and `shift`.\n    thinking=True,\n)\n\nconfig = GenerationConfig(batch_size=1, audio_format=\"flac\")\nresult = generate_music(dit_handler, llm_handler, params, config, save_dir=\"/output\")\n```\n\n### 7.6 — Switching to a shift-distilled turbo variant\n\n```bash\n# 1. Stop the server\n# 2. Set ACESTEP_CONFIG_PATH to the shift-distilled checkpoint\nACESTEP_CONFIG_PATH=acestep-v15-turbo-shift3 \\\nACESTEP_LM_MODEL_PATH=acestep-5Hz-lm-1.7B \\\nACESTEP_LM_BACKEND=mlx \\\nuv run acestep-api --port 8001\n\n# 3. Pass shift=3.0 (matches the distillation)\nTASK_ID=$(curl -s -X POST http://127.0.0.1:8001/release_task \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"prompt\": \"minimalist piano ballad, sparse arrangement, intimate\",\n    \"inference_steps\": 8,\n    \"shift\": 3.0\n  }' | python3 -c \"import json,sys; print(json.load(sys.stdin).get('data',{}).get('task_id',''))\")\n```\n\n### 7.7 — A/B test recipe (locked seed, only `shift` changes)\n\n```bash\n# Pick one prompt and one seed; sweep shift in 1.0, 2.0, 3.0, 4.0, 5.0.\n# Compare the five outputs by listening + ebur128 LRA.\n\nPROMPT='{\"prompt\":\"warm indie folk, fingerstyle guitar, soft brushed drums\",\"audio_duration\":60,\"thinking\":true,\"inference_steps\":8,\"seed\":42,\"shift\":1.0,\"audio_format\":\"wav\"}'\n\ncurl -s -X POST http://127.0.0.1:8001/release_task \\\n  -H \"Content-Type: application/json\" \\\n  -d \"${PROMPT//shift\\\":1.0/shift\\\":2.0}\"\n\n# Repeat with shift=3.0, 4.0, 5.0. Listen to all five back-to-back.\n```\n\n## 8. Troubleshooting\n\n| Symptom | Likely cause | Fix |\n| --- | --- | --- |\n| Output feels \"mushy\" / flat structure (verse ≈ chorus) | `shift` is too low for the prompt's macro arc | Bump `shift` toward 3.0; on SFT/XL, try 4.0 or 5.0 |\n| Output feels \"thin\" / under-arranged | `shift` is too high for the prompt's detail needs | Drop `shift` toward 1.0; if already at 1.0, try `inference_steps=64` |\n| Vocals render harsh / sibilant | `shift=3.0` + `guidance_scale ≥ 7.0` over-fits to late-step noise | Drop to `shift=1.0` AND `guidance_scale=5.0` (the softer combo for SFT — see [`acestep-xl-models.md`](acestep-xl-models.md) § 3) |\n| BASE-only task (`extract`/`lego`/`complete`) silently fails | Wrong shift for the loaded BASE checkpoint | Confirm `shift=3.0` is passed; BASE tasks default to shift=3.0 in the API but won't reject wrong values — they'll just produce subpar output |\n| 8-step XL produces \"soup\" output | `shift` is not the cause — `num_inference_steps` is | This is the `num_inference_steps=50` footgun from [`acestep-xl-models.md`](acestep-xl-models.md) § 3, not a shift issue. Pass `inference_steps=50` explicitly. |\n| `shift=1.0` on `turbo-shift3` produces noisy detail | Wrong shift for the loaded checkpoint | Switch to `turbo-shift1` for `shift=1.0` workloads, or accept the `shift=3.0` default for `turbo-shift3` |\n| Continuous shift (`timesteps=[…]`) ignores `shift` parameter | By design — custom `timesteps` overrides `shift` | If you want a continuous schedule, supply `timesteps` and don't pass `shift`. The two are mutually exclusive. |\n| Same `shift`, different `inference_methods` produce different outputs | `infer_method=\"sde\"` injects stochastic noise per step | Switch to `infer_method=\"ode\"` for deterministic comparison; or accept the variability if using `sde` |\n| Output is identical across seeds | Probably `num_inference_steps=8` on XL (wrong footgun) — not a shift issue | See 8-step XL footgun above |\n| Output sounds great on `turbo` but \"off\" on `xl-turbo` | The XL variant is more sensitive to schedule mismatches | Match the `shift` to the prompt shape more carefully on XL; the 4B DiT is less forgiving of mis-shaped schedules |\n| Custom `timesteps` schedule produces artifacts | Schedule is mis-shaped (e.g. monotonicity violation, too few points) | Verify the schedule is strictly decreasing from `~1.0` to `0.0`; aim for 8–16 points; cross-check by running the same schedule with `infer_method=\"ode\"` (deterministic) to isolate the schedule from sampling noise |\n| BMAD shift-sweep shows no audible difference | Either the prompt is shift-insensitive (e.g. simple percussive loop) or the step count is too low to reveal schedule effects | Accept the result; or raise `inference_steps` to 32+ on SFT/XL to expose schedule effects |\n\n## 9. Footguns (must read)\n\nThese three footguns are specific to `shift` and silently degrade\nquality if missed. They sit alongside the `num_inference_steps=50` and\n`shift=3.0` footguns already documented in\n[`acestep-xl-models.md`](acestep-xl-models.md) § 3.\n\n1. **`shift` does NOT control vocal/instrument separation.** That is a\n   VAE-level operation. Don't tune `shift` to chase cleaner stems —\n   tune `audio_cover_strength` (for cover) or use Demucs for full\n   stem separation.\n\n2. **`shift` is fully effective on `sft` / `xl-sft` even though the\n   upstream `INFERENCE.md` says \"only effective for base models\".** That\n   doc line was written before the SFT-shift interactions were\n   characterized. On SFT, `shift` is a real knob — try 1.0–5.0.\n\n3. **Custom `timesteps` overrides `shift`.** Passing both is\n   unnecessary; only one will be honored (the `timesteps` list, per\n   upstream `GenerationParams` doc). For continuous shift, supply\n   `timesteps` and omit `shift`.\n\n## 10. Sources\n\n- [`ACE-Step-1.5/docs/en/INFERENCE.md`](/Users/luis/Repos/ACE-Step-1.5/docs/en/INFERENCE.md)\n  — `GenerationParams.shift` formula\n  (`t = shift * t / (1 + (shift - 1) * t)`),\n  default `1.0`, range `1.0–5.0`, \"Recommended 3.0 for turbo models\".\n- [`ACE-Step-1.5/docs/en/Tutorial.md`](/Users/luis/Repos/ACE-Step-1.5/docs/en/Tutorial.md)\n  § DiT Models — turbo variant distillation configs (joint / shift1 /\n  shift3 / continuous), semantic-vs-detail description of the shift\n  parameter, \"draw outline first then fill details\" mental model.\n- [`ACE-Step-1.5/README.md`](/Users/luis/Repos/ACE-Step-1.5/README.md) § Model\n  Zoo (DiT Models / XL DiT Models) — checkpoint names and properties.\n- [`references/acestep-xl-models.md`](acestep-xl-models.md) — `shift=3.0`\n  pinned as XL Base default; `shift` does NOT control vocal/instrument\n  separation.\n- [`references/acestep-task-types.md`](acestep-task-types.md) —\n  `extract`/`lego`/`complete` all use `shift: 3.0` as the BASE-only\n  default.\n- [`references/acestep-generation.md`](acestep-generation.md) —\n  `shift: 3.0` documented default for base models; the\n  \"only effective for base models, not turbo models\" upstream note and\n  the xl-sft-is-not-turbo caveat.\n- `music-craft_ROADMAP.md` item **13i** — the original taxonomy\n  request, tier-to-shift mapping table, and verification criterion.\n- `music-craft-ENHANCEMENT-PLAN.md` — sequenced under item 12\n  (Turbo shift-schedule taxonomy).\n\n## 11. See also\n\n- [`acestep-generation.md`](acestep-generation.md) — base 3-step\n  workflow, full parameter table, quality tiers with ML-budget probe.\n- [`acestep-xl-models.md`](acestep-xl-models.md) — XL (4B) DiT\n  reference, including the `num_inference_steps=50` and `shift=3.0`\n  footguns.\n- [`acestep-task-types.md`](acestep-task-types.md) — the six\n  audio-conditioned task types; `extract`/`lego`/`complete` (BASE-only).\n- [`setup-and-preflight.md`](setup-and-preflight.md) — ML-budget probe,\n  env-var reference, model download consent flow.\n- [`quality-and-revision.md`](quality-and-revision.md) — listenability\n  rubric for \"is this output good enough to ship?\" decisions when\n  shift-sweeping.\n- [`prompt-formula.md`](prompt-formula.md) — production-sheet prompt\n  construction that pairs naturally with the right shift choice.\n\nFile v1.6.0:references/acestep-task-types.md\n\n# ACE-Step Task Types Reference\n\nPer-task reference for the five ACE-Step 1.5 **audio-conditioned** task types:\n`cover`, `repaint`, `extract`, `lego`, and `complete`. Use this when the\nrequest requires anything beyond plain text-to-music generation. Load\n[`acestep-generation.md`](acestep-generation.md) first for the base\n3-step workflow (submit, poll, collect), quality tiers, and the\n`task_type` table.\n\n## Audio-Conditioned Generation (umbrella category)\n\nEvery task below conditions on an input audio file. They differ in **what\nthe model is asked to produce** from that audio:\n\n| `task_type` | What it does | Output count | Model support |\n| --- | --- | --- | --- |\n| `cover` | Re-style a song while following its melody and structure | 1 | BASE + SFT + TURBO |\n| `repaint` | Regenerate only a time window, keep the rest of the audio | 1 | BASE + SFT + TURBO |\n| `extract` | Stem separation (vocals / drums / bass / etc.) | 1–2 | **BASE only** |\n| `lego` | Add a new instrument layer on top of existing audio | 2+ | **BASE only** |\n| `complete` | Fill out a partial track with mixed accompaniment (e.g. Vocal2BGM) | 1 | **BASE only** |\n\n> **BASE-only warning** — `extract`, `lego`, and `complete` are only\n> supported by `acestep-v15-base` and `acestep-v15-xl-base`. They\n> silently fail or refuse on `*turbo` and `*sft` checkpoints. The\n> BASE model is research-grade in this skill's stack as of 2026-07;\n> smoke-test on your hardware before promising production reliability.\n\n## Model support matrix (verified upstream)\n\n| DiT Model | Pre-Train | SFT | RL | CFG | Steps | Cover | Repaint | Extract | Lego | Complete |\n| --- | :---: | :---: | :--: | :---: | :---: | :---: | :---: | :---: | :---: | :---: |\n| `acestep-v15-base` | ✅ | ❌ | ❌ | ✅ | 50 | ✅ | ✅ | ✅ | ✅ | ✅ |\n| `acestep-v15-sft` | ✅ | ✅ | ❌ | ✅ | 50 | ✅ | ✅ | ❌ | ❌ | ❌ |\n| `acestep-v15-turbo` | ✅ | ✅ | ❌ | ❌ | 8 | ✅ | ✅ | ❌ | ❌ | ❌ |\n| `acestep-v15-xl-base` | ✅ | ❌ | ❌ | ✅ | 50 | ✅ | ✅ | ✅ | ✅ | ✅ |\n| `acestep-v15-xl-sft` | ✅ | ✅ | ❌ | ✅ | 50 | ✅ | ✅ | ❌ | ❌ | ❌ |\n| `acestep-v15-xl-turbo` | ✅ | ✅ | ❌ | ❌ | 8 | ✅ | ✅ | ❌ | ❌ | ❌ |\n\n> Source: `ACE-Step-1.5/README.md` § Model Zoo (DiT Models / XL DiT\n> Models) plus the base `GenerationParams.task_type` enum in\n> `acestep/inference.py`.\n\n## Switching to a BASE-only model\n\nBASE-only tasks fail with turbo/SFT. Before submitting an `extract`,\n`lego`, or `complete` request, switch the loaded DiT:\n\n```bash\n# Switch to BASE (download acestep-v15-base first if not on disk)\ncurl -s -X POST http://127.0.0.1:8001/v1/init \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"dit_model\": \"acestep-v15-base\", \"lm_model\": \"acestep-5Hz-lm-1.7B\"}'\n\n# Switch back to standard (turbo) after the task\ncurl -s -X POST http://127.0.0.1:8001/v1/init \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"dit_model\": \"acestep-v15-turbo\", \"lm_model\": \"acestep-5Hz-lm-1.7B\"}'\n```\n\n**XL-BASE path** (≥20 GB VRAM recommended):\n\n```bash\ncurl -s -X POST http://127.0.0.1:8001/v1/init \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"dit_model\": \"acestep-v15-xl-base\", \"lm_model\": \"acestep-5Hz-lm-1.7B\"}'\n```\n\n> Switching models reloads the DiT weights (~10–90 s cold start). Do\n> not batch BASE-only and turbo/SFT tasks in the same minute.\n\n## Uploading source audio (all tasks)\n\nThe API **rejects absolute file paths** in JSON (`{\"detail\":\"absolute audio\nfile paths are not allowed\"}`). Upload via multipart form-data:\n\n| Field name | When |\n| --- | --- |\n| `src_audio` | cover, repaint, extract, lego, complete — the audio the model edits / conditions on |\n| `reference_audio` / `ref_audio` | style-transfer reference (global timbre / mix feel, NOT melody) |\n\nSend all other parameters as form fields:\n\n```bash\ncurl -s -X POST http://127.0.0.1:8001/release_task \\\n  -F \"task_type=<TASK>\" \\\n  -F \"src_audio=@/path/to/source.wav\" \\\n  -F \"<other_field>=<value>\"\n```\n\n> **Path caveat:** the file must exist on the server's filesystem or be\n> uploaded through the multipart `src_audio` field. A path on the\n> client's machine is never accepted.\n\n## Polling and output collection (all tasks)\n\nSame 3-step pattern as text2music:\n\n1. Submit via `/release_task` (returns `task_id`).\n2. Poll `/query_result` with `{\"task_ids\": [\"...\"]}`. Treat empty `data`\n   as \"still running\" — see `acestep-generation.md` Polling caveat.\n3. Collect from `${ACE_STEP_PATH}/.cache/acestep/tmp/api_audio/` once\n   `status: 1` (success) appears. Cache-file detection (rather than\n   `/query_result`) is the reliable completion signal under load.\n\nUse [`scripts/wait_for_acestep.py`](../scripts/wait_for_acestep.py)\nwhen available — it reconciles polling with cache-file detection.\n\n## LM behavior per task type\n\n| `task_type` | LM (`thinking`) | Why |\n| --- | --- | --- |\n| `text2music` | runs if `thinking=true` | default flow |\n| `cover` | **skipped** | caption + lyrics come directly from the user; LM rewriting can fight the source |\n| `repaint` | **skipped** | same reason as cover |\n| `extract` | **skipped** | upstream note: LM-generated captions can cause DiT to reconstruct input instead of extracting stems |\n| `lego` | runs if `thinking=true` | uses LM planning for the new layer |\n| `complete` | runs if `thinking=true` | uses LM planning for the accompaniment |\n\n**Implication:** for `cover`/`repaint`/`extract`, the `caption` and\n`lyrics` fields you submit are used **verbatim**. Write them well.\n`thinking: true` is silently ignored.\n\n---\n\n## 1. Cover\n\n**Purpose:** Re-style a song while following its melody and structure.\nGenerate \"in the style of X\" renditions of an existing song. Local,\nmelody-aware alternative to cloud cover services.\n\n### Inputs\n\n- `src_audio` — the song you want to restyle (multipart upload, not a\n  path). Format: any of `wav`, `mp3`, `flac`, `opus`, `aac`.\n- `caption` — the style you want (genre + instruments + vocal character\n  - mood + production). Use the same detailed multi-dimensional format\n  as text2music (see `acestep-generation.md` Prompt format).\n- `lyrics` — optional; defaults to empty (instrumental cover) or what\n  the model infers.\n- `audio_cover_strength` — how strictly to follow the source melody\n  (see Parameters).\n- Metas (`bpm`, `key_scale`, `time_signature`, `vocal_language`) — set\n  explicitly for tighter control.\n\n### Outputs\n\n- **1 file** at `${ACE_STEP_PATH}/.cache/acestep/tmp/api_audio/`,\n  duration auto-locked to the source length (`audio_duration` is\n  ignored).\n\n### Use cases\n\n- Genre-jump covers (rock → synthwave, acoustic → orchestral).\n- A/B a melody in different keys or tempos without re-recording.\n- Local cover generation when the user wants a private / offline\n  workflow.\n\n### Parameters\n\n| Parameter | Type | Default | Notes |\n| --- | --- | --- | --- |\n| `task_type` | string | — | Must be `\"cover\"`. |\n| `src_audio` | file | required | Multipart upload. |\n| `caption` | string | required | Style description. Detailed, multi-dimensional. |\n| `lyrics` | string | empty | If set, model sings this against the source melody. |\n| `audio_cover_strength` | float | `1.0` | 0.0–1.0. Lower = bigger restyle. See table below. |\n| `bpm` | int | none | Recommended: read source BPM first and set it explicitly. |\n| `key_scale` | string | `\"\"` | E.g. `\"D major\"`, `\"A minor\"`. |\n| `time_signature` | string | `\"\"` | E.g. `\"4/4\"`. |\n| `vocal_language` | string | `\"unknown\"` | Set explicitly; LM is skipped. |\n| `audio_format` | string | `\"flac\"` | Lossless for cover work — lossy MP3 round-trip degrades melody. |\n\n**`audio_cover_strength` heuristic (verified in `acestep-generation.md`):**\n\n| Value | Behavior |\n| --- | --- |\n| `0.2`–`0.4` | Strong genre jump (rock → synthwave). Most free interpretation. |\n| `0.5`–`0.7` | Balanced — keeps the melody, allows noticeable style change. |\n| `0.8`–`0.9` | Subtle restyle. Closer to source timbre. |\n| `1.0` | Closest possible to source. |\n\n### Example command\n\n```bash\nTASK_ID=$(curl -s -X POST http://127.0.0.1:8001/release_task \\\n  -F \"task_type=cover\" \\\n  -F \"src_audio=@/Users/luis/Music mix/source/rock_track.wav\" \\\n  -F \"audio_cover_strength=0.35\" \\\n  -F \"prompt=dreamy 80s synthwave, warm analog synths, gated-reverb drums, arpeggiated bass, neon night-drive mood\" \\\n  -F \"bpm=129\" \\\n  -F \"key_scale=D major\" \\\n  -F \"audio_format=wav\" \\\n  | python3 -c \"import json,sys; print(json.load(sys.stdin).get('data',{}).get('task_id',''))\")\n\n# Poll until ready\ncurl -s -X POST http://127.0.0.1:8001/query_result \\\n  -H \"Content-Type: application/json\" \\\n  -d \"{\\\"task_ids\\\": [\\\"$TASK_ID\\\"]}\"\n```\n\n### Limitations\n\n- **Source-length cap.** A full ~5-minute cover hits the default\n  600 s server timeout on 12 GB-class hardware. Trim the source with\n  `ffmpeg -ss <start> -t 60 -i in.wav out.wav` for short experiments,\n  or raise `ACESTEP_GENERATION_TIMEOUT=3600`.\n- **Melody capture is approximate.** Sparse, mid/slow-tempo songs work\n  best; dense / fast / percussive sources tend to lose melodic detail.\n  Expect variation, not an exact pitch-perfect copy.\n- **LM is skipped.** Your caption and lyrics are used verbatim — a weak\n  caption degrades results more than on text2music.\n- **No style-transfer / mashup support.** For two-song mashup or\n  emotion-driven prompts, switch to `music-craft-minimax`.\n- **No cancel endpoint.** Lint the request before submitting; cancel\n  is not available mid-job.\n\n### Quality tips\n\n- **Always request lossless output** (`audio_format=wav` or `flac`)\n  when iterating on style — MP3 round-trip hides melody artifacts.\n- **Trim the source** to a section with clear melodic content for\n  faster, more reliable covers. Cover the full song only on the\n  final pass.\n- **Set explicit BPM / key.** LM is skipped, so the model cannot infer\n  from your prompt.\n- **Generate a small batch** (`batch_size: 2–4`) and pick the best —\n  cover quality varies more across seeds than text2music.\n- **Validate with `tests/analyzers/task_validator.py`:**\n\n  ```bash\n  python3 tests/analyzers/task_validator.py cover \\\n      /path/to/source.wav /path/to/cover_output.wav --json\n  ```\n\n  Confirms output duration matches source within ±0.5 s.\n\n### Routing gate\n\nUse ACE-Step cover only when the user explicitly accepts a local,\nslower, experimental workflow. Use `music-craft-minimax` for fast\ncloud cover, long sources, mashups, or emotion analysis.\n\n---\n\n## 2. Repaint\n\n**Purpose:** Regenerate only a time window of an existing audio file,\nkeep the rest untouched. Fix a bad section without throwing away the\nwhole track.\n\n### Inputs\n\n- `src_audio` — the audio you want to patch (multipart upload).\n- `repainting_start`, `repainting_end` — seconds, define the window\n  to regenerate. `repainting_end: -1` means \"until end of audio\".\n- `caption` — describes the new content for that window only.\n- `lyrics` — optional; if set, replaces lyrics in the window.\n- `repaint_mode` — `conservative` / `balanced` / `aggressive`.\n- `repaint_strength` — float 0–1 (balanced mode only); higher = stick\n  closer to source.\n\n### Outputs\n\n- **1 file** at the cache dir, **same total duration as the source**.\n  Only the `[start, end)` window is regenerated; everything outside\n  the window is preserved.\n\n### Use cases\n\n- Fix a mispronounced word in `[Chorus]` of an otherwise good track.\n- Replace a weak bridge with a stronger arrangement.\n- Add a section transition (e.g. change Verse → Pre-Chorus feel at\n  `30 s`).\n- \"Continue writing\" — extend the end of a track by repainting the\n  last 10–30 s.\n\n### Parameters\n\n| Parameter | Type | Default | Notes |\n| --- | --- | --- | --- |\n| `task_type` | string | — | Must be `\"repaint\"`. |\n| `src_audio` | file | required | Multipart upload. |\n| `repainting_start` | float | `0.0` | Seconds. |\n| `repainting_end` | float | `-1` | Seconds. `-1` = end of file. |\n| `caption` | string | required | Content for the repainted window. |\n| `lyrics` | string | empty | Lyrics inside the window only. |\n| `repaint_mode` | string | `\"balanced\"` | `conservative` (safer, sticks close to source) / `balanced` (default) / `aggressive` (more freedom). |\n| `repaint_strength` | float | `0.5` | 0.0–1.0; only used in `balanced` mode. Higher = closer to source. |\n| `chunk_mask_mode` | string | `\"auto\"` | `\"explicit\"` = hard 0/1 mask from range; `\"auto\"` = model decides per chunk. |\n| `repaint_latent_crossfade_frames` | int | `10` | Latent-level blend width at boundaries (~0.4 s). |\n| `repaint_wav_crossfade_sec` | float | `0.0` | Waveform splice crossfade in seconds; `0` = hard cut. |\n| `audio_format` | string | `\"flac\"` | Lossless strongly recommended — lossy boundaries are audible. |\n| `bpm` / `key_scale` / etc. | — | — | Set explicitly to anchor the new window. |\n\n### Example command\n\n```bash\nTASK_ID=$(curl -s -X POST http://127.0.0.1:8001/release_task \\\n  -F \"task_type=repaint\" \\\n  -F \"src_audio=@/Users/luis/Music mix/source/track_with_bad_chorus.wav\" \\\n  -F \"repainting_start=72.0\" \\\n  -F \"repainting_end=104.0\" \\\n  -F \"repaint_mode=balanced\" \\\n  -F \"repaint_strength=0.6\" \\\n  -F \"prompt=anthemic chorus, soaring female vocal, full band, big drums\" \\\n  -F \"lyrics=[Chorus]\\nWe rise together\\nInto the light\" \\\n  -F \"audio_format=wav\" \\\n  | python3 -c \"import json,sys; print(json.load(sys.stdin).get('data',{}).get('task_id',''))\")\n```\n\n### Limitations\n\n- **Window size.** Operates best on 3–90 s windows. Very short windows\n  (<3 s) may not give the model enough context; very long windows\n  (>90 s) start to drift from the source.\n- **Boundary artifacts.** Even with `repaint_wav_crossfade_sec: 0.2`,\n  edges between preserved and regenerated audio are audible if the\n  source and the new content disagree on tempo/key. Sample-rate and\n  bit-depth must match (default 48 kHz / float32).\n- **Same total duration as source.** The repaint does not extend or\n  shorten the track — only replaces content inside the window.\n- **LM is skipped.** Caption is used verbatim. A vague caption gives\n  a vague repaint.\n- **No cancel.** Repaints on long sources (>3 min) take several\n  minutes; verify the window before submitting.\n- **Single-worker queue.** A failed repaint blocks subsequent jobs\n  until the timeout fires.\n\n### Quality tips\n\n- **Pick the window from a structural analysis.** Use\n  `references/structure-tags.md` to identify `[Verse]` / `[Chorus]` /\n  `[Bridge]` boundaries, then target a single section.\n- **Start with `repaint_mode=balanced` + `repaint_strength=0.6`**.\n  Move to `aggressive` if the result is too similar to the source.\n- **Match metas to the source.** Set `bpm` and `key_scale` to the\n  detected values; mismatched keys make the boundary click.\n- **Keep the window ≤ 60 s** for the first attempt. Smaller windows\n  are faster, safer, and easier to validate.\n- **Validate with `tests/analyzers/task_validator.py`:**\n\n  ```bash\n  python3 tests/analyzers/task_validator.py repaint \\\n      /path/to/source.wav /path/to/repainted.wav --json\n  ```\n\n  Output duration must match source within ±0.5 s.\n\n---\n\n## 3. Extract (**BASE-only**)\n\n**Purpose:** Stem separation — extract a specific instrument or vocal\ntrack from a mixed audio file. Local alternative to Demucs/Spleeter,\nbut experimental.\n\n### Inputs\n\n- `src_audio` — the mixed audio (multipart upload).\n- `instruction` — must specify which track to extract. Auto-generated\n  from `task_type` if omitted. Common forms:\n  - `\"Extract the vocals track from the audio:\"`\n  - `\"Extract the drums track from the audio:\"`\n  - `\"Extract the bass track from the audio:\"`\n\n### Outputs\n\n- **1 file** (the extracted stem) at the cache dir, **same duration as\n  the source**. The model writes the requested stem; it does NOT also\n  write the residual (use `complete` or Demucs for residual\n  extraction).\n\n### Use cases\n\n- Quick vocals / drums / bass isolation from a finished mix.\n- Pre-process for `lego` (extract vocals → lego new drums on top).\n- Local stem analysis when Demucs is not installed or too slow.\n\n### Supported tracks (12 instrument families)\n\nFrom `acestep/constants.py` § `TRACK_NAMES`:\n\n`vocals`, `backing_vocals`, `drums`, `bass`, `guitar`, `keyboard`,\n`percussion`, `strings`, `synth`, `fx`, `brass`, `woodwinds`.\n\n### Parameters\n\n| Parameter | Type | Default | Notes |\n| --- | --- | --- | --- |\n| `task_type` | string | — | Must be `\"extract\"`. |\n| `src_audio` | file | required | Multipart upload. |\n| `instruction` | string | auto | `\"Extract the {TRACK_NAME} track from the audio:\"`. Pick one of the 12 names. |\n| `audio_format` | string | `\"flac\"` | Lossless recommended. |\n| `inference_steps` | int | `8` | BASE supports 32–64. Higher = cleaner separation but slower. |\n| `guidance_scale` | float | `7.0` | BASE only. |\n| `shift` | float | `3.0` | BASE only. Try `1.0` or `5.0` if defaults feel off. |\n| `thinking` | bool | true | **Skipped** for extract (per upstream). Do not expect CoT planning. |\n\n### Example command\n\n```bash\nTASK_ID=$(curl -s -X POST http://127.0.0.1:8001/release_task \\\n  -F \"task_type=extract\" \\\n  -F \"src_audio=@/Users/luis/Music mix/source/full_mix.wav\" \\\n  -F \"instruction=Extract the vocals track from the audio:\" \\\n  -F \"audio_format=wav\" \\\n  -F \"inference_steps=32\" \\\n  | python3 -c \"import json,sys; print(json.load(sys.stdin).get('data',{}).get('task_id',''))\")\n```\n\n### Limitations\n\n- **BASE-only.** Fails on `*turbo` and `*sft` checkpoints. Switch\n  models first (see \"Switching to a BASE-only model\" above).\n- **No residual output.** You get one stem per request. To get\n  \"vocals + accompaniment\", submit two extracts (one for vocals, one\n  for the rest via `instrumental`-style instruction) or use Demucs.\n- **LM is skipped.** Do not pass `caption`/`lyrics` — extract uses\n  `instruction` only. Upstream note: LM-generated captions can cause\n  DiT to reconstruct input audio instead of extracting stems.\n- **Lower quality than Demucs** in published benchmarks. For\n  production-quality stem separation, prefer `scripts/extract_stems.py`\n  (Demucs wrapper) which writes a `stems.json` with normalized\n  paths.\n- **Output can leak source audio** when the chosen stem does not\n  exist (e.g. asking for \"saxophone\" on a track with no sax). Treat\n  unexpected content as a failure signal, not a success.\n- **No cancel.** Long sources take minutes; check the duration\n  before submitting.\n\n### Quality tips\n\n- **Start with vocals.** Vocals separation is the most reliably\n  trained task; drums and bass are next. Fx / woodwinds are the\n  weakest.\n- **Match source sample rate.** Default 48 kHz; resample with ffmpeg\n  if your source is 44.1 kHz.\n- **For full stems**, use the Demucs helper path in\n  `acestep-generation.md` (`scripts/extract_stems.py`) and skip\n  ACE-Step extract entirely.\n- **Validate with `tests/analyzers/task_validator.py`:**\n\n  ```bash\n  python3 tests/analyzers/task_validator.py extract \\\n      /path/to/source.wav /path/to/vocals.wav /path/to/instrumental.wav --json\n  ```\n\n  Confirms both stems are within ±0.5 s of the source duration.\n\n### Routing gate\n\nACE-Step extract is **experimental**. For arranger workflows, prefer\nthe Demucs path documented in `acestep-generation.md`. Use ACE-Step\nextract only when Demucs is not available and the user accepts a\nresearch-grade result.\n\n---\n\n## 4. Lego (**BASE-only**)\n\n**Purpose:** Add a new instrument layer on top of existing audio. Feed\nin a backing track, ask for \"the guitar track based on the audio\ncontext\", and the model renders a matching guitar layer that fits\nrhythm, harmony, and timbre.\n\n### Inputs\n\n- `src_audio` — the existing audio (multipart upload). The \"context\"\n  the model layers against.\n- `instruction` — must specify which track to add. Forms:\n  - `\"Generate the vocals track based on the audio context:\"`\n  - `\"Generate the drums track based on the audio context:\"`\n  - `\"Generate the guitar track based on the audio context:\"`\n- `caption` — style description for the new layer.\n- `repainting_start` / `repainting_end` — optional. Defines where in\n  the source the new layer plays. Default: full length. If you set a\n  window, the layer only renders there.\n\n### Outputs\n\n- **2+ files** at the cache dir:\n  1. The **new layer** (e.g. `vocals.wav`) — only the added instrument.\n  2. (Optional / depending on mode) A **combined mix** of source + new\n     layer.\n\n  Exact file count is implementation-dependent — always check the\n  cache dir for new files, not just `/query_result`.\n\n### Use cases\n\n- Layer a guitar part on top of an existing drum + bass recording.\n- Add backing vocals to a lead-vocal-only take.\n- Build a multi-track composition iteratively (one instrument per\n  request).\n- Create a Stem2Track conversion: take a stem from `extract`, add a\n  new layer via `lego`.\n\n### Supported tracks (12)\n\nSame as `extract`: `vocals`, `backing_vocals`, `drums`, `bass`,\n`guitar`, `keyboard`, `percussion`, `strings`, `synth`, `fx`, `brass`,\n`woodwinds`.\n\n### Parameters\n\n| Parameter | Type | Default | Notes |\n| --- | --- | --- | --- |\n| `task_type` | string | — | Must be `\"lego\"`. |\n| `src_audio` | file | required | The audio to layer against. |\n| `instruction` | string | auto | `\"Generate the {TRACK_NAME} track based on the audio context:\"`. |\n| `caption` | string | empty | Style description for the new layer. |\n| `lyrics` | string | empty | If the new layer is vocals. |\n| `repainting_start` | float | `0.0` | Optional window start (seconds). |\n| `repainting_end` | float | `-1` | Optional window end; `-1` = end. |\n| `global_caption` | string | empty | Song-level caption for \"SFT-stems lego\" mode (when using SFT-stems). |\n| `inference_steps` | int | `8` | BASE supports 32–64. |\n| `guidance_scale` | float | `7.0` | BASE only. |\n| `shift` | float | `3.0` | BASE only. |\n| `thinking` | bool | true | **Runs** for lego (unlike cover/repaint/extract). LM helps plan the new layer. |\n\n### Example command\n\n```bash\nTASK_ID=$(curl -s -X POST http://127.0.0.1:8001/release_task \\\n  -F \"task_type=lego\" \\\n  -F \"src_audio=@/Users/luis/Music mix/source/drums_and_bass.wav\" \\\n  -F \"instruction=Generate the guitar track based on the audio context:\" \\\n  -F \"prompt=clean electric guitar, bluesy lead, mid-gain, expressive bends\" \\\n  -F \"audio_format=wav\" \\\n  -F \"inference_steps=32\" \\\n  | python3 -c \"import json,sys; print(json.load(sys.stdin).get('data',{}).get('task_id',''))\")\n```\n\n### Limitations\n\n- **BASE-only.** Fails on turbo/SFT. Switch first.\n- **LM is required for best results.** Unlike cover/repaint/extract,\n  lego uses LM planning — keep `thinking=true` and a non-empty\n  `caption`.\n- **New layer matches source rhythm/feel, not exact pitches.** Expect\n  a complementary part, not a transposed copy.\n- **File count varies.** Always inspect the cache dir after a lego\n  job; do not assume a specific output filename pattern.\n- **Cannot replace existing tracks.** Lego adds; it does not\n  re-record. To change a track that's already in the mix, use\n  `extract` first, then `lego` the replacement.\n- **No cancel.** Long sources (5+ min) take a long time; trim the\n  source to the section you actually need layered.\n- **Stereo image is approximate.** The new layer lands roughly in\n  the right place but does not preserve precise panning.\n\n### Quality tips\n\n- **Provide a strong context track.** A clean drums+bass mix gives\n  better lego results than a busy full mix — the model has clearer\n  rhythmic anchors.\n- **Match BPM / key explicitly.** Unlike cover/repaint, lego runs LM,\n  but explicit metas still anchor the new layer to the source's\n  tempo / tonality.\n- **Use `repainting_start` / `repainting_end` for sections.** A\n  guitar solo from 60 s to 90 s is much easier to validate than a\n  full-length layer.\n- **Iterate.** Lego is a \"first draft\" task — expect to layer\n  multiple times (vocals → guitar → keys) and combine with\n  `scripts/remix_stems.py`.\n- **Validate with `tests/analyzers/task_validator.py`:**\n\n  ```bash\n  python3 tests/analyzers/task_validator.py lego \\\n      /path/to/source.wav /path/to/original.wav /path/to/new_layer.wav --json\n  ```\n\n  Source duration and original duration must match; new layer\n  duration may differ if a window was used (warning, not error).\n\n---\n\n## 5. Complete (**BASE-only**)\n\n**Purpose:** Fill out a partial track with a full mixed accompaniment\n(\"Vocal2BGM\" pattern). Feed in a single vocal take; the model\ngenerates a complete backing track (drums + bass + keys + etc.) that\nmatches rhythm, harmony, and style.\n\n### Inputs\n\n- `src_audio` — the partial / a-cappella track (multipart upload).\n- `instruction` — must specify which tracks to add. Forms:\n  - `\"Complete the input track with drums, bass, guitar:\"`\n  - `\"Complete the input track with drums, bass, guitar, keyboard:\"`\n  - `\"Complete the input track with piano, strings:\"`\n- `caption` — style description for the accompaniment.\n\n### Outputs\n\n- **1 file** at the cache dir — a mix of the source (preserved) plus\n  the generated accompaniment. **Output duration is typically ≥ 1.5×\n  source duration** (the model often extends intros/outros).\n\n### Use cases\n\n- Turn a vocal-only phone recording into a demo with drums + bass +\n  keys.\n- Add a backing track to a melodic sketch (piano/vocal) for a fuller\n  arrangement.\n- Reverse-Vocal2BGM: feed a stripped instrumental and ask for vocal-\n  like top-line.\n\n### Parameters\n\n| Parameter | Type | Default | Notes |\n| --- | --- | --- | --- |\n| `task_type` | string | — | Must be `\"complete\"`. |\n| `src_audio` | file | required | Partial / a-cappella track. |\n| `instruction` | string | auto | `\"Complete the input track with {TRACK_CLASSES}:\"`. Comma-separated track list from the 12 supported families. |\n| `caption` | string | empty | Style description for the accompaniment. |\n| `lyrics` | string | empty | Optional; if the source is vocal, the model can align. |\n| `inference_steps` | int | `8` | BASE supports 32–64. |\n| `guidance_scale` | float | `7.0` | BASE only. |\n| `shift` | float | `3.0` | BASE only. |\n| `thinking` | bool | true | **Runs** for complete (LM plans the accompaniment arrangement). |\n\n### Example command\n\n```bash\nTASK_ID=$(curl -s -X POST http://127.0.0.1:8001/release_task \\\n  -F \"task_type=complete\" \\\n  -F \"src_audio=@/Users/luis/Music mix/source/vocal_only_take.wav\" \\\n  -F \"instruction=Complete the input track with drums, bass, guitar, keyboard:\" \\\n  -F \"prompt=indie folk, acoustic drums, upright bass, fingerstyle guitar, soft piano\" \\\n  -F \"lyrics=[Verse]\\nwalking through the morning light\\nnothing feels the same\" \\\n  -F \"audio_format=wav\" \\\n  -F \"inference_steps=32\" \\\n  | python3 -c \"import json,sys; print(json.load(sys.stdin).get('data',{}).get('task_id',''))\")\n```\n\n### Limitations\n\n- **BASE-only.** Fails on turbo/SFT.\n- **Extends the source.** The model often adds intro/outro pads to\n  smooth the entry and exit. If you need the exact source length,\n  trim the output with ffmpeg after the job completes.\n- **Single accompaniment shape.** The model picks one arrangement;\n  re-running with different seeds gives different arrangements.\n  There is no control over individual instrument layers here — use\n  `lego` after `complete` for that.\n- **LM is required for best results.** Keep `thinking=true` and\n  write a specific `caption`.\n- **Vocal-source alignment depends on quality.** A noisy / roomy\n  phone vocal gives weaker alignment than a clean studio take.\n- **No cancel.** Long sources take minutes; trim the source to the\n  section you actually need filled.\n- **Output style is global, not sectional.** The whole\n  accompaniment uses one style; per-section variation requires\n  multiple `repaint` passes after `complete`.\n\n### Quality tips\n\n- **Source vocal quality matters.** A clean, pitched vocal gives\n  better accompaniment than a noisy phone recording. Noise removal\n  with ffmpeg before the request is worth it.\n- **Be specific in `instruction`.** `\"drums, bass, guitar\"` gives a\n  standard trio; `\"drums, bass, guitar, keyboard, strings\"` gives\n  a fuller arrangement. Empty / vague instructions give vague\n  results.\n- **Set BPM / key from the source.** Detect first (use the librosa\n  pipeline or ACE-Step's audio-understanding mode) and pass\n  explicitly — the model anchors the accompaniment to those.\n- **Generate 2–4 variations** (`batch_size: 2–4`) and pick the best.\n  Arrangement quality varies more across seeds than across caption\n  changes.\n- **Validate with `tests/analyzers/task_validator.py`:**\n\n  ```bash\n  python3 tests/analyzers/task_validator.py complete \\\n      /path/to/partial.wav /path/to/completed.wav --json\n  ```\n\n  Output duration must be ≥ source duration; warns if < 1.5×.\n\n---\n\n## Cross-task operational notes\n\n### Server timeouts\n\nDefault `ACESTEP_GENERATION_TIMEOUT=600` (10 min). BASE tasks on long\nsources routinely exceed this. **Set `ACESTEP_GENERATION_TIMEOUT=3600`\n(1 h)** whenever you intend to run BASE-only tasks:\n\n```bash\nACESTEP_GENERATION_TIMEOUT=3600 \\\nACESTEP_CONFIG_PATH=acestep-v15-base \\\nACESTEP_LM_MODEL_PATH=acestep-5Hz-lm-1.7B \\\nuv run acestep-api --port 8001\n```\n\n### Memory budgets (24 GB M3 reference)\n\n| Task | Model | Approx peak RAM | Wall time per 60 s audio |\n| --- | --- | --- | --- |\n| cover | turbo | ~11 GB | ~3–5 min |\n| cover | xl-sft | ~25–30 GB | ~10–15 min |\n| repaint | turbo | ~11 GB | ~3–5 min |\n| repaint | xl-sft | ~25–30 GB | ~10–15 min |\n| extract | base | ~10 GB | ~3–6 min |\n| extract | xl-base | ~25–30 GB | ~15–25 min |\n| lego | base | ~10 GB | ~3–6 min |\n| lego | xl-base | ~25–30 GB | ~15–25 min |\n| complete | base | ~10 GB | ~3–6 min |\n| complete | xl-base | ~25–30 GB | ~15–25 min |\n\nSee `acestep-generation.md` § Quality Tiers for the full\nmemory-probe protocol.\n\n### Smoke-test before production\n\nPer item 13e of `music-craft_ROADMAP.md`: **do not promise production\nreliability for `extract`, `lego`, or `complete` until smoke tests\nland on this machine.** Minimum smoke test for each:\n\n1. Download the BASE model (or XL-BASE if RAM allows):\n\n   ```bash\n   uv run acestep-download --model acestep-v15-base\n   ```\n\n2. Switch the loaded DiT:\n\n   ```bash\n   curl -s -X POST http://127.0.0.1:8001/v1/init \\\n     -H \"Content-Type: application/json\" \\\n     -d '{\"dit_model\": \"acestep-v15-base\"}'\n   ```\n\n3. Run a 30-second task (extract a vocal, lego a guitar on a 30 s\n   backing, complete a 30 s vocal) and validate with\n   `tests/analyzers/task_validator.py`.\n4. Only after that passes on your hardware, accept user requests.\n\n### Output naming and caching\n\nAll outputs land under\n`${ACE_STEP_PATH}/.cache/acestep/tmp/api_audio/` with auto-generated\nfilenames. There is no per-task subdirectory — sort by `task_id` and\ncollection time:\n\n```bash\n# After a task completes\nfind \"${ACE_STEP_PATH:-$HOME/ACE-Step-1.5}/.cache/acestep/tmp/api_audio\" \\\n  -type f -newer /tmp/last_marker -name '*.wav' -o -name '*.flac' -o -name '*.mp3'\n```\n\nCache accumulation is the same caveat as text2music — see\n`acestep-generation.md` Cache caveat. Review before deleting.\n\n### Validation helper\n\n`tests/analyzers/task_validator.py` accepts task type + file paths\nand validates output shape per task:\n\n| Task | Required files | Checks |\n| --- | --- | --- |\n| `cover` | 2 (source, cover) | output duration ≈ source |\n| `repaint` | 2 (source, repainted) | output duration ≈ source |\n| `extract` | 3 (source, vocal, instrumental) | both stems ≈ source duration |\n| `lego` | 3 (source, original, new_layer) | original ≈ source; new_layer may be shorter |\n| `complete` | 2 (partial, completed) | completed ≥ partial; warns if < 1.5× |\n\n```bash\npython3 tests/analyzers/task_validator.py <task> <files...> --json\n```\n\nExits 0 = valid, 1 = invalid, 2 = error (missing deps / wrong arg count).\nUse `--json` for machine-readable output; omit for human summary.\n\n### Common errors\n\n| Symptom | Cause | Fix |\n| --- | --- | --- |\n| `task_type` rejected | wrong task for current DiT | switch to BASE for extract/lego/complete |\n| Output is a reconstruction of input | extract with bad `instruction` or LM-coaxed caption | drop `caption`/`lyrics`; verify `instruction` is exact |\n| Window boundary click | repaint with mismatched key/tempo | match source metas; raise `repaint_wav_crossfade_sec` |\n| Server timeout fired mid-job | default 600 s on long source | set `ACESTEP_GENERATION_TIMEOUT=3600` |\n| Output duration wrong | turbo steps used with BASE | BASE tasks need `inference_steps` ≥ 32 |\n| Empty `/query_result` while running | known server quirk | use cache-file detection, not just `/query_result` |\n| `absolute audio file paths are not allowed` | JSON body with file path | use multipart `-F \"src_audio=@/path\"` |\n\n### Routing decision tree\n\nWhen the request involves source audio:\n\n| Request shape | Recommended path |\n| --- | --- |\n| Re-style a song while keeping the melody | ACE-Step `cover` (this skill) |\n| Fix a bad section of an existing track | ACE-Step `repaint` (this skill) |\n| Two-song mashup or emotion-driven style transfer | `music-craft-minimax` (cloud) |\n| Stem separation (vocals / drums / bass) | Demucs via `scripts/extract_stems.py` (faster, higher quality) — ACE-Step `extract` only if Demucs unavailable |\n| Add an instrument layer to a backing track | ACE-Step `lego` (this skill) |\n| Add full backing to a vocal-only take | ACE-Step `complete` (this skill) — Vocal2BGM |\n| Generate from a prompt with no source audio | text2music (see `acestep-generation.md`) |\n\n---\n\n## See also\n\n- [`acestep-generation.md`](acestep-generation.md) — base 3-step workflow,\n  quality tiers, full prompt/parameter table, MP3/FLAC cache details.\n- [`setup-and-preflight.md`](setup-and-preflight.md) — dependency consent,\n  platform detection, ML budget probe.\n- [`wait-and-collect.md`](wait-and-collect.md) — M1 → wait → collect → M2\n  sequencing pattern.\n- [`local-ace-step-curl-template.md`](local-ace-step-curl-template.md) —\n  JSON-safe `/release_task` template.\n- [`structure-tags.md`](structure-tags.md) — section tag vocabulary for\n  picking repaint windows.\n- [`../../tests/analyzers/task_validator.py`](../../tests/analyzers/task_validator.py)\n  — output-shape validator for all five task types.\n- ACE-Step upstream:\n  [`README.md`](../../../../ACE-Step-1.5/README.md) § Model Zoo,\n  [`docs/en/INFERENCE.md`](../../../../ACE-Step-1.5/docs/en/INFERENCE.md)\n  § Task Types,\n  [`docs/en/Tutorial.md`](../../../../ACE-Step-1.5/docs/en/Tutorial.md)\n  § \"Base Model Advanced Audio Control Tasks\".\n\nFile v1.6.0:references/acestep-xl-models.md\n\n# ACE-Step XL (4B) DiT Models\n\nReference for the ACE-Step 1.5 XL (4B-parameter) DiT family: how the two\ndeployment shapes — **`xl-base`** and **`xl-mixed`** — differ, what\nhardware they need, the **must-pass parameters** that silently degrade\nquality if forgotten, and a decision tree for picking XL vs. the standard\n2B tier. Load this when a request mentions the XL DiT, the 4B DiT,\n`acestep-v15-xl-*` checkpoints, or \"best quality\" generation.\n\n> **Terminology note.** Elsewhere in this skill (`acestep-generation.md`),\n> **`xl-mixed`** refers to the *4B DiT + 1.7B LM* combo (the\n> `acestep-v15-xl-sft` + `acestep-5Hz-lm-1.7B` stack). This document uses\n> the stricter XL definition: **`xl-mixed` = 4B DiT + 4B LM**\n> (`acestep-v15-xl-base` + `acestep-5Hz-lm-4B`, ≈22 GB peak) — i.e. the\n> \"best\" tier from the existing tier table. The two meanings are two\n> different points on the same DiT + LM curve and the footguns below\n> apply to both.\n\n## TL;DR\n\n- **XL DiT = ~4B parameters, ~9 GB bf16 / ~11 GB on disk for the DiT\n  alone** (vs. ~4.7 GB for the 2B DiT). All HF checkpoints:\n  [`acestep-v15-xl-base`](https://huggingface.co/ACE-Step/acestep-v15-xl-base),\n  [`acestep-v15-xl-sft`](https://huggingface.co/ACE-Step/acestep-v15-xl-sft),\n  [`acestep-v15-xl-turbo`](https://huggingface.co/ACE-Step/acestep-v15-xl-turbo).\n- **Two deployment shapes**:\n  - **`xl-base`** — XL DiT only, BASE-only training. Required for\n    `extract` / `lego` / `complete` task types (the BASE-only family).\n  - **`xl-mixed`** — XL DiT + 4B LM. Heavy memory and slow on consumer\n    hardware, but the highest-quality path on a 32 GB+ machine.\n- **Hardware floor**: ≥12 GB VRAM (with CPU offload + quantization),\n  ≥20 GB VRAM recommended; **24 GB M3 is the practical reference**\n  (Luis's machine).\n- **Two critical footguns**, both silently degrade output if missed:\n  1. **`num_inference_steps=50` MUST be passed \n\nArchive v1.5.1: 28 files, 117153 bytes\n\nFiles: README.md (4331b), references/acestep-generation.md (30458b), references/changelog.md (1410b), references/error-handling.md (26912b), references/examples.md (10077b), references/free-tool-inputs.md (16582b), references/input-workflows.md (17275b), references/local-ace-step-curl-template.md (2486b), references/lyrics-cleanup.md (2590b), references/other-backends.md (9817b), references/prompt-formula.md (13390b), references/quality-and-revision.md (6279b), references/request-intake.md (8792b), references/setup-and-preflight.md (20709b), references/structure-tags.md (14374b), references/style-categories.md (6418b), references/user-preference-flow.md (7172b), references/wait-and-collect.md (3136b), references/windows-wsl-setup.md (11456b), scripts/extract_stems.py (4873b), scripts/lint_lyrics.py (4425b), scripts/remix_stems.py (2417b), scripts/smoke_test.py (5107b), scripts/verify_lyrics_alignment.py (3768b), scripts/wait_for_acestep.py (4564b), skill-card.md (2798b), SKILL.md (33314b), _meta.json (130b)\n\nArchive v1.5.0: 28 files, 115624 bytes\n\nFiles: README.md (3530b), references/acestep-generation.md (30440b), references/changelog.md (1033b), references/error-handling.md (26912b), references/examples.md (10077b), references/free-tool-inputs.md (16582b), references/input-workflows.md (17275b), references/local-ace-step-curl-template.md (2486b), references/lyrics-cleanup.md (2590b), references/other-backends.md (9270b), references/prompt-formula.md (13390b), references/quality-and-revision.md (6279b), references/request-intake.md (8792b), references/setup-and-preflight.md (20650b), references/structure-tags.md (14374b), references/style-categories.md (6418b), references/user-preference-flow.md (7172b), references/wait-and-collect.md (3136b), references/windows-wsl-setup.md (11456b), scripts/extract_stems.py (4873b), scripts/lint_lyrics.py (4425b), scripts/remix_stems.py (2417b), scripts/smoke_test.py (5107b), scripts/verify_lyrics_alignment.py (3768b), scripts/wait_for_acestep.py (4564b), skill-card.md (2972b), SKILL.md (31622b), _meta.json (130b)\n\nArchive v1.4.1: 27 files, 117697 bytes\n\nFiles: README.md (3505b), references/acestep-generation.md (30440b), references/error-handling.md (26912b), references/examples.md (10077b), references/free-tool-inputs.md (19922b), references/input-workflows.md (19843b), references/local-ace-step-curl-template.md (2486b), references/lyrics-cleanup.md (2590b), references/other-backends.md (9270b), references/prompt-formula.md (13390b), references/quality-and-revision.md (6868b), references/request-intake.md (8792b), references/setup-and-preflight.md (20650b), references/structure-tags.md (14374b), references/style-categories.md (6418b), references/user-preference-flow.md (7172b), references/wait-and-collect.md (3136b), references/windows-wsl-setup.md (11456b), scripts/extract_stems.py (4873b), scripts/lint_lyrics.py (4425b), scripts/remix_stems.py (2417b), scripts/smoke_test.py (5107b), scripts/verify_lyrics_alignment.py (3768b), scripts/wait_for_acestep.py (4564b), skill-card.md (3250b), SKILL.md (31919b), _meta.json (130b)\n\nArchive v1.4.0: 25 files, 113330 bytes\n\nFiles: README.md (3475b), references/acestep-generation.md (30440b), references/error-handling.md (26912b), references/examples.md (10077b), references/free-tool-inputs.md (19922b), references/input-workflows.md (19843b), references/local-ace-step-curl-template.md (2486b), references/lyrics-cleanup.md (2590b), references/other-backends.md (9270b), references/prompt-formula.md (13390b), references/quality-and-revision.md (6868b), references/request-intake.md (8792b), references/setup-and-preflight.md (20650b), references/structure-tags.md (13762b), references/style-categories.md (6418b), references/user-preference-flow.md (7172b), references/wait-and-collect.md (3136b), references/windows-wsl-setup.md (11456b), scripts/extract_stems.py (4873b), scripts/remix_stems.py (2417b), scripts/smoke_test.py (3953b), scripts/wait_for_acestep.py (4564b), skill-card.md (3291b), SKILL.md (30956b), _meta.json (130b)\n\nArchive v1.3.0: 21 files, 104853 bytes\n\nFiles: README.md (3475b), references/acestep-generation.md (29576b), references/error-handling.md (26600b), references/examples.md (10077b), references/free-tool-inputs.md (19588b), references/input-workflows.md (19038b), references/local-ace-step-curl-template.md (2486b), references/lyrics-cleanup.md (2590b), references/other-backends.md (9270b), references/prompt-formula.md (13390b), references/quality-and-revision.md (6868b), references/request-intake.md (8770b), references/setup-and-preflight.md (20517b), references/structure-tags.md (13568b), references/style-categories.md (6418b), references/user-preference-flow.md (6936b), references/wait-and-collect.md (2653b), references/windows-wsl-setup.md (11557b), skill-card.md (2841b), SKILL.md (29323b), _meta.json (130b)\n\nArchive v1.1.0: 18 files, 98415 bytes\n\nFiles: README.md (3122b), references/acestep-generation.md (27936b), references/error-handling.md (26600b), references/examples.md (10077b), references/free-tool-inputs.md (19588b), references/input-workflows.md (18476b), references/other-backends.md (9270b), references/prompt-formula.md (13390b), references/quality-and-revision.md (6868b), references/request-intake.md (8400b), references/setup-and-preflight.md (20517b), references/structure-tags.md (13244b), references/style-categories.md (6418b), references/user-preference-flow.md (6936b), references/windows-wsl-setup.md (11557b), skill-card.md (3136b), SKILL.md (27420b), _meta.json (130b)\n\nArchive v1.0.1: 13 files, 89535 bytes\n\nFiles: README.md (2380b), references/error-handling.md (26153b), references/examples.md (10077b), references/free-tool-inputs.md (17425b), references/input-workflows.md (16225b), references/prompt-formula.md (12266b), references/structure-tags.md (12588b), references/style-categories.md (6418b), references/user-preference-flow.md (6556b), references/windows-wsl-setup.md (11557b), skill-card.md (3111b), SKILL.md (92702b), _meta.json (130b)\n\nArchive v1.0.0: 13 files, 89748 bytes\n\nFiles: README.md (2392b), references/error-handling.md (26153b), references/examples.md (10077b), references/free-tool-inputs.md (17425b), references/input-workflows.md (16225b), references/prompt-formula.md (12266b), references/structure-tags.md (12588b), references/style-categories.md (6418b), references/user-preference-flow.md (6556b), references/windows-wsl-setup.md (11557b), skill-card.md (3636b), SKILL.md (92714b), _meta.json (130b)","readmeExcerpt":"Skill: music-craft Owner: luischarro Summary: Generate songs, instrumentals, or lyrics-driven tracks through a structured OpenClaw-native workflow with anti-sparse prompt engineering and quality verification. Provider-agnostic — works with any music backend the runtime exposes (ACE-Step, MusicGen, Stable Audio, mmx). Tags: latest:1.6.0 Version history: v1.6.0 | 2026-07-30T11:36:34.536Z | user v1.6.0: discoverability ","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"git clone https://github.com/ace-step/ACE-Step-1.5.git \"${ACE_STEP_PATH}\"\ncd \"${ACE_STEP_PATH}\" && uv sync\nuv run acestep-api --port 8001   # or: ./start_api_server_macos.sh"},{"language":"bash","snippet":"curl -s -X POST http://127.0.0.1:8001/query_result \\\n  -H \"Content-Type: application/json\" \\\n  -d \"{\\\"task_ids\\\": [\\\"$TASK_ID\\\"]}\""},{"language":"bash","snippet":"# 1. Submit task\nTASK_ID=$(curl -s -X POST http://127.0.0.1:8001/release_task \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"prompt\": \"<detailed caption, e.g.: dreamy 80s synthwave, warm analog synths, gated-reverb drums, arpeggiated bass, neon night-drive mood>\",\n    \"lyrics\": \"[Verse]\\n<lyrics here>\\n\\n[Chorus]\\n<lyrics here>\",\n    \"audio_duration\": 210,\n    \"bpm\": 96,\n    \"key_scale\": \"D major\",\n    \"time_signature\": \"4/4\",\n    \"vocal_language\": \"en\",\n    \"thinking\": true,\n    \"inference_steps\": 8,\n    \"guidance_scale\": 7.0\n  }' | python3 -c \"import json,sys; print(json.load(sys.stdin).get('data',{}).get('task_id',''))\")\n\n# 2. Poll for completion. Treat empty `data` as pending, not failed.\n# Wait, then check:\ncurl -s -X POST http://127.0.0.1:8001/query_result \\\n  -H \"Content-Type: application/json\" \\\n  -d \"{\\\"task_ids\\\": [\\\"$TASK_ID\\\"]}\"\n\n# 3. Copy audio from cache dir when done\n# Files saved to: ${ACE_STEP_PATH}/.cache/acestep/tmp/api_audio/"},{"language":"bash","snippet":"find \"${ACE_STEP_PATH:-$HOME/ACE-Step-1.5}/.cache/acestep/tmp/api_audio\" \\\n  -type f -name '*.mp3' -mtime +7 -print"},{"language":"bash","snippet":"TASK_ID=$(curl -s -X POST http://127.0.0.1:8001/release_task \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"prompt\": \"<your detailed multi-dimensional caption here>\",\n    \"lyrics\": \"[Verse 1]\\n<line>\\n<line>\\n\\n[Pre-Chorus]\\n<line>\\n<line>\\n\\n[Chorus]\\n<line>\\n<line>\\n\\n[Verse 2]\\n<line>\\n<line>\\n\\n[Bridge]\\n<line>\\n<line>\\n\\n[Outro]\\n<line>\\n<line>\\n\",\n    \"audio_duration\": 210,\n    \"bpm\": 96,\n    \"key_scale\": \"D major\",\n    \"time_signature\": \"4/4\",\n    \"vocal_language\": \"en\",\n    \"thinking\": true,\n    \"inference_steps\": 8,\n    \"guidance_scale\": 7.0\n  }' | python3 -c \"import json,sys; print(json.load(sys.stdin).get('data',{}).get('task_id',''))\")"},{"language":"text","snippet":"# Good (detailed, multi-dimensional — matches ACE-Step's own examples)\n\"A groovy funk track with slap bass, tight horn stabs, rhythmic guitar scratching, a charismatic male lead with call-and-response backing vocals, and an irresistible pocket groove\"\n\"Dreamy 80s synthwave: warm analog synths, gated-reverb drums, arpeggiated bassline, shimmering pads, nostalgic neon night-drive mood\"\n\n# Also fine (short tag; LM expands it with thinking=true)\n\"dreamy synthwave, 80s retro, atmospheric pads\"\n\n# Avoid: contradictory styles stacked in one static caption (express as evolution instead)\n\"classical chamber strings AND crushing hardcore metal AND lo-fi hip-hop, all at once\""}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: music-craft\nversion: 1.6.0\ndescription: Generate songs, instrumentals, or lyrics-driven tracks through a structured OpenClaw-native workflow with anti-sparse prompt engineering and quality verification. Provider-agnostic — works with any music backend the runtime exposes (ACE-Step, MusicGen, Stable Audio, mmx).\nmetadata: {\"openclaw\":{\"requires\":{\"anyBins\":[\"python3\",\"python\"]},\"emoji\":\"\\ud83c\\udfb5\",\"homepage\":\"https://github.com/LuisCharro/skills/tree/main/publish/music-craft\",\"envVars\":[{\"name\":\"MUSIC_PROVIDER_API_KEY\",\"required\":false,\"description\":\"Generic API key for any music provider.\"},{\"name\":\"STABILITY_API_KEY\",\"required\":false,\"description\":\"Stability AI API key. Only needed if using Stable Audio as backend.\"}]}}\n---\n# Music Craft\n\n## What is Music Craft?\n\nMusic Craft is the **provider-agnostic entry point** for music generation in OpenClaw. It guides the model end-to-end — from intake, through a structured production-sheet prompt, to quality verification and delivery — so you get songs, instrumentals, or lyrics-driven tracks that sound complete (no sparse dropouts, no silent sections), in your chosen length, with vocals that match your lyrics.\n\n**It picks the right backend for the job.** Local ACE-Step for exact-duration vocal tracks with lyrics. MusicGen for local instrumental experiments. Stable Audio for production-friendly instrumentals. The `mmx` CLI for fast cloud generation. You pick the outcome; the skill routes.\n\n## Key Capabilities\n\n- **Production-sheet prompt engineering** — every prompt is a structured brief with genre, mood, BPM, key, instruments, structure, vocals, and an avoid list.\n- **Anti-sparse guards by default** — instruments stay playing, no a cappella dropouts, no silent mid-sections.\n- **Exact-duration vocal tracks** — local ACE-Step returns requested length to the millisecond (verified 18/18 local jobs).\n- **Structure-tagged lyrics** — canonical `[Verse]`, `[Chorus]`, `[Break]`, etc., so the model sings the song you wrote.\n- **Pre-generation linting** — duration density, tag whitelist, prompt/flag conflicts caught before you spend a generation.\n- **Quality verification on every output** — duration, loudness, file size, audible completeness, lyrics alignment.\n- **Provider-agnostic** — local ACE-Step, MusicGen, Stable Audio, or the MiniMax cloud CLI; same workflow, same verification.\n\n## Why use this skill?\n\nMost music models treat \"sparse\" as \"remove everything\" or \"3 minutes\" as \"however long the model wants\". Music Craft encodes the rules once, so every generation runs through the same loop with the same anti-sparse guard, the same prompt validation, the same post-generation verification. You get coherent, full, audibly correct songs — not first-draft lottery tickets.\n\n## Quick Start\n\n1. Say what you want — e.g. _\"Make a sad love song in Spanish, ~3 minutes.\"_\n2. The skill auto-detects language, genre, mood, duration, and asks only the 1-3 things that aren't recoverable from your message.\n3. It buil"},{"path":"README.md","content":"# Music Craft\n\nGenerate songs, instrumentals, and lyrics-driven tracks through a disciplined OpenClaw-native workflow.\n\nCurrent release: v1.6.0.\n\nThis skill is **provider-agnostic**. It works with any music backend the OpenClaw runtime exposes via the `music_generate` tool — no special CLI, API, or library required.\n\n## Data and consent\n\nDepending on the chosen backend, prompts, lyrics, reference URLs, or generated/derived music instructions may be sent to a cloud provider. Local backends may download models and write temporary/generated audio files on the user's machine. Ask before installing/downloading large dependencies, uploading user-owned media, or overwriting existing outputs.\n\n## Licensing and commercial use\n\nClawHub publishes this skill bundle under MIT-0, so the skill instructions\nand bundled helper code may be used, modified, and redistributed commercially\nwithout attribution. MIT-0 does not grant rights to third-party models,\nsoftware, source material, or generated music. The operator must accept the\nselected backend's current terms on their own computer and use their own\nprovider account. ACE-Step 1.5\nis currently documented by its project as MIT/commercial-ready. MusicGen's\ncode is MIT, but its model weights are CC-BY-NC 4.0 and must not be used for\ncommercial output. For any other backend, identify and verify its model,\naccount tier, and output terms before commercial use. Source audio, lyrics,\nsamples, and voices must also be owned or properly licensed.\n\n## Platform support\n\nThis base workflow is effectively OS-neutral: it should work on macOS, Linux, and Windows as long as the active OpenClaw runtime exposes the `music_generate` tool. Platform differences only matter if the runtime provider itself needs local setup.\n\n## What it does\n\n- Translates your request into a production-sheet prompt with anti-sparse guards\n- Structures your lyrics (or auto-generated lyrics) with whitelisted section tags\n- Calls the runtime's `music_generate` tool\n- Verifies duration, loudness, structure, lyrics alignment, and audible quality before delivery\n- Documents prompt-length, lyrics-transcription, direct ACE-Step submission, wait-and-collect, and post-generation finalization safeguards\n\n## When to use\n\nUse this skill for any music generation task that does not require:\n\n- cover or style transfer from a reference audio file\n- emotion analysis or two-song mashup\n- separate `--avoid`, `--bpm`, `--key`, or `--structure` flags\n\nFor those, see [`music-craft-minimax`](../music-craft-minimax/) (requires MiniMax Token Plan). Audio input must be a local file path. URLs are not accepted in v1.5.0+; if you want to fetch audio by title from the internet, use the private `music-source-fetch` skill first (not published on ClawHub).\n\nIf the request is a standard song, instrumental, jingle, or lyrics-driven track, stay here and infer defaults first.\n\nFor exact-duration vocal tracks, prefer the local ACE-Step path documented in\n[`references/acestep-generation.md`]("},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7fkh362pnj2zckq8pcxfaxr9821zv8\",\n  \"slug\": \"music-craft\",\n  \"version\": \"1.6.0\",\n  \"publishedAt\": 1785411394536\n}"},{"path":"references/acestep-generation.md","content":"# ACE-Step Generation\n\nComplete ACE-Step 1.5 operating guide: API workflow (submit, poll, copy),\nfull-song generation, quality tiers with memory-safe selection, and\naudio-conditioned generation (cover, repaint, reference audio). Load this\nwhen the selected backend is ACE-Step.\n\n## ACE-Step 1.5 (local — free, vocals + lyrics, best local quality)\n\n**Best for:** local generation with real vocals, separate lyrics, song structure, up to 10 minutes (600s). No API key, no quota. Runs natively on Apple Silicon via MLX.\n\n**Verified routing note (2026-06-12):** ACE-Step is the exact-duration route.\nIn a 9-song field run, local M1+M2 generations returned the requested\n`audio_duration` exactly for 18/18 jobs. MiniMax cloud was faster but returned\n57-135% of requested duration for the paired cloud jobs.\n\n**Prerequisites:** REST API must be running on `http://127.0.0.1:8001`. Install with:\n```bash\ngit clone https://github.com/ace-step/ACE-Step-1.5.git \"${ACE_STEP_PATH}\"\ncd \"${ACE_STEP_PATH}\" && uv sync\nuv run acestep-api --port 8001   # or: ./start_api_server_macos.sh\n```\n\n(See [`setup-and-preflight.md`](setup-and-preflight.md) for how `ACE_STEP_PATH` is determined.)\n\n**Generation (3-step async):**\n\n```bash\n# 1. Submit task\nTASK_ID=$(curl -s -X POST http://127.0.0.1:8001/release_task \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"prompt\": \"<detailed caption, e.g.: dreamy 80s synthwave, warm analog synths, gated-reverb drums, arpeggiated bass, neon night-drive mood>\",\n    \"lyrics\": \"[Verse]\\n<lyrics here>\\n\\n[Chorus]\\n<lyrics here>\",\n    \"audio_duration\": 210,\n    \"bpm\": 96,\n    \"key_scale\": \"D major\",\n    \"time_signature\": \"4/4\",\n    \"vocal_language\": \"en\",\n    \"thinking\": true,\n    \"inference_steps\": 8,\n    \"guidance_scale\": 7.0\n  }' | python3 -c \"import json,sys; print(json.load(sys.stdin).get('data',{}).get('task_id',''))\")\n\n# 2. Poll for completion. Treat empty `data` as pending, not failed.\n# Wait, then check:\ncurl -s -X POST http://127.0.0.1:8001/query_result \\\n  -H \"Content-Type: application/json\" \\\n  -d \"{\\\"task_ids\\\": [\\\"$TASK_ID\\\"]}\"\n\n# 3. Copy audio from cache dir when done\n# Files saved to: ${ACE_STEP_PATH}/.cache/acestep/tmp/api_audio/\n```\n\n**Polling caveat:** The `/query_result` endpoint may return `{\"data\": [], \"code\": 200}` even while the task is actively running. This is a known server-side quirk. Don't treat empty data as \"task failed\" — instead, check for new MP3 files in the cache directory, or look at the server log (`/tmp/acestep-api.log`) for actual progress markers (e.g. `MLX DiT diffusion: 24/50`). If available, use `scripts/wait_for_acestep.py` because it reconciles `/query_result` with cache-file detection.\n\n**Stats caveat:** `GET /v1/stats` exposes top-level state such as queued,\nrunning, succeeded, failed, `queue_size`, and `avg_job_seconds`. It does not\nshow current sub-stage progress. For long jobs, use `/tmp/acestep-api.log` as\nthe source of truth for LM, DiT, CFG, and VAE progress.\n\n**Cache caveat:** generated audio acc"},{"path":"references/acestep-shift-schedule.md","content":"# ACE-Step Shift Schedule Taxonomy\n\nReference for the ACE-Step 1.5 **shift** parameter: what it controls in the\nflow-matching diffusion process, how the three documented schedules\n(`shift=1.0`, `shift=3.0`, and **continuous shift**) map to each model tier\n(turbo, base, sft, xl-base, xl-mixed), and how to pick the right shift for\nthe request at hand. Load this when a request mentions `shift`,\ntimestep-shifting, or any of the four turbo variants (`turbo`,\n`turbo-shift1`, `turbo-shift3`, `turbo-continuous`).\n\n> **Status:** docs-only. Backed by `music-craft_ROADMAP.md` item **13i** and\n> verified upstream against `ACE-Step-1.5/docs/en/INFERENCE.md` +\n> `Tutorial.md`. No runtime changes; if the upstream semantics shift\n> (e.g. a new default), update § 1 first.\n\n## TL;DR\n\n- **`shift` is a timestep-reshaping factor in flow-matching diffusion.**\n  When `shift != 1.0`, the scheduler applies\n  `t' = shift * t / (1 + (shift - 1) * t)` to every timestep `t` before\n  the step. Higher `shift` = the early (high-noise, structure-defining)\n  timesteps get **more compute**, low-noise detail timesteps get **less**.\n- **Three documented schedules:**\n  - `shift=1.0` — **default in the upstream `GenerationParams`**. Flat\n    schedule; the original cosine/linear timesteps are untouched. Pairs\n    with turbo's few-step (8) inference.\n  - `shift=3.0` — **documented default for base / sft / xl-base / xl-sft\n    checkpoints** (per upstream Tutorial + INFERENCE.md). Steep schedule;\n    more effort on structure, less on micro-detail. The \"best quality\"\n    default in upstream docs.\n  - **continuous shift** — a time-varying schedule where the shift\n    factor changes across timesteps instead of staying constant.\n    Available in the `turbo-continuous` checkpoint (experimental) and\n    as a research lever on custom flow-matching pipelines.\n- **Tier-to-shift mapping:**\n\n  | Tier | Default shift | Notes |\n  | --- | --- | --- |\n  | `turbo` (2B, 8 steps) | `1.0` (joint-distilled 1/2/3) | Most flexible; works at any shift in 1.0–5.0 |\n  | `turbo-shift1` | `1.0` | Distilled only at shift=1; richer details, weaker semantics |\n  | `turbo-shift3` | `3.0` | Distilled only at shift=3; clearer, drier, less orchestration |\n  | `turbo-continuous` | continuous (1–5) | Experimental; not thoroughly tested upstream |\n  | `base` / `xl-base` | `3.0` | Recommended; try `1.0` or `5.0` if defaults feel off |\n  | `sft` / `xl-sft` | `3.0` (experiment 1.0–5.0) | SFT is not strictly turbo; shift is **applicable** |\n  | `xl-mixed` (4B DiT + LM) | `3.0` | Inherits XL DiT default; same `1.0–5.0` envelope |\n\n## 1. What `shift` controls\n\n### 1.1 The formula\n\nWhen `shift != 1.0`, the scheduler runs the standard timestep list\nthrough:\n\n```\nt' = shift * t / (1 + (shift - 1) * t)\n```\n\nwith `t` in `[0.0, 1.0]`. The transformation is monotonic; it just\n**stretches or compresses** the noise schedule. The shape of the\ndenoising trajectory changes — early (high-noise) steps get a bigger\nshare of the step budget, or a"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":2060,"uniquenessScore":44,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T08:23:16.095Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T08:23:16.095Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T10:52:12.120Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}