{"id":"c06da4d6-f4eb-4f47-ad66-84fa1010df45","entityType":"agent","slug":"clawhub-pruna-ai-music-2-5","name":"music-2.5","canonicalUrl":"https://www.xpersona.co/agent/clawhub-pruna-ai-music-2-5","canonicalPath":"/agent/clawhub-pruna-ai-music-2-5","generatedAt":"2026-10-10T17:37:43.255Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-10T13:57:59.884Z","emptyReason":null},"description":"Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. Skill: music-2.5 Owner: pruna-ai Summary: Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. Tags: ai:1.0.14, generative:1.0.14, latest:1.0.14, pruna:1.0.14 Version history: v1.0.14 | 2026-09-29T15:36:02.415Z | auto - Version bump: updated metadata version to 1.0.14. - Removed skill-card.md file. - No changes to usage, features, or required/o","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.4K downloads reported by the source. Last updated 10/10/2026.","installCommand":"clawhub skill install s173djpkkm1x4yfzg522h5vfp989mrn5:music-2-5","sourceUrl":"https://clawhub.ai/pruna-ai/music-2-5","homepage":"https://clawhub.ai/pruna-ai/skills/music-2-5","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/pruna-ai/music-2-5","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/pruna-ai/skills/music-2-5","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":63,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. Skill: music-2.5 Owner: pruna-ai "},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-10T13:57:59.884Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T13:57:59.884Z","emptyReason":null},"stars":null,"forks":null,"downloads":1400,"packageName":null,"latestVersion":"1.0.14","tractionLabel":"1.4K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T13:57:59.884Z","emptyReason":null},"lastUpdatedAt":"2026-10-10T13:57:59.884Z","lastCrawledAt":"2026-10-10T13:57:59.884Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-11T13:57:59.884Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.14","createdAt":"2026-09-29T15:36:02.415Z","changelog":"- Version bump: updated metadata version to 1.0.14. - Removed skill-card.md file. - No changes to usage, features, or required/optional fields.","fileCount":4,"zipByteSize":3808},{"version":"1.0.13","createdAt":"2026-09-17T13:58:54.457Z","changelog":"music-2-5 1.0.13 changelog: - Updated SKILL.md with an incremented version and minor instruction/prompt tweaks. - Improved `p-video` follow-on skill description for more precise guidance. - Removed skill-card.md from the repository.","fileCount":4,"zipByteSize":3968},{"version":"1.0.12","createdAt":"2026-09-10T13:57:25.026Z","changelog":"music-2-5 v1.0.12 - Updated SKILL.md with minor clarifications in the \"Typical next steps\" section, refining the description for `p-video`. - Removed the file: skill-card.md.","fileCount":4,"zipByteSize":4005},{"version":"1.0.11","createdAt":"2026-09-03T14:11:42.562Z","changelog":"- Bumped version to 1.0.11 in SKILL.md metadata. - Removed the file skill-card.md. - No changes to requirements or usage instructions.","fileCount":4,"zipByteSize":3965},{"version":"1.0.10","createdAt":"2026-08-28T07:57:33.330Z","changelog":"music-2-5 v1.0.10 - Updated metadata version to 1.0.10 in SKILL.md. - Removed the file skill-card.md. - No changes to core functionality or usage instructions.","fileCount":4,"zipByteSize":4006},{"version":"1.0.9","createdAt":"2026-08-04T06:18:21.689Z","changelog":"- Version bump from 1.0.8 to 1.0.9 in metadata. - Removed skill-card.md file. - No changes to functionality or documented usage instructions.","fileCount":4,"zipByteSize":3990},{"version":"1.0.8","createdAt":"2026-07-28T17:20:26.189Z","changelog":"- Added intake step: now opens a generation-diversity clarification intake before the first POST. - agent habit now clarifies that intake should start before calling the API. - Removed the file skill-card.md. - Version updated to 1.0.8.","fileCount":4,"zipByteSize":3872},{"version":"1.0.7","createdAt":"2026-07-23T12:35:25.250Z","changelog":"music-2-5 1.0.7 is a streamline and refocus update, tightening documentation and required dependencies. - Consolidated documentation, removing 7 files and merging key guidance into SKILL.md. - Switched to a clear prerequisites section, listing required supporting skills and install commands. - Added explicit usage directions: state skill name, confirm API token, and clarify redirect criteria. - Moved related guides and shared policy details to referenced skills for less duplication. - Updated \"When NOT to use\" and \"Typical next steps\" with cross-links to complementary skills.","fileCount":4,"zipByteSize":4060}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s173djpkkm1x4yfzg522h5vfp989mrn5:music-2-5","setupComplexity":"low","setupSteps":["Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-music-2-5/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-music-2-5/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-music-2-5/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-music-2-5/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-music-2-5/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-music-2-5/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T17:37:43.252Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-music-2-5/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-music-2-5/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-music-2-5/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-music-2-5/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-10T13:57:59.884Z","emptyReason":null},"readme":"Skill: music-2.5\n\nOwner: pruna-ai\n\nSummary: Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video.\n\nTags: ai:1.0.14, generative:1.0.14, latest:1.0.14, pruna:1.0.14\n\nVersion history:\n\nv1.0.14 | 2026-09-29T15:36:02.415Z | auto\n\n- Version bump: updated metadata version to 1.0.14.\n- Removed skill-card.md file.\n- No changes to usage, features, or required/optional fields.\n\nv1.0.13 | 2026-09-17T13:58:54.457Z | auto\n\nmusic-2-5 1.0.13 changelog:\n\n- Updated SKILL.md with an incremented version and minor instruction/prompt tweaks.\n- Improved `p-video` follow-on skill description for more precise guidance.\n- Removed skill-card.md from the repository.\n\nv1.0.12 | 2026-09-10T13:57:25.026Z | auto\n\nmusic-2-5 v1.0.12\n\n- Updated SKILL.md with minor clarifications in the \"Typical next steps\" section, refining the description for `p-video`.\n- Removed the file: skill-card.md.\n\nv1.0.11 | 2026-09-03T14:11:42.562Z | auto\n\n- Bumped version to 1.0.11 in SKILL.md metadata.\n- Removed the file skill-card.md.\n- No changes to requirements or usage instructions.\n\nv1.0.10 | 2026-08-28T07:57:33.330Z | auto\n\nmusic-2-5 v1.0.10\n\n- Updated metadata version to 1.0.10 in SKILL.md.\n- Removed the file skill-card.md.\n- No changes to core functionality or usage instructions.\n\nv1.0.9 | 2026-08-04T06:18:21.689Z | auto\n\n- Version bump from 1.0.8 to 1.0.9 in metadata.\n- Removed skill-card.md file.\n- No changes to functionality or documented usage instructions.\n\nv1.0.8 | 2026-07-28T17:20:26.189Z | auto\n\n- Added intake step: now opens a generation-diversity clarification intake before the first POST.\n- agent habit now clarifies that intake should start before calling the API.\n- Removed the file skill-card.md.\n- Version updated to 1.0.8.\n\nv1.0.7 | 2026-07-23T12:35:25.250Z | auto\n\nmusic-2-5 1.0.7 is a streamline and refocus update, tightening documentation and required dependencies.\n\n- Consolidated documentation, removing 7 files and merging key guidance into SKILL.md.\n- Switched to a clear prerequisites section, listing required supporting skills and install commands.\n- Added explicit usage directions: state skill name, confirm API token, and clarify redirect criteria.\n- Moved related guides and shared policy details to referenced skills for less duplication.\n- Updated \"When NOT to use\" and \"Typical next steps\" with cross-links to complementary skills.\n\nv1.0.6 | 2026-07-16T20:57:54.055Z | auto\n\n- Introduced a shared generation policy section with links to new references for seed rituals, generation diversity, and quality checklists.\n- Added references/generation-quality-checklists.md; updated links throughout to point to new and reorganized reference files.\n- Updated example repo helper invocation path for clarity.\n- Improved descriptions in SKILL.md for broader context and simpler guidance.\n- Removed obsolete skill-card.md file.\n\nv1.0.2 | 2026-07-16T13:25:26.386Z | auto\n\n- Version bump to 1.0.2.\n- Documentation updates in SKILL.md and multiple reference files.\n- Minor clarifications and adjustments to usage instructions and model input descriptions.\n- Removed redundant skill-card.md file.\n\nv1.0.1 | 2026-07-14T15:46:06.459Z | auto\n\n- Adds skill documentation for music-2.5, enabling AI-generated full-length songs with natural vocals from lyrics and style prompts.\n- Describes primary workflow integration with the music-video skill for audio-driven video creation.\n- Details input fields (lyrics, prompt, sample rate, bitrate, audio format) and structure tag system for song arrangement.\n- Includes usage notes, prompt tips, language support, and example API calls for practical integration.\n- Links to related skills and further documentation for music and audio production workflows.\n\nv0.0.1 | 2026-06-30T14:46:18.998Z | auto\n\nInitial release of music-2.5 skill.\n\n- Generate full-length songs with natural vocals from user-supplied lyrics and style prompts.\n- Supports structure tags for flexible song arrangements (e.g., [Verse], [Chorus], [Solo]).\n- Customizable output options: sample rate, bitrate, and audio format (MP3 default).\n- Designed for use as source tracks in AI-driven music video workflows.\n- Strong performance in English and Mandarin; outputs are unique per generation.\n- Not for narration or instrumental-only needs—see related skills for those cases.\n\nArchive index:\n\nArchive v1.0.14: 4 files, 3808 bytes\n\nFiles: skill-card.md (1658b), skill.manifest.json (23b), SKILL.md (4932b), _meta.json (129b)\n\nFile v1.0.14:SKILL.md\n\n---\nname: music-2.5\ndescription: Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video.\nlicense: MIT\nmetadata:\n  version: \"1.0.14\"\n  package: pruna-skills\n  provider: replicate\n  replicate_model: minimax/music-2.5\n---\n\n## Prerequisites\n\nInstall and load these skills before generating (skip if already in context via `@pruna`):\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFollow each skill's **Before generating** / craft sections — do not restate guide content here.\n\n## Agent habit\n\nIn the **first reply**, name `` `music-2.5` `` in backticks, confirm `REPLICATE_API_TOKEN` (or stop with signup links from `pruna-api`), then ask for required inputs. Open intake → **`generation-diversity`** clarification intake before the first `POST`. Redirect when **When NOT to use** fits better.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `gemini-3.1-flash-tts` | Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. | `npx skills add PrunaAI/pruna-skills@gemini-3.1-flash-tts -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n\n## Environment\n\n```bash\nexport REPLICATE_API_TOKEN=r8_...\n```\n\nRequires **`ffmpeg`** / **`ffprobe`** for slicing and assembly in the music-video workflow.\n\n## HTTP (curl)\n\n```bash\ncurl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"lyrics\": \"[Verse]\\nWe built it line by line\\nEvery skill a stepping stone\\n\\n[Chorus]\\nRun the pipeline, watch it grow\\nPruna models, let them flow\",\n      \"prompt\": \"Indie pop, uplifting, warm female vocal, 92 BPM, acoustic guitar and mellow synth pads, no harsh distortion\",\n      \"sample_rate\": 44100,\n      \"bitrate\": 256000,\n      \"audio_format\": \"mp3\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/minimax/music-2.5/predictions\"\n```\n\nPoll `urls.get` until `status` is `succeeded`; download `output`.\n\n## Before generating\n\n1. Complete Prerequisites guide reading order.\n2. Confirm **`lyrics`** (with structure tags) and optional style **`prompt`**. When listing required fields, name **`REPLICATE_API_TOKEN`** (Replicate — not `PRUNA_API_KEY`).\n3. **Model notes:** structure tags on their own lines — `[Intro]` `[Verse]` `[Pre Chorus]` `[Chorus]` `[Hook]` `[Bridge]` `[Solo]` `[Inst]` `[Build Up]` `[Drop]` `[Interlude]` `[Break]` `[Transition]` `[Outro]`. `\\n` = line break (also a safe video cut boundary); `\\n\\n` = pause. Max ~5 minutes per generation. English and Mandarin have strongest pronunciation. Data is sent to MiniMax via Replicate — see their [privacy policy](https://www.minimax.io/platform/protocol/privacy-policy).\n\n## Required input\n\n- `lyrics` (string) — 1–3,500 characters\n\n## Common optional fields\n\n- `prompt` — genre, mood, tempo, vocal timbre, instruments (up to ~2,000 chars)\n- `sample_rate`: `16000` · `24000` · `32000` · **`44100`** (default)\n- `bitrate`: `32000` · `64000` · `128000` · **`256000`** (default)\n- `audio_format`: **`mp3`** (default) · `wav` · `pcm`\n\n## Typical next steps\n\nCommon follow-ons after this skill:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `music-video` | Use when someone wants a full music video — original song or vocals, performance clips, B-roll, and lyric-synced edits. | `npx skills add PrunaAI/pruna-skills@music-video -y` |\n| `whisperx` | Use when someone needs word-level timestamps from audio — lyric alignment, cut-safe line boundaries, or caption source timing before burn-in with video-editing. | `npx skills add PrunaAI/pruna-skills@whisperx -y` |\n| `p-video` | Use when someone wants a simple short clip from text or images — quick B-roll, drafts, or start/end frame animation. Not when the brief needs cinematic generation, highest quality, tight lip-sync, or imported audio at 1080p. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n\nFile v1.0.14:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"music-2-5\",\n  \"version\": \"1.0.14\",\n  \"publishedAt\": 1790696162415\n}\n\nFile v1.0.14:skill-card.md\n\n## Description:\n\nGuides agents in creating original vocal songs from lyrics and optional style prompts using MiniMax Music 2.5 via Replicate.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[pruna-ai](https://clawhub.ai/user/pruna-ai)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nCreators and developers use this skill to guide an agent through preparing lyrics, specifying a musical style, and generating a vocal song for standalone listening or a music video.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Lyrics and prompts are sent to Replicate and MiniMax.\n\nMitigation: Do not submit confidential or regulated text; review the providers' terms before use.\n\nRisk: Unpinned skill-install examples can introduce supply-chain changes.\n\nMitigation: Prefer reviewed, pinned skill versions when available.\n\n## Reference(s):\n\n- [Replicate MiniMax Music 2.5 prediction endpoint](https://api.replicate.com/v1/models/minimax/music-2.5/predictions)\n\n## Skill Output:\n\n**Output Type(s):** [Guidance, Shell commands, JSON]\n\n**Output Format:** [Markdown instructions with shell and JSON examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Guides retrieval of generated song audio; MP3, WAV, and PCM are supported output formats.]\n\n## Skill Version(s):\n\n1.0.14 (source: skill frontmatter and server release)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.14:skill.manifest.json\n\n{\n  \"references\": []\n}\n\nArchive v1.0.13: 4 files, 3968 bytes\n\nFiles: skill-card.md (2037b), skill.manifest.json (23b), SKILL.md (4932b), _meta.json (129b)\n\nFile v1.0.13:SKILL.md\n\n---\nname: music-2.5\ndescription: Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video.\nlicense: MIT\nmetadata:\n  version: \"1.0.13\"\n  package: pruna-skills\n  provider: replicate\n  replicate_model: minimax/music-2.5\n---\n\n## Prerequisites\n\nInstall and load these skills before generating (skip if already in context via `@pruna`):\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFollow each skill's **Before generating** / craft sections — do not restate guide content here.\n\n## Agent habit\n\nIn the **first reply**, name `` `music-2.5` `` in backticks, confirm `REPLICATE_API_TOKEN` (or stop with signup links from `pruna-api`), then ask for required inputs. Open intake → **`generation-diversity`** clarification intake before the first `POST`. Redirect when **When NOT to use** fits better.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `gemini-3.1-flash-tts` | Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. | `npx skills add PrunaAI/pruna-skills@gemini-3.1-flash-tts -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n\n## Environment\n\n```bash\nexport REPLICATE_API_TOKEN=r8_...\n```\n\nRequires **`ffmpeg`** / **`ffprobe`** for slicing and assembly in the music-video workflow.\n\n## HTTP (curl)\n\n```bash\ncurl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"lyrics\": \"[Verse]\\nWe built it line by line\\nEvery skill a stepping stone\\n\\n[Chorus]\\nRun the pipeline, watch it grow\\nPruna models, let them flow\",\n      \"prompt\": \"Indie pop, uplifting, warm female vocal, 92 BPM, acoustic guitar and mellow synth pads, no harsh distortion\",\n      \"sample_rate\": 44100,\n      \"bitrate\": 256000,\n      \"audio_format\": \"mp3\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/minimax/music-2.5/predictions\"\n```\n\nPoll `urls.get` until `status` is `succeeded`; download `output`.\n\n## Before generating\n\n1. Complete Prerequisites guide reading order.\n2. Confirm **`lyrics`** (with structure tags) and optional style **`prompt`**. When listing required fields, name **`REPLICATE_API_TOKEN`** (Replicate — not `PRUNA_API_KEY`).\n3. **Model notes:** structure tags on their own lines — `[Intro]` `[Verse]` `[Pre Chorus]` `[Chorus]` `[Hook]` `[Bridge]` `[Solo]` `[Inst]` `[Build Up]` `[Drop]` `[Interlude]` `[Break]` `[Transition]` `[Outro]`. `\\n` = line break (also a safe video cut boundary); `\\n\\n` = pause. Max ~5 minutes per generation. English and Mandarin have strongest pronunciation. Data is sent to MiniMax via Replicate — see their [privacy policy](https://www.minimax.io/platform/protocol/privacy-policy).\n\n## Required input\n\n- `lyrics` (string) — 1–3,500 characters\n\n## Common optional fields\n\n- `prompt` — genre, mood, tempo, vocal timbre, instruments (up to ~2,000 chars)\n- `sample_rate`: `16000` · `24000` · `32000` · **`44100`** (default)\n- `bitrate`: `32000` · `64000` · `128000` · **`256000`** (default)\n- `audio_format`: **`mp3`** (default) · `wav` · `pcm`\n\n## Typical next steps\n\nCommon follow-ons after this skill:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `music-video` | Use when someone wants a full music video — original song or vocals, performance clips, B-roll, and lyric-synced edits. | `npx skills add PrunaAI/pruna-skills@music-video -y` |\n| `whisperx` | Use when someone needs word-level timestamps from audio — lyric alignment, cut-safe line boundaries, or caption source timing before burn-in with video-editing. | `npx skills add PrunaAI/pruna-skills@whisperx -y` |\n| `p-video` | Use when someone wants a simple short clip from text or images — quick B-roll, drafts, or start/end frame animation. Not when the brief needs cinematic generation, highest quality, tight lip-sync, or imported audio at 1080p. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n\nFile v1.0.13:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"music-2-5\",\n  \"version\": \"1.0.13\",\n  \"publishedAt\": 1789653534457\n}\n\nFile v1.0.13:skill-card.md\n\n## Description:\n\nUse when someone wants an original AI song with vocals, sung lyrics, a style prompt track, or source audio for a music video.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[pruna-ai](https://clawhub.ai/user/pruna-ai)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nExternal users and developers use this skill to collect lyrics and style inputs, call Replicate's MiniMax music-2.5 model, and download the generated song for use directly or as source audio for video workflows.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Unpinned remote skill package installs can change installed agent behavior.\n\nMitigation: Review the remote package and prefer pinned, reviewed versions before running the npx skills add commands, especially the full-suite install.\n\nRisk: Lyrics, prompts, and related generation data are sent to Replicate/MiniMax.\n\nMitigation: Do not submit confidential lyrics, secrets, regulated data, or content you are not authorized to send to Replicate/MiniMax.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/pruna-ai/skills/music-2-5)\n- [Replicate MiniMax music-2.5 predictions API](https://api.replicate.com/v1/models/minimax/music-2.5/predictions)\n- [MiniMax privacy policy](https://www.minimax.io/platform/protocol/privacy-policy)\n\n## Skill Output:\n\n**Output Type(s):** [Guidance, Shell commands, API calls, Configuration]\n\n**Output Format:** [Markdown with inline bash and JSON request examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Agent guidance may include Replicate API polling and download steps; the remote model returns generated audio.]\n\n## Skill Version(s):\n\n1.0.13 (source: evidence release and SKILL.md metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.13:skill.manifest.json\n\n{\n  \"references\": []\n}\n\nArchive v1.0.12: 4 files, 4005 bytes\n\nFiles: skill-card.md (2185b), skill.manifest.json (23b), SKILL.md (4888b), _meta.json (129b)\n\nFile v1.0.12:SKILL.md\n\n---\nname: music-2.5\ndescription: Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video.\nlicense: MIT\nmetadata:\n  version: \"1.0.12\"\n  package: pruna-skills\n  provider: replicate\n  replicate_model: minimax/music-2.5\n---\n\n## Prerequisites\n\nInstall and load these skills before generating (skip if already in context via `@pruna`):\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFollow each skill's **Before generating** / craft sections — do not restate guide content here.\n\n## Agent habit\n\nIn the **first reply**, name `` `music-2.5` `` in backticks, confirm `REPLICATE_API_TOKEN` (or stop with signup links from `pruna-api`), then ask for required inputs. Open intake → **`generation-diversity`** clarification intake before the first `POST`. Redirect when **When NOT to use** fits better.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `gemini-3.1-flash-tts` | Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. | `npx skills add PrunaAI/pruna-skills@gemini-3.1-flash-tts -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n\n## Environment\n\n```bash\nexport REPLICATE_API_TOKEN=r8_...\n```\n\nRequires **`ffmpeg`** / **`ffprobe`** for slicing and assembly in the music-video workflow.\n\n## HTTP (curl)\n\n```bash\ncurl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"lyrics\": \"[Verse]\\nWe built it line by line\\nEvery skill a stepping stone\\n\\n[Chorus]\\nRun the pipeline, watch it grow\\nPruna models, let them flow\",\n      \"prompt\": \"Indie pop, uplifting, warm female vocal, 92 BPM, acoustic guitar and mellow synth pads, no harsh distortion\",\n      \"sample_rate\": 44100,\n      \"bitrate\": 256000,\n      \"audio_format\": \"mp3\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/minimax/music-2.5/predictions\"\n```\n\nPoll `urls.get` until `status` is `succeeded`; download `output`.\n\n## Before generating\n\n1. Complete Prerequisites guide reading order.\n2. Confirm **`lyrics`** (with structure tags) and optional style **`prompt`**. When listing required fields, name **`REPLICATE_API_TOKEN`** (Replicate — not `PRUNA_API_KEY`).\n3. **Model notes:** structure tags on their own lines — `[Intro]` `[Verse]` `[Pre Chorus]` `[Chorus]` `[Hook]` `[Bridge]` `[Solo]` `[Inst]` `[Build Up]` `[Drop]` `[Interlude]` `[Break]` `[Transition]` `[Outro]`. `\\n` = line break (also a safe video cut boundary); `\\n\\n` = pause. Max ~5 minutes per generation. English and Mandarin have strongest pronunciation. Data is sent to MiniMax via Replicate — see their [privacy policy](https://www.minimax.io/platform/protocol/privacy-policy).\n\n## Required input\n\n- `lyrics` (string) — 1–3,500 characters\n\n## Common optional fields\n\n- `prompt` — genre, mood, tempo, vocal timbre, instruments (up to ~2,000 chars)\n- `sample_rate`: `16000` · `24000` · `32000` · **`44100`** (default)\n- `bitrate`: `32000` · `64000` · `128000` · **`256000`** (default)\n- `audio_format`: **`mp3`** (default) · `wav` · `pcm`\n\n## Typical next steps\n\nCommon follow-ons after this skill:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `music-video` | Use when someone wants a full music video — original song or vocals, performance clips, B-roll, and lyric-synced edits. | `npx skills add PrunaAI/pruna-skills@music-video -y` |\n| `whisperx` | Use when someone needs word-level timestamps from audio — lyric alignment, cut-safe line boundaries, or caption source timing before burn-in with video-editing. | `npx skills add PrunaAI/pruna-skills@whisperx -y` |\n| `p-video` | Use when someone wants a simple short clip from text or images — quick B-roll, drafts, or start/end frame animation. Not when the brief needs the highest quality or tight lip-sync. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n\nFile v1.0.12:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"music-2-5\",\n  \"version\": \"1.0.12\",\n  \"publishedAt\": 1789048645026\n}\n\nFile v1.0.12:skill-card.md\n\n## Description:\n\nUse when someone wants an original AI song with vocals - sung lyrics, a style prompt track, or source audio for a music video.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[pruna-ai](https://clawhub.ai/user/pruna-ai)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nExternal users, developers, and creative teams use this skill to guide an agent through generating original vocal music with lyrics, style prompting, Replicate API calls, polling, and audio download steps.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill asks users to install multiple external skills through unpinned npx commands with automatic confirmation.\n\nMitigation: Review and pin external skill install commands before running them, and remove automatic confirmation when users need to inspect changes first.\n\nRisk: Lyrics, prompts, and related creative material are submitted to the Replicate/MiniMax workflow for third-party processing.\n\nMitigation: Do not submit confidential lyrics, unreleased business material, secrets, or regulated data unless that third-party processing is acceptable.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/pruna-ai/skills/music-2-5)\n- [Replicate MiniMax music-2.5 predictions API](https://api.replicate.com/v1/models/minimax/music-2.5/predictions)\n- [MiniMax privacy policy](https://www.minimax.io/platform/protocol/privacy-policy)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown with bash, curl, and JSON examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Guides the agent to collect lyrics and style inputs, confirm REPLICATE_API_TOKEN, call Replicate/MiniMax, poll prediction status, and download generated audio.]\n\n## Skill Version(s):\n\n1.0.12 (source: release evidence and SKILL.md metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.12:skill.manifest.json\n\n{\n  \"references\": []\n}\n\nArchive v1.0.11: 4 files, 3965 bytes\n\nFiles: skill-card.md (2025b), skill.manifest.json (23b), SKILL.md (4884b), _meta.json (129b)\n\nFile v1.0.11:SKILL.md\n\n---\nname: music-2.5\ndescription: Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video.\nlicense: MIT\nmetadata:\n  version: \"1.0.11\"\n  package: pruna-skills\n  provider: replicate\n  replicate_model: minimax/music-2.5\n---\n\n## Prerequisites\n\nInstall and load these skills before generating (skip if already in context via `@pruna`):\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFollow each skill's **Before generating** / craft sections — do not restate guide content here.\n\n## Agent habit\n\nIn the **first reply**, name `` `music-2.5` `` in backticks, confirm `REPLICATE_API_TOKEN` (or stop with signup links from `pruna-api`), then ask for required inputs. Open intake → **`generation-diversity`** clarification intake before the first `POST`. Redirect when **When NOT to use** fits better.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `gemini-3.1-flash-tts` | Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. | `npx skills add PrunaAI/pruna-skills@gemini-3.1-flash-tts -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n\n## Environment\n\n```bash\nexport REPLICATE_API_TOKEN=r8_...\n```\n\nRequires **`ffmpeg`** / **`ffprobe`** for slicing and assembly in the music-video workflow.\n\n## HTTP (curl)\n\n```bash\ncurl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"lyrics\": \"[Verse]\\nWe built it line by line\\nEvery skill a stepping stone\\n\\n[Chorus]\\nRun the pipeline, watch it grow\\nPruna models, let them flow\",\n      \"prompt\": \"Indie pop, uplifting, warm female vocal, 92 BPM, acoustic guitar and mellow synth pads, no harsh distortion\",\n      \"sample_rate\": 44100,\n      \"bitrate\": 256000,\n      \"audio_format\": \"mp3\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/minimax/music-2.5/predictions\"\n```\n\nPoll `urls.get` until `status` is `succeeded`; download `output`.\n\n## Before generating\n\n1. Complete Prerequisites guide reading order.\n2. Confirm **`lyrics`** (with structure tags) and optional style **`prompt`**. When listing required fields, name **`REPLICATE_API_TOKEN`** (Replicate — not `PRUNA_API_KEY`).\n3. **Model notes:** structure tags on their own lines — `[Intro]` `[Verse]` `[Pre Chorus]` `[Chorus]` `[Hook]` `[Bridge]` `[Solo]` `[Inst]` `[Build Up]` `[Drop]` `[Interlude]` `[Break]` `[Transition]` `[Outro]`. `\\n` = line break (also a safe video cut boundary); `\\n\\n` = pause. Max ~5 minutes per generation. English and Mandarin have strongest pronunciation. Data is sent to MiniMax via Replicate — see their [privacy policy](https://www.minimax.io/platform/protocol/privacy-policy).\n\n## Required input\n\n- `lyrics` (string) — 1–3,500 characters\n\n## Common optional fields\n\n- `prompt` — genre, mood, tempo, vocal timbre, instruments (up to ~2,000 chars)\n- `sample_rate`: `16000` · `24000` · `32000` · **`44100`** (default)\n- `bitrate`: `32000` · `64000` · `128000` · **`256000`** (default)\n- `audio_format`: **`mp3`** (default) · `wav` · `pcm`\n\n## Typical next steps\n\nCommon follow-ons after this skill:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `music-video` | Use when someone wants a full music video — original song or vocals, performance clips, B-roll, and lyric-synced edits. | `npx skills add PrunaAI/pruna-skills@music-video -y` |\n| `whisperx` | Use when someone needs word-level timestamps from audio — lyric alignment, cut-safe line boundaries, or caption source timing before burn-in with video-editing. | `npx skills add PrunaAI/pruna-skills@whisperx -y` |\n| `p-video` | Use when someone wants one short video clip from text or images — B-roll, start/end frame animation, or a quick motion shot. Not for full multi-scene films or lip-synced hosts. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n\nFile v1.0.11:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"music-2-5\",\n  \"version\": \"1.0.11\",\n  \"publishedAt\": 1788444702562\n}\n\nFile v1.0.11:skill-card.md\n\n## Description:\n\nUse when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[pruna-ai](https://clawhub.ai/user/pruna-ai)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nExternal creators and developers use this skill to guide an agent through generating original vocal songs from lyrics and style prompts with the MiniMax music-2.5 model on Replicate.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill asks agents to install multiple unpinned remote skills with automatic confirmation before use.\n\nMitigation: Review the referenced PrunaAI skills before installation, prefer pinned or reviewed versions, and avoid installing the full suite unless it is needed.\n\nRisk: Lyrics, style prompts, and generated-song requests are sent to Replicate and MiniMax.\n\nMitigation: Do not submit secrets, private lyrics, regulated data, or content that cannot be shared with those external services.\n\n## Reference(s):\n\n- [music-2.5 on ClawHub](https://clawhub.ai/pruna-ai/skills/music-2-5)\n- [MiniMax music-2.5 Replicate prediction endpoint](https://api.replicate.com/v1/models/minimax/music-2.5/predictions)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, Shell commands, Configuration, Guidance, API calls]\n\n**Output Format:** [Markdown guidance with inline shell commands and HTTP request examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Requires REPLICATE_API_TOKEN; ffmpeg and ffprobe are needed only for slicing and assembly in the music-video workflow.]\n\n## Skill Version(s):\n\n1.0.11 (source: server release evidence and SKILL.md metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.11:skill.manifest.json\n\n{\n  \"references\": []\n}\n\nArchive v1.0.10: 4 files, 4006 bytes\n\nFiles: skill-card.md (2289b), skill.manifest.json (23b), SKILL.md (4884b), _meta.json (129b)\n\nFile v1.0.10:SKILL.md\n\n---\nname: music-2.5\ndescription: Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video.\nlicense: MIT\nmetadata:\n  version: \"1.0.10\"\n  package: pruna-skills\n  provider: replicate\n  replicate_model: minimax/music-2.5\n---\n\n## Prerequisites\n\nInstall and load these skills before generating (skip if already in context via `@pruna`):\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFollow each skill's **Before generating** / craft sections — do not restate guide content here.\n\n## Agent habit\n\nIn the **first reply**, name `` `music-2.5` `` in backticks, confirm `REPLICATE_API_TOKEN` (or stop with signup links from `pruna-api`), then ask for required inputs. Open intake → **`generation-diversity`** clarification intake before the first `POST`. Redirect when **When NOT to use** fits better.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `gemini-3.1-flash-tts` | Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. | `npx skills add PrunaAI/pruna-skills@gemini-3.1-flash-tts -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n\n## Environment\n\n```bash\nexport REPLICATE_API_TOKEN=r8_...\n```\n\nRequires **`ffmpeg`** / **`ffprobe`** for slicing and assembly in the music-video workflow.\n\n## HTTP (curl)\n\n```bash\ncurl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"lyrics\": \"[Verse]\\nWe built it line by line\\nEvery skill a stepping stone\\n\\n[Chorus]\\nRun the pipeline, watch it grow\\nPruna models, let them flow\",\n      \"prompt\": \"Indie pop, uplifting, warm female vocal, 92 BPM, acoustic guitar and mellow synth pads, no harsh distortion\",\n      \"sample_rate\": 44100,\n      \"bitrate\": 256000,\n      \"audio_format\": \"mp3\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/minimax/music-2.5/predictions\"\n```\n\nPoll `urls.get` until `status` is `succeeded`; download `output`.\n\n## Before generating\n\n1. Complete Prerequisites guide reading order.\n2. Confirm **`lyrics`** (with structure tags) and optional style **`prompt`**. When listing required fields, name **`REPLICATE_API_TOKEN`** (Replicate — not `PRUNA_API_KEY`).\n3. **Model notes:** structure tags on their own lines — `[Intro]` `[Verse]` `[Pre Chorus]` `[Chorus]` `[Hook]` `[Bridge]` `[Solo]` `[Inst]` `[Build Up]` `[Drop]` `[Interlude]` `[Break]` `[Transition]` `[Outro]`. `\\n` = line break (also a safe video cut boundary); `\\n\\n` = pause. Max ~5 minutes per generation. English and Mandarin have strongest pronunciation. Data is sent to MiniMax via Replicate — see their [privacy policy](https://www.minimax.io/platform/protocol/privacy-policy).\n\n## Required input\n\n- `lyrics` (string) — 1–3,500 characters\n\n## Common optional fields\n\n- `prompt` — genre, mood, tempo, vocal timbre, instruments (up to ~2,000 chars)\n- `sample_rate`: `16000` · `24000` · `32000` · **`44100`** (default)\n- `bitrate`: `32000` · `64000` · `128000` · **`256000`** (default)\n- `audio_format`: **`mp3`** (default) · `wav` · `pcm`\n\n## Typical next steps\n\nCommon follow-ons after this skill:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `music-video` | Use when someone wants a full music video — original song or vocals, performance clips, B-roll, and lyric-synced edits. | `npx skills add PrunaAI/pruna-skills@music-video -y` |\n| `whisperx` | Use when someone needs word-level timestamps from audio — lyric alignment, cut-safe line boundaries, or caption source timing before burn-in with video-editing. | `npx skills add PrunaAI/pruna-skills@whisperx -y` |\n| `p-video` | Use when someone wants one short video clip from text or images — B-roll, start/end frame animation, or a quick motion shot. Not for full multi-scene films or lip-synced hosts. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n\nFile v1.0.10:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"music-2-5\",\n  \"version\": \"1.0.10\",\n  \"publishedAt\": 1787903853330\n}\n\nFile v1.0.10:skill-card.md\n\n## Description:\n\nUse when someone wants an original AI song with vocals: sung lyrics, a style prompt track, or source audio for a music video.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[pruna-ai](https://clawhub.ai/user/pruna-ai)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nExternal users, developers, and creative teams use this skill to guide an agent through generating original songs with vocals from lyrics and an optional style prompt through Replicate's MiniMax music-2.5 model.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill sends lyrics and prompts to an external provider through Replicate.\n\nMitigation: Avoid submitting private or sensitive lyrics, prompts, or source material, and review MiniMax and Replicate handling policies before use.\n\nRisk: The skill depends on Replicate credentials and may trigger paid external API calls.\n\nMitigation: Use a Replicate token with appropriate account limits and confirm required inputs before making prediction requests.\n\nRisk: The skill asks agents to install related Pruna dependency skills before generation.\n\nMitigation: Verify the Pruna dependency skills before installation and allow only dependencies needed for the intended workflow.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/pruna-ai/skills/music-2-5)\n- [MiniMax privacy policy](https://www.minimax.io/platform/protocol/privacy-policy)\n- [Replicate MiniMax music-2.5 prediction endpoint](https://api.replicate.com/v1/models/minimax/music-2.5/predictions)\n\n## Skill Output:\n\n**Output Type(s):** [Guidance, Shell commands, Configuration, API Calls]\n\n**Output Format:** [Markdown with inline bash, curl, and JSON examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Guides credential setup, required lyrics input, optional music settings, Replicate prediction polling, and generated audio download.]\n\n## Skill Version(s):\n\n1.0.10 (source: server release metadata and artifact metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.10:skill.manifest.json\n\n{\n  \"references\": []\n}\n\nArchive v1.0.9: 4 files, 3990 bytes\n\nFiles: skill-card.md (2216b), skill.manifest.json (23b), SKILL.md (4883b), _meta.json (128b)\n\nFile v1.0.9:SKILL.md\n\n---\nname: music-2.5\ndescription: Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video.\nlicense: MIT\nmetadata:\n  version: \"1.0.9\"\n  package: pruna-skills\n  provider: replicate\n  replicate_model: minimax/music-2.5\n---\n\n## Prerequisites\n\nInstall and load these skills before generating (skip if already in context via `@pruna`):\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFollow each skill's **Before generating** / craft sections — do not restate guide content here.\n\n## Agent habit\n\nIn the **first reply**, name `` `music-2.5` `` in backticks, confirm `REPLICATE_API_TOKEN` (or stop with signup links from `pruna-api`), then ask for required inputs. Open intake → **`generation-diversity`** clarification intake before the first `POST`. Redirect when **When NOT to use** fits better.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `gemini-3.1-flash-tts` | Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. | `npx skills add PrunaAI/pruna-skills@gemini-3.1-flash-tts -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n\n## Environment\n\n```bash\nexport REPLICATE_API_TOKEN=r8_...\n```\n\nRequires **`ffmpeg`** / **`ffprobe`** for slicing and assembly in the music-video workflow.\n\n## HTTP (curl)\n\n```bash\ncurl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"lyrics\": \"[Verse]\\nWe built it line by line\\nEvery skill a stepping stone\\n\\n[Chorus]\\nRun the pipeline, watch it grow\\nPruna models, let them flow\",\n      \"prompt\": \"Indie pop, uplifting, warm female vocal, 92 BPM, acoustic guitar and mellow synth pads, no harsh distortion\",\n      \"sample_rate\": 44100,\n      \"bitrate\": 256000,\n      \"audio_format\": \"mp3\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/minimax/music-2.5/predictions\"\n```\n\nPoll `urls.get` until `status` is `succeeded`; download `output`.\n\n## Before generating\n\n1. Complete Prerequisites guide reading order.\n2. Confirm **`lyrics`** (with structure tags) and optional style **`prompt`**. When listing required fields, name **`REPLICATE_API_TOKEN`** (Replicate — not `PRUNA_API_KEY`).\n3. **Model notes:** structure tags on their own lines — `[Intro]` `[Verse]` `[Pre Chorus]` `[Chorus]` `[Hook]` `[Bridge]` `[Solo]` `[Inst]` `[Build Up]` `[Drop]` `[Interlude]` `[Break]` `[Transition]` `[Outro]`. `\\n` = line break (also a safe video cut boundary); `\\n\\n` = pause. Max ~5 minutes per generation. English and Mandarin have strongest pronunciation. Data is sent to MiniMax via Replicate — see their [privacy policy](https://www.minimax.io/platform/protocol/privacy-policy).\n\n## Required input\n\n- `lyrics` (string) — 1–3,500 characters\n\n## Common optional fields\n\n- `prompt` — genre, mood, tempo, vocal timbre, instruments (up to ~2,000 chars)\n- `sample_rate`: `16000` · `24000` · `32000` · **`44100`** (default)\n- `bitrate`: `32000` · `64000` · `128000` · **`256000`** (default)\n- `audio_format`: **`mp3`** (default) · `wav` · `pcm`\n\n## Typical next steps\n\nCommon follow-ons after this skill:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `music-video` | Use when someone wants a full music video — original song or vocals, performance clips, B-roll, and lyric-synced edits. | `npx skills add PrunaAI/pruna-skills@music-video -y` |\n| `whisperx` | Use when someone needs word-level timestamps from audio — lyric alignment, cut-safe line boundaries, or caption source timing before burn-in with video-editing. | `npx skills add PrunaAI/pruna-skills@whisperx -y` |\n| `p-video` | Use when someone wants one short video clip from text or images — B-roll, start/end frame animation, or a quick motion shot. Not for full multi-scene films or lip-synced hosts. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n\nFile v1.0.9:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"music-2-5\",\n  \"version\": \"1.0.9\",\n  \"publishedAt\": 1785824301689\n}\n\nFile v1.0.9:skill-card.md\n\n## Description: <br>\nUse when someone wants an original AI song with vocals - sung lyrics, a style prompt track, or source audio for a music video. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[pruna-ai](https://clawhub.ai/user/pruna-ai) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nExternal users and developers use this skill to guide an agent through generating original vocal music with Replicate's MiniMax music-2.5 model, including lyric intake, style prompting, API invocation, polling, and download steps. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Lyrics and style prompts are sent to Replicate/MiniMax for music generation. <br>\nMitigation: Use the skill only when the user accepts that data sharing, and avoid submitting confidential lyrics, prompts, or source material. <br>\nRisk: The workflow requires a Replicate API token and may perform paid API calls. <br>\nMitigation: Confirm REPLICATE_API_TOKEN availability and user intent before making generation requests. <br>\nRisk: Follow-on slicing and assembly workflows depend on ffmpeg and ffprobe. <br>\nMitigation: Verify ffmpeg and ffprobe are installed before starting music-video post-processing. <br>\n\n\n## Reference(s): <br>\n- [Replicate MiniMax music-2.5 predictions API](https://api.replicate.com/v1/models/minimax/music-2.5/predictions) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [guidance, shell commands, configuration, code] <br>\n**Output Format:** [Markdown with inline bash and JSON examples] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Guides API-token setup, required lyric input, optional generation parameters, prediction polling, output download, and ffmpeg-dependent post-processing workflows.] <br>\n\n## Skill Version(s): <br>\n1.0.9 (source: server release evidence and frontmatter metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.0.9:skill.manifest.json\n\n{\n  \"references\": []\n}\n\nArchive v1.0.8: 4 files, 3872 bytes\n\nFiles: skill-card.md (1966b), skill.manifest.json (23b), SKILL.md (4883b), _meta.json (128b)\n\nFile v1.0.8:SKILL.md\n\n---\nname: music-2.5\ndescription: Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video.\nlicense: MIT\nmetadata:\n  version: \"1.0.8\"\n  package: pruna-skills\n  provider: replicate\n  replicate_model: minimax/music-2.5\n---\n\n## Prerequisites\n\nInstall and load these skills before generating (skip if already in context via `@pruna`):\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFollow each skill's **Before generating** / craft sections — do not restate guide content here.\n\n## Agent habit\n\nIn the **first reply**, name `` `music-2.5` `` in backticks, confirm `REPLICATE_API_TOKEN` (or stop with signup links from `pruna-api`), then ask for required inputs. Open intake → **`generation-diversity`** clarification intake before the first `POST`. Redirect when **When NOT to use** fits better.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `gemini-3.1-flash-tts` | Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. | `npx skills add PrunaAI/pruna-skills@gemini-3.1-flash-tts -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n\n## Environment\n\n```bash\nexport REPLICATE_API_TOKEN=r8_...\n```\n\nRequires **`ffmpeg`** / **`ffprobe`** for slicing and assembly in the music-video workflow.\n\n## HTTP (curl)\n\n```bash\ncurl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"lyrics\": \"[Verse]\\nWe built it line by line\\nEvery skill a stepping stone\\n\\n[Chorus]\\nRun the pipeline, watch it grow\\nPruna models, let them flow\",\n      \"prompt\": \"Indie pop, uplifting, warm female vocal, 92 BPM, acoustic guitar and mellow synth pads, no harsh distortion\",\n      \"sample_rate\": 44100,\n      \"bitrate\": 256000,\n      \"audio_format\": \"mp3\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/minimax/music-2.5/predictions\"\n```\n\nPoll `urls.get` until `status` is `succeeded`; download `output`.\n\n## Before generating\n\n1. Complete Prerequisites guide reading order.\n2. Confirm **`lyrics`** (with structure tags) and optional style **`prompt`**. When listing required fields, name **`REPLICATE_API_TOKEN`** (Replicate — not `PRUNA_API_KEY`).\n3. **Model notes:** structure tags on their own lines — `[Intro]` `[Verse]` `[Pre Chorus]` `[Chorus]` `[Hook]` `[Bridge]` `[Solo]` `[Inst]` `[Build Up]` `[Drop]` `[Interlude]` `[Break]` `[Transition]` `[Outro]`. `\\n` = line break (also a safe video cut boundary); `\\n\\n` = pause. Max ~5 minutes per generation. English and Mandarin have strongest pronunciation. Data is sent to MiniMax via Replicate — see their [privacy policy](https://www.minimax.io/platform/protocol/privacy-policy).\n\n## Required input\n\n- `lyrics` (string) — 1–3,500 characters\n\n## Common optional fields\n\n- `prompt` — genre, mood, tempo, vocal timbre, instruments (up to ~2,000 chars)\n- `sample_rate`: `16000` · `24000` · `32000` · **`44100`** (default)\n- `bitrate`: `32000` · `64000` · `128000` · **`256000`** (default)\n- `audio_format`: **`mp3`** (default) · `wav` · `pcm`\n\n## Typical next steps\n\nCommon follow-ons after this skill:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `music-video` | Use when someone wants a full music video — original song or vocals, performance clips, B-roll, and lyric-synced edits. | `npx skills add PrunaAI/pruna-skills@music-video -y` |\n| `whisperx` | Use when someone needs word-level timestamps from audio — lyric alignment, cut-safe line boundaries, or caption source timing before burn-in with video-editing. | `npx skills add PrunaAI/pruna-skills@whisperx -y` |\n| `p-video` | Use when someone wants one short video clip from text or images — B-roll, start/end frame animation, or a quick motion shot. Not for full multi-scene films or lip-synced hosts. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n\nFile v1.0.8:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"music-2-5\",\n  \"version\": \"1.0.8\",\n  \"publishedAt\": 1785259226189\n}\n\nFile v1.0.8:skill-card.md\n\n## Description: <br>\nUse this skill when someone wants an original AI song with vocals, sung lyrics, a style prompt track, or source audio for a music video. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[pruna-ai](https://clawhub.ai/user/pruna-ai) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nCreators, developers, and agents use this skill to generate original vocal songs through Replicate/MiniMax from lyrics, style prompts, and audio settings. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Lyrics, style prompts, and generated-audio requests are sent to Replicate and MiniMax. <br>\nMitigation: Confirm the user is comfortable sending this content to those services before generating. <br>\nRisk: The workflow requires a Replicate API token and points agents to related Pruna skills. <br>\nMitigation: Verify REPLICATE_API_TOKEN handling and review the related Pruna skills before using them in the workspace. <br>\n\n\n## Reference(s): <br>\n- [music-2.5 on ClawHub](https://clawhub.ai/pruna-ai/skills/music-2-5) <br>\n- [MiniMax privacy policy](https://www.minimax.io/platform/protocol/privacy-policy) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [guidance, shell commands, configuration] <br>\n**Output Format:** [Markdown guidance with curl examples and environment variable configuration.] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Requires REPLICATE_API_TOKEN and may reference ffmpeg/ffprobe for follow-on music-video assembly.] <br>\n\n## Skill Version(s): <br>\n1.0.8 (source: server release evidence and artifact metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.0.8:skill.manifest.json\n\n{\n  \"references\": []\n}\n\nArchive v1.0.7: 4 files, 4060 bytes\n\nFiles: skill-card.md (2488b), skill.manifest.json (23b), SKILL.md (4794b), _meta.json (128b)\n\nFile v1.0.7:SKILL.md\n\n---\nname: music-2.5\ndescription: Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video.\nlicense: MIT\nmetadata:\n  version: \"1.0.7\"\n  package: pruna-skills\n  provider: replicate\n  replicate_model: minimax/music-2.5\n---\n\n## Prerequisites\n\nInstall and load these skills before generating (skip if already in context via `@pruna`):\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFollow each skill's **Before generating** / craft sections — do not restate guide content here.\n\n## Agent habit\n\nIn the **first reply**, name `` `music-2.5` `` in backticks, confirm `REPLICATE_API_TOKEN` (or stop with signup links from `pruna-api`), then ask for required inputs. Redirect when **When NOT to use** fits better.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `gemini-3.1-flash-tts` | Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. | `npx skills add PrunaAI/pruna-skills@gemini-3.1-flash-tts -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n\n## Environment\n\n```bash\nexport REPLICATE_API_TOKEN=r8_...\n```\n\nRequires **`ffmpeg`** / **`ffprobe`** for slicing and assembly in the music-video workflow.\n\n## HTTP (curl)\n\n```bash\ncurl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"lyrics\": \"[Verse]\\nWe built it line by line\\nEvery skill a stepping stone\\n\\n[Chorus]\\nRun the pipeline, watch it grow\\nPruna models, let them flow\",\n      \"prompt\": \"Indie pop, uplifting, warm female vocal, 92 BPM, acoustic guitar and mellow synth pads, no harsh distortion\",\n      \"sample_rate\": 44100,\n      \"bitrate\": 256000,\n      \"audio_format\": \"mp3\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/minimax/music-2.5/predictions\"\n```\n\nPoll `urls.get` until `status` is `succeeded`; download `output`.\n\n## Before generating\n\n1. Complete Prerequisites guide reading order.\n2. Confirm **`lyrics`** (with structure tags) and optional style **`prompt`**. When listing required fields, name **`REPLICATE_API_TOKEN`** (Replicate — not `PRUNA_API_KEY`).\n3. **Model notes:** structure tags on their own lines — `[Intro]` `[Verse]` `[Pre Chorus]` `[Chorus]` `[Hook]` `[Bridge]` `[Solo]` `[Inst]` `[Build Up]` `[Drop]` `[Interlude]` `[Break]` `[Transition]` `[Outro]`. `\\n` = line break (also a safe video cut boundary); `\\n\\n` = pause. Max ~5 minutes per generation. English and Mandarin have strongest pronunciation. Data is sent to MiniMax via Replicate — see their [privacy policy](https://www.minimax.io/platform/protocol/privacy-policy).\n\n## Required input\n\n- `lyrics` (string) — 1–3,500 characters\n\n## Common optional fields\n\n- `prompt` — genre, mood, tempo, vocal timbre, instruments (up to ~2,000 chars)\n- `sample_rate`: `16000` · `24000` · `32000` · **`44100`** (default)\n- `bitrate`: `32000` · `64000` · `128000` · **`256000`** (default)\n- `audio_format`: **`mp3`** (default) · `wav` · `pcm`\n\n## Typical next steps\n\nCommon follow-ons after this skill:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `music-video` | Use when someone wants a full music video — original song or vocals, performance clips, B-roll, and lyric-synced edits. | `npx skills add PrunaAI/pruna-skills@music-video -y` |\n| `whisperx` | Use when someone needs word-level timestamps from audio — lyric alignment, cut-safe line boundaries, or caption source timing before burn-in with video-editing. | `npx skills add PrunaAI/pruna-skills@whisperx -y` |\n| `p-video` | Use when someone wants one short video clip from text or images — B-roll, start/end frame animation, or a quick motion shot. Not for full multi-scene films or lip-synced hosts. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n\nFile v1.0.7:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"music-2-5\",\n  \"version\": \"1.0.7\",\n  \"publishedAt\": 1784810125250\n}\n\nFile v1.0.7:skill-card.md\n\n## Description: <br>\nUse when someone wants an original AI song with vocals: sung lyrics, a style prompt track, or source audio for a music video. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[pruna-ai](https://clawhub.ai/user/pruna-ai) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nExternal developers and agents use this skill to prepare Replicate requests for MiniMax music-2.5, collecting lyrics, style prompts, audio format settings, and follow-on workflow guidance for original AI songs with vocals. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Lyrics and style prompts are sent to Replicate/MiniMax for generation. <br>\nMitigation: Confirm the user is comfortable sharing those inputs with the third-party model provider before making API calls. <br>\nRisk: The Replicate API token could be exposed if pasted into prompts, code, or generated files. <br>\nMitigation: Use REPLICATE_API_TOKEN from the environment and avoid echoing or writing the secret value. <br>\nRisk: The workflow depends on referenced Pruna helper skills and may require ffmpeg or ffprobe for music-video slicing and assembly. <br>\nMitigation: Install or load the prerequisite skills and confirm required local tools are available before generation or downstream video work. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/pruna-ai/skills/music-2-5) <br>\n- [Replicate MiniMax music-2.5 predictions endpoint](https://api.replicate.com/v1/models/minimax/music-2.5/predictions) <br>\n- [MiniMax privacy policy](https://www.minimax.io/platform/protocol/privacy-policy) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [Guidance, Shell commands, Configuration, API requests] <br>\n**Output Format:** [Markdown with inline bash and JSON request examples] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Guides the agent to confirm REPLICATE_API_TOKEN, collect lyrics and optional music settings, call Replicate, poll for completion, and download generated audio.] <br>\n\n## Skill Version(s): <br>\n1.0.7 (source: server release evidence and skill frontmatter metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.0.7:skill.manifest.json\n\n{\n  \"references\": []\n}\n\nArchive v1.0.6: 10 files, 22780 bytes\n\nFiles: README-INSTALL.md (894b), references/api-credentials.md (3128b), references/generation-diversity.md (25978b), references/generation-quality-checklists.md (10208b), references/random-seed-ritual.md (4011b), references/replicate-api.md (2251b), skill-card.md (2802b), skill.manifest.json (128b), SKILL.md (5528b), _meta.json (128b)\n\nFile v1.0.6:SKILL.md\n\n---\nname: music-2.5\ndescription: Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video.\nlicense: MIT\nmetadata:\n  version: \"1.0.6\"\n  provider: replicate\n  replicate_model: minimax/music-2.5\n---\n\n## Shared generation policy\n\n<!-- shared-generation-policy -->\n\nBefore any paid `POST /v1/predictions`:\n\n1. **[Random seed ritual](./references/random-seed-ritual.md)** — always first; derive axes via sum-mod.\n2. **[Generation diversity](./references/generation-diversity.md)** — explicit prompts; rotate ≥2 scenario axes per session.\n3. **[Quality checklists](./references/generation-quality-checklists.md)** — open output files and judge pass/fail before advancing.\n\n# Music 2.5 (MiniMax · Replicate)\n\nFull-length **songs with natural vocals** from lyrics + style description. Not a Pruna P-model — runs on [Replicate](https://replicate.com/minimax/music-2.5).\n\n**Primary workflow:** [music-video](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/music-video/skills/music-video/SKILL.md) — lyrics → song → lyric-safe cuts → `p-video-avatar` / `p-video` clips → assembly.\n\n## When to use\n\n| Goal | Use this |\n|------|----------|\n| Sung track for a music video | Yes — write lyrics with section tags first |\n| Drive **`p-video`** clip length | Yes — export MP3 → upload to Pruna → `audio` input |\n| Documentary narration | No — use [gemini-3.1-flash-tts](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/gemini-3.1-flash-tts/skills/gemini-3.1-flash-tts/SKILL.md) |\n| Instrumental bed under VO | No — use [stable-audio-2.5](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/stable-audio-2.5/skills/stable-audio-2.5/SKILL.md) |\n\n## Environment\n\n```bash\nexport REPLICATE_API_TOKEN=r8_...\n```\n\nRequires **`ffmpeg`** / **`ffprobe`** for slicing and assembly in the music-video workflow.\n\n## Model input (Replicate)\n\n| Field | Notes |\n|-------|-------|\n| `lyrics` | **Required.** 1–3,500 characters. Use [structure tags](#structure-tags) and `\\n` line breaks. |\n| `prompt` | Optional style string — genre, mood, tempo, vocal timbre, key instruments (up to ~2,000 chars). |\n| `sample_rate` | `16000` · `24000` · `32000` · **`44100`** (default) |\n| `bitrate` | `32000` · `64000` · `128000` · **`256000`** (default) |\n| `audio_format` | **`mp3`** (default) · `wav` · `pcm` |\n\nOutput: audio file URL (typically **MP3**, ~2:30–4:30 for full songs).\n\n## Structure tags\n\nControl arrangement with tags on their own lines (see [Music 2.5 readme](https://replicate.com/minimax/music-2.5)):\n\n`[Intro]` · `[Verse]` · `[Pre Chorus]` · `[Chorus]` · `[Hook]` · `[Bridge]` · `[Solo]` · `[Inst]` · `[Build Up]` · `[Drop]` · `[Interlude]` · `[Break]` · `[Transition]` · `[Outro]`\n\n- One tag per section; follow with 2–4 lyric lines per section for clean melodies.\n- `\\n` = line break (also a **safe video cut boundary** in the music-video workflow).\n- `\\n\\n` = pause between sections.\n- Parentheticals work for ad-libs and directions: `(Ooh, yeah)` · `(Guitar solo — slow, bluesy)`.\n\n## Prompt tips\n\nFollow the model’s [prompt guide](https://replicate.com/minimax/music-2.5): genre + mood + vocal description + tempo + instruments + production feel.\n\n**Example prompt:**\n\n```text\nIndie pop, uplifting, warm female vocal, 92 BPM, acoustic guitar and soft synth pads,\nwide soundstage, crisp modern production, anthemic chorus\n```\n\n**Instrumental sections:** use `[Inst]` or `[Solo]` tags with parenthetical instrument directions instead of sung lines.\n\n## HTTP (curl)\n\n```bash\ncurl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"lyrics\": \"[Verse]\\nWe built it line by line\\nEvery skill a stepping stone\\n\\n[Chorus]\\nRun the pipeline, watch it grow\\nPruna models, let them flow\",\n      \"prompt\": \"Indie pop, uplifting, warm female vocal, 92 BPM, acoustic guitar and mellow synth pads, no harsh distortion\",\n      \"sample_rate\": 44100,\n      \"bitrate\": 256000,\n      \"audio_format\": \"mp3\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/minimax/music-2.5/predictions\"\n```\n\nPoll `urls.get` until `status` is `succeeded`; download `output`.\n\n## Repo helper\n\n```bash\npython3 workflows/music-video/scripts/generate_song.py \\\n  --plan output/my-music-video/music_video_plan.json \\\n  --out-dir output/my-music-video\n```\n\n## Good to know\n\n- **English and Mandarin** have strongest pronunciation; other languages vary.\n- Each generation is unique — same lyrics + prompt produce different arrangements.\n- Max ~5 minutes per generation.\n- Data is sent to MiniMax via Replicate — see their [privacy policy](https://www.minimax.io/platform/protocol/privacy-policy).\n\n## Related\n\n- [audio-post-production.md](https://github.com/PrunaAI/pruna-skills/tree/main/references/shared/audio-post-production.md) — when to use songs vs narration vs beds\n- [music-video workflow](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/music-video/skills/music-video/SKILL.md)\n- [gemini-3.1-flash-tts](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/gemini-3.1-flash-tts/skills/gemini-3.1-flash-tts/SKILL.md) — spoken narration (not song)\n- [stable-audio-2.5](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/stable-audio-2.5/skills/stable-audio-2.5/SKILL.md) — instrumental beds only\n- [replicate-api.md](https://github.com/PrunaAI/pruna-skills/tree/main/references/policies/replicate-api.md)\n\nFile v1.0.6:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"music-2-5\",\n  \"version\": \"1.0.6\",\n  \"publishedAt\": 1784235474055\n}\n\nFile v1.0.6:references/api-credentials.md\n\n# API credentials (Pruna + Replicate)\n\n**Agent rule:** Before any `POST /v1/predictions`, Replicate prediction, or paid runner — check env vars. If a required key is **missing or empty**, **stop** and tell the user how to sign up. Do not guess, mock, or skip with placeholder keys.\n\n## Pruna P-API\n\n| | |\n|--|--|\n| **Env var** | `PRUNA_API_KEY` |\n| **Header** | `apikey: ${PRUNA_API_KEY}` (not `Authorization: Bearer`) |\n| **Sign up / get key** | [Pruna dashboard](https://dashboard.pruna.ai/) |\n| **Docs** | [Quickstart](https://docs.api.pruna.ai/guides/quickstart) · [pruna-api.md](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/pruna-api/SKILL.md) |\n\n**Used by:** all `p-image*`, `p-video*` tool skills and Pruna workflow runners.\n\n### If `PRUNA_API_KEY` is missing — agent message template\n\n> Pruna generation needs an API key. Sign up or sign in at **[dashboard.pruna.ai](https://dashboard.pruna.ai/)**, create an API key, then set:\n>\n> ```bash\n> export PRUNA_API_KEY=\"your_key_here\"\n> ```\n>\n> Add that to your shell profile or project `.env` (never commit the key). Reply when it’s set and we can continue.\n\n## Replicate\n\n| | |\n|--|--|\n| **Env var** | `REPLICATE_API_TOKEN` |\n| **Header** | `Authorization: Bearer ${REPLICATE_API_TOKEN}` |\n| **Sign up / get token** | [Replicate API tokens](https://replicate.com/account/api-tokens) ([sign in](https://replicate.com/signin) first if needed) |\n| **Docs** | [replicate-api.md](./replicate-api.md) |\n\n**Used by:** `music-2.5`, `gemini-3.1-flash-tts`, `stable-audio-2.5`, `whisperx`, and workflow beds/TTS/song phases.\n\n### If `REPLICATE_API_TOKEN` is missing — agent message template\n\n> This step uses Replicate (song, TTS, transcription, or background bed). Create a token at **[replicate.com/account/api-tokens](https://replicate.com/account/api-tokens)**, then set:\n>\n> ```bash\n> export REPLICATE_API_TOKEN=\"r8_...\"\n> ```\n>\n> Reply when it’s set and we can continue.\n\n## Which key does this job need?\n\n| Task | Keys required |\n|------|----------------|\n| `p-image`, `p-image-edit`, `p-image-upscale`, `p-image-try-on` | `PRUNA_API_KEY` |\n| `p-video`, `p-video-avatar`, `p-video-animate`, `p-video-replace` | `PRUNA_API_KEY` |\n| Music 2.5 song generation | `REPLICATE_API_TOKEN` |\n| Gemini TTS narration | `REPLICATE_API_TOKEN` |\n| Stable Audio background bed | `REPLICATE_API_TOKEN` |\n| WhisperX transcription | `REPLICATE_API_TOKEN` |\n| Music video / explainer (full pipeline) | **Both** — Pruna for stills/video; Replicate for song/TTS/bed as needed |\n\nWhen only one key is missing, suggest **only** that provider’s signup link — not both.\n\n## Security\n\n- Never print full keys in chat or commit them to git.\n- `.env` is gitignored; prefer env vars over hardcoding in plans or manifests.\n- Never embed keys in prompts, manifests, plan JSON, logs, or **subagent task text**.\n- Prefer the **parent agent** to own API calls; do not fan credentials across parallel subagents unless the host documents isolated secret injection.\n- Full rules: [agent-safety.md](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/agent-safety/SKILL.md).\n\nFile v1.0.6:references/generation-diversity.md\n\n# Generation diversity (all models)\n\nOne checklist so **every** Pruna output — **`p-image`**, **`p-video`**, try-on, avatar, replace, animate — is as **diverse** as the brief allows. Details live in linked docs; this page is the agent shortcut.\n\nUse the **full** checklist here for every generation.\n\n## Contents\n\n- [Three steps (every job)](#three-steps-every-job)\n- [Explicit prompt structure](#explicit-prompt-structure-required)\n- [Text & typography by model](#text--typography-by-model)\n- [SSoT axis derivation](#ssot-axis-derivation-sum-mod)\n- [Scenario axes](#scenario-axes-rotate-across-outputs)\n- [Render categories](#render-categories)\n- [Crowded scenes](#crowded-scenes-p-image)\n- [Body type spread](#body-type-spread)\n- [Location-matched crowds](#location-matched-crowds)\n- [Group classes](#group-classes--courses)\n- [Framing & camera](#framing--camera)\n- [Scene spice](#scene-spice-when-it-fits)\n- [Photoreal anti-slop](#photoreal-anti-slop-neon--stylized-briefs)\n- [Aspect ratio](#aspect-ratio-multi-example-sets)\n- [By model](#by-model-minimum-diversity)\n- [When not to maximize diversity](#when-not-to-maximize-diversity)\n- [Anti-patterns](#anti-patterns)\n\n## Three steps (every job)\n\n1. **[Random seed ritual](./random-seed-ritual.md) (SSoT)** — **always first**, before the prompt. Generate a fresh random string, **state it in the turn**, derive axes via [sum-mod](#ssot-axis-derivation-sum-mod). **Do not** pass the ritual string to API `seed`. **One new ritual string per independent generation**; reuse only on same-brief slop retry.\n2. **Write an [explicit prompt](#explicit-prompt-structure-required)** — name specific people, animals, objects, actions, setting, and camera/light. Add text/typography only when the brief needs it — see [text rules by model](#text--typography-by-model).\n3. **Diversify the scenario row** — change at least **two axes** from the previous output in the same session (cast, setting, camera, **`render_category_tag`**, **aspect_ratio**, creatures, props, … — unless user asked for continuity).\n4. **Log** — `ritual_seed`, axes chosen, prediction id (manifest or turn text).\n\n## Explicit prompt structure (required)\n\n**Vague prompts produce generic AI slop.** After the ritual and axis picks, every still prompt must be **specific and dynamic** — concrete nouns, frozen actions, named places. Prefer playground/creative briefs over marketing abstractions.\n\n**Name at least four of these per prompt (log tags in manifest):**\n\n| Clause | Log as | Agent must specify |\n|--------|--------|-------------------|\n| **People** | `cast_descriptor` | Named role + age band + expression (`fearless grandmother in floral apron`, not `woman`) |\n| **Animals / creatures** | `creature_tag` | Species + attitude (`otter DJ`, `luna moth knight`, `VIP anglerfish`) |\n| **Objects** | `prop_tag` | Concrete props (`vinyl record`, `chrome rocket sled`, `velvet rope`, `tiny boombox`) |\n| **Action** | `action_tag` | Frozen mid-motion verb (`scratching vinyl`, `lassoing runaway taco truck`, `cape mid-swing`) |\n| **Duration** | `duration_tag` | When timing matters (`1970s`, `8PM`, `45-minute spin class`, `Saturday-morning cartoon`) |\n| **Setting** | `setting_tag` | Named place + era + materials (`packed 1970s roller rink`, `abyss-depth jellyfish nightclub`, `Monument Valley dust storm`) |\n| **Text / typography** | `text_spec` | Only when brief needs readable type — exact strings + surface (see [by model](#text--typography-by-model)) |\n| **Camera + light** | `camera_tag`, `lighting_tag` | `fish-eye lens`, `tilt-shift macro`, `teal-magenta cinematic`, `golden hour sparkle` |\n| **Style** | `render_category_tag` | Medium (`cel-shaded anime`, `baroque oil painting`, `ink-wash storybook`, `photoreal documentary`) |\n\n**Template:**\n\n```text\n{people and/or creatures} {action} with/at {specific objects} in {named setting},\n{style or era cues}, {camera_tag}, {lighting_tag}\n```\n\n**Good examples (dynamic / specific):**\n\n```text\nDisco ball reflections on an otter DJ scratching vinyl at a packed 1970s roller rink,\nfish-eye lens, glitter confetti mid-air, funky energy\n```\n\n```text\nBioluminescent jellyfish nightclub at abyss depth, VIP anglerfish in sunglasses at velvet rope,\nteal-magenta cinematic lighting\n```\n\n```text\nCorgi cowboy lassoing a runaway taco truck through Monument Valley dust storm,\npulp western poster energy, dynamic diagonal composition\n```\n\n**Anti-pattern:** `cool cyberpunk portrait, neon vibes` — no subject, no action, no place. **Right:** name who, what they're doing, where, with which props.\n\n## Text & typography by model\n\n**Never use negation to suppress text** — `no text`, `without signs`, `no typography` often **invoke** the thing you are trying to avoid. Describe surfaces positively when you want blank walls (`plain unmarked walls`, `matte unprinted props`).\n\n| Model | Prompt upsampling | Typography in prompt |\n|-------|-------------------|----------------------|\n| **`p-image`** | **No** effective prompt upsampling | **Avoid** dense readable-type requests unless user explicitly wants `text_rendering`. Short prompts; skip `readable`, `legible`, `headline`, multi-sign lists — they drift to gibberish. Collage triggers still apply: [interactive-explainer-prompts.md](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/interactive-explainer-prompts.md) (`flat lay`, `grid`, `collage`, …). |\n\n**`p-image` text hygiene:** prefer scenes without copy. If a screen appears: `monitor soft colorful blur glow only` — not legible UI unless the user explicitly asked for readable text (then simplify the brief or drop copy).\n\n**Collage triggers (all T2I models):** still avoid `flat lay`, `packshot`, `grid`, `collage`, `montage`, `contact sheet`, `split`, `before and after` — use `single frame`, `one camera angle` instead. Full table: [interactive-explainer-prompts.md](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/interactive-explainer-prompts.md).\n\n## SSoT axis derivation (sum-mod)\n\nAfter stating `ritual_seed` (random string), derive prompt choices — sum Unicode/ASCII char codes, mod list length:\n\n```text\nRATIOS = [\"1:1\", \"16:9\", \"9:16\", \"4:3\", \"3:4\", \"3:2\", \"2:3\"]\naspect_ratio  ← RATIOS[ sum(codes(ritual_seed)) % 7 ]\ncamera_tag    ← camera_tags[ sum(codes(ritual_seed[0:4])) % len(camera_tags) ]\nrender_tag    ← render_tags[ sum(codes(ritual_seed[4:8])) % len(render_tags) ]\n```\n\n`camera_tags` and `render_tags` — see [framing & camera](#framing--camera) and [render categories](#render-categories). State derived picks in the turn (*\"Aspect ratio: 16:9, camera: over-shoulder\"*).\n\n**User `api_seed`:** when the user supplies an integer for reproducibility, pass it as `input.seed` — separate from the ritual string.\n\n## Scenario axes (rotate across outputs)\n\n| Axis | Vary with | Applies to |\n|------|-----------|------------|\n| **Cast** | age, ethnicity, gender, archetype, **hairstyle**, **body type** (rotate — see [below](#body-type-spread)), disability aids (wheelchair, cane), visible age band twice in prompt | all person/content gens |\n| **Medium** | `render_category_tag` — rotate across [render categories](#render-categories) | `p-image`, avatar stills |\n| **Setting** | unique `setting_tag` — specific room/street/venue/era, not repeat adjacent rows | stills + video plates |\n| **Camera** | `camera_tag` — rotate across [framing ladder](#framing--camera); never default MC facing lens | stills, `video_prompt` |\n| **Lighting** | `lighting_tag` — golden hour · neon · overcast · practical | stills, video mood |\n| **Motion** | unique `video_prompt` per clip | `p-video`, `p-video-avatar`, animate |\n| **Voice** | natural `voice_script`; one `voice` preset per character | avatar, TTS-led video |\n| **Seed** | new ritual string per **independent** job; reuse only on same-brief slop retry | all generation skills |\n| **Aspect ratio** | different `aspect_ratio` per independent still in a batch — see [below](#aspect-ratio-multi-example-sets) | `p-image`, `p-image-edit` |\n| **Crowd density** | layered background population + activity cues — see [below](#crowded-scenes-p-image) | `p-image` plates with busy worlds |\n\nFull style/camera/lighting ladders: [visual-variety-bible.md](https://github.com/PrunaAI/pruna-skills/tree/main/shared/visual-variety-bible.md). Persona + try-on bar: [realistic-persona-showcase.md](https://github.com/PrunaAI/pruna-skills/tree/main/shared/realistic-persona-showcase.md).\n\n## Render categories\n\nRotate **`render_category_tag`** (and log it) so diversity batches cover more than photoreal portraits or anime. Category families below mirror arena leaderboards — pick a **different tag per independent output**.\n\n**Random seed ritual still applies** to every generation in [step 1](#three-steps-every-job); categories describe *what* to vary, not *when* to pick `seed`.\n\n### Text-to-image — `p-image`\n\nSources: [Arena text-to-image](https://arena.ai/leaderboard/text-to-image) · [AA text-to-image](https://artificialanalysis.ai/image/leaderboard/text-to-image)\n\n**Unified `render_category_tag`** (Arena bucket = tag — pick one per still):\n\n`product_branding_commercial` · `3d_imaging_modeling` · `cartoon_anime_fantasy` · `photoreal_cinematic` · `art` · `portraits` · `nature_environment` · `animals_creature` · `text_rendering`\n\n| Tag | Typical prompt lane |\n|-----|---------------------|\n| `product_branding_commercial` | single product on seamless studio, person + product in named setting, showroom (not `flat lay` / `packshot` words) |\n| `3d_imaging_modeling` | CG film still, clay/stop-motion, rounded 3D forms |\n| `cartoon_anime_fantasy` | cel anime, fantasy character, crowded stylized world |\n| `photoreal_cinematic` | documentary crowd scenes, film-scale wide, urban march |\n| `art` | oil, watercolor, gouache, charcoal, flat vector |\n| `portraits` | single-subject editorial or documentary portrait (crowd optional behind) |\n| `nature_environment` | landscape-wide; subject small in frame |\n| `animals_creature` | named species + handler; crowded market/park when it fits |\n| `text_rendering` | **user-requested only** — otherwise no readable text |\n\nLog `render_category_tag` in manifest. Combine with [crowded scenes](#crowded-scenes-p-image), [body type](#body-type-spread), and [scene spice](#scene-spice-when-it-fits) when the brief allows.\n\n### Image edit — `p-image-edit`\n\nSources: [Arena image edit](https://arena.ai/leaderboard/image-edit) · [AA image editing](https://artificialanalysis.ai/image/leaderboard/editing)\n\nArena modalities: `single_image_edit` · `multi_image_edit`\n\nEdit diversity tags: `background_swap` · `relight` · `wardrobe_on_plate` · `pose_or_angle_delta` · `multi_ref_composite` · `region_inpaint`\n\nVary **instruction** and **what changes** while identity URL stays fixed on character arcs.\n\n### Text-to-video — `p-video`\n\nSources: [Arena text-to-video](https://arena.ai/leaderboard/text-to-video) · [AA text-to-video](https://artificialanalysis.ai/video/leaderboard/text-to-video)\n\nMotion/scene tags: `character_performance` · `landscape_broll` · `urban_street` · `product_demo` · `abstract_mood` · `crowd_scene` · `dialogue_beat`\n\nRotate `video_prompt` grammar, start plate world, and `camera_tag` per clip.\n\n### Image-to-video — `p-video` (+ plate upload)\n\nSources: [Arena image-to-video](https://arena.ai/leaderboard/image-to-video) · [AA image-to-video](https://artificialanalysis.ai/video/leaderboard/image-to-video)\n\nPlate-driven tags: `animate_hero_still` · `camera_move_on_plate` · `environmental_parallax` · `avatar_lip_sync` · `hands_or_prop_motion`\n\nMatch motion to what the **still** already shows — do not contradict the plate.\n\n### Video edit — `p-video-replace` (and edit-style video)\n\nSource: [Arena video edit](https://arena.ai/leaderboard/video-edit)\n\nEdit tags: `face_recast` · `wardrobe_swap` · `accessory_swap` · `background_replace` · `object_in_hand_swap` · `style_transfer_on_subject`\n\nSame-gender / identity rules for talking-head beats still apply — see [visual-variety-bible.md](https://github.com/PrunaAI/pruna-skills/tree/main/shared/visual-variety-bible.md).\n\n## Crowded scenes (`p-image`)\n\nWhen the brief asks for **busy**, **crowded**, or **lively** worlds — not a lone subject on a blank wall — stack density in the prompt:\n\n1. **Three depth layers** — sharp foreground subject · readable midground faces/hands/props · landmark bokeh (stage, temple, billboards, ferris wheel).\n2. **Named population count** — `hundreds of pedestrians`, `dozens of faces in midground`, `20+ tiny clay figures` (stylized sets need explicit counts; models under-deliver on vague \"busy\").\n3. **Activity verbs** — raised hands, umbrellas open, food steam, confetti, market haggling, commuters pressed shoulder-to-shoulder.\n4. **Shallow DOF + single subject** — `single subject one frame` keeps one identity readable while the crowd stays behind them.\n5. **Age & angle lock** — repeat age band twice (`woman in her late 50s, visibly fifty`) and use [framing & camera](#framing--camera) — models drift younger, center-frame, and front-facing without it.\n\n| Crowd family | Density cues |\n|--------------|--------------|\n| **Urban rush** | crosswalk stripes, wet reflections, umbrellas, billboard bokeh |\n| **Festival / parade** | confetti, raised hands, costume layers, smoke haze |\n| **Market / bazaar** | overflowing stalls, hanging goods, steam, price tags as color blobs |\n| **Transit crush** | strap hangers, door windows, blurred faces pressed together |\n| **Stylized miniature** | counted clay/figurine shoppers (`20+`), cramped aisle, stacked crates |\n| **Institutional / ER** | framed oil portraits on beige walls, triage number board, wall sanitizer, vending machine, scuffed linoleum, TV blur, mixed-age seated patients |\n| **Urban march / protest** | named city, local landmarks, multiracial crowd cues separate from hero — see [location-matched crowds](#location-matched-crowds) |\n| **Group fitness class** | class name + duration, mixed-gender riders, realistic warm studio light — see [group classes](#group-classes--courses) |\n\n**Anti-pattern:** one blurred smear behind a portrait — name **what** the crowd is doing and **where** layers sit. **Institutional** scenes (ER, airport, classroom) need `benches full`, `standing room only`, or `shoulder-to-shoulder` — otherwise models default to a quiet hallway. Name **set dressing** too: framed portraits on walls, triage number board, vending machine glow, scuffed linoleum — generic mint corridors read AI-empty.\n\n## Body type spread\n\nModels default to one “average fitness” body. In diversity batches, **name build on the hero and vary background bodies**:\n\n| Build tag | Prompt cue |\n|-----------|------------|\n| **Plus-size / curvy** | `plus-size`, `curvy build`, `full-figured` |\n| **Athletic / muscular** | `broad shoulders`, `muscular arms`, `athletic build` |\n| **Petite / slim** | `petite frame`, `slim build`, `narrow shoulders` |\n| **Tall / lanky** | `tall and lanky`, `6-foot frame`, `long limbs` |\n| **Stocky / heavyset** | `stocky build`, `heavyset`, `barrel chest` |\n| **Lean wiry** | `lean wiry frame`, `weathered thin face` |\n\n**Rule:** rotate build across independent panels in a session — not every hero “athletic build”. Background crowd should mix ages **and** silhouettes (`elderly thin woman`, `heavyset man`, `pregnant woman seated`, `toddler on lap`).\n\n## Location-matched crowds\n\nWhen the prompt names a **real city or country**, background faces must match that place’s **demographic mix** — not clone the hero’s ethnicity.\n\n| Wrong | Right |\n|-------|--------|\n| South Asian hero + only South Asian protesters in “New York” | Hero is one identity; crowd explicitly `multiracial NYC march — Black, Latino, white, East Asian protesters` |\n| “Dense city march” with no geography | Name city + 3–4 crowd ethnicity cues + local landmarks (yellow cabs, art deco towers, steam vent) |\n| Festival in Lagos with only Nordic faces | Match crowd to `setting_tag` region |\n\n**Prompt pattern:** lock hero cast in sentence 1; sentence 2 lists **four+ distinct background silhouettes** unrelated to hero ethnicity; sentence 3 names **local landmarks** so the plate cannot read as generic stock.\n\n**Applies to:** protests, airports, transit, street markets, sports crowds — any scene where “crowded” implies a real place.\n\n## Group classes & courses\n\nWhen the scene is a **class, workshop, or team activity**, name the **course type** and **who else is in the room** — models default to monochrome crowds (all men, all one age).\n\n| Specify | Example cues |\n|---------|----------------|\n| **Class type** | `45-minute evening spin class`, `beginner yoga flow`, `HIIT bootcamp circuit` |\n| **Room realism** | warm overhead track lights, mirror wall, rubber floor, water bottles, towels — **not** magenta-cyan neon strips unless brief is explicitly nightclub |\n| **Gender mix** | hero is one person; crowd `mixed-gender class — women with ponytails, men with beards, nonbinary cyclist` |\n| **Body + age mix** | plus-size rider, petite woman, athletic man, woman in her 50s — same as [body type spread](#body-type-spread) |\n\n**Lighting rule for fitness:** real boutique studios are **dim warm overhead** or **single spotlight on instructor** — avoid `split gel`, `neon LED strips`, `magenta-cyan` on photoreal gym plates; those read AI-fake.\n\n**Prompt pattern:** `Documentary fitness portrait` + class name + instructor on bike at front + `20+ mixed-gender cyclists` with 3–4 named background silhouettes + realistic room props.\n\n## Framing & camera\n\nModels default to **centered subject, eyes at camera**. In diversity batches, **rotate `camera_tag` and frame placement** every row — log both in manifest.\n\n**Gaze rule:** `glance off-lens`, `profile`, `back to camera`, `looking down at [prop]`, or `watching the crowd` — **not** `facing camera` or `looking at viewer` unless the user asked for a direct-address avatar plate.\n\n**Placement rule:** name where the subject sits in frame — `left third`, `right third`, `lower right corner`, `edge of frame`, `small in environmental wide` — **not** centered mugshot every time.\n\n| `camera_tag` | Prompt cue |\n|--------------|------------|\n| **Overhead / bird's eye** | `overhead aerial view`, `top-down`, `drone shot looking straight down` |\n| **High corner** | `high angle from corner`, `surveillance-style downward angle` |\n| **Worm's eye** | `ground-level worm's eye`, `camera on pavement` |\n| **Crane-down** | `slight high angle crane-down` |\n| **Over-shoulder** | `over-shoulder from behind`, `seen past someone's shoulder` |\n| **Profile / side** | `profile side angle`, `walking across frame` |\n| **From behind** | `back to camera`, `three-quarter from behind` |\n| **Dutch tilt** | `dutch tilt` — tension scenes only |\n| **Through crowd** | `subject visible through gap in crowd`, `foreground heads out of focus` |\n\n**Batch rule:** no two adjacent stills share the same `camera_tag` **and** placement corner (e.g. don't do `left third` twice in a row).\n\nAvatar / lip-sync exception: face must stay readable and mouth visible — use `slight angle from the side` or `three-quarter`, still **off-center** and **off-lens gaze** when not delivering VO to camera.\n\n## Scene spice (when it fits)\n\nDefault plates are person + crowd + place. Add **one or two specific attributes** when the setting naturally supports them — not random clutter on every row.\n\n| Spice type | When to add | Example |\n|------------|-------------|---------|\n| **Animals** | setting implies them | dog park → `golden retriever on leash`; harbor → `seagulls overhead`; rooftop → `pigeons on water tower`; parade → `police horse midground` |\n| **Held / worn props** | role or weather | `red umbrella tucked under arm`, `wire beekeeper smoker`, `chipped ceramic mug`, `sample strawberry basket` |\n| **Micro-detail** | one thumb-stopping oddity | `muddy paw prints on pavement`, `honey jar on crate`, `green parade beads on fence` |\n\nCamera and placement live in [framing & camera](#framing--camera) — not optional spice.\n\n**Rule:** pick **at most two** spice items per prompt. They must answer “what would a photographer notice here?” — not a checklist dump.\n\n**Skip spice when:** product hero, avatar MC talking head, try-on full-body (garment is the focus), or minimal studio brief.\n\n## Photoreal anti-slop (neon / stylized briefs)\n\nStylized settings still need **documentary skin discipline** or outputs go waxy:\n\n- Lead with `documentary portrait, natural skin pores, not CGI, not illustration` even for neon/cyberpunk worlds.\n- Prefer **worn real materials** — matte leather, faded denim, scratched CRT bezels, sticky carpet — over `holographic puffer`, `chrome armor`, `HUD`.\n- Name **gritty location cues** — basement arcade, wet alley, scuffed linoleum — not abstract `neon corridor`.\n- Background crowd faces need **imperfect texture**; blur is fine, plastic skin in midground is not.\n\n## Aspect ratio (multi-example sets)\n\nWhen generating **two or more** stills in one session (playground grid, demo batch, mood board), give each independent output a **different** `aspect_ratio` unless the user locked a format.\n\n**Allowed `p-image` values:** `1:1` · `16:9` · `9:16` · `4:3` · `3:4` · `3:2` · `2:3`\n\n**How to pick:** after the [random seed ritual](./random-seed-ritual.md), use [sum-mod](#ssot-axis-derivation-sum-mod) on `ritual_seed` — state it in the turn (*\"Aspect ratio: 16:9\"*). Do **not** default every example to `9:16` or `1:1`.\n\n| Ratio | Typical use |\n|-------|-------------|\n| `9:16` | vertical UGC, full-body fashion, avatar talking head |\n| `16:9` | environmental wide, cinematic landscape plate |\n| `3:4` | editorial portrait, try-on full-body |\n| `4:3` | classic portrait, product + person |\n| `1:1` | packshot grid, social tile |\n| `3:2` · `2:3` | magazine / poster crops |\n\nMatch prompt framing to ratio (e.g. `16:9 horizontal wide shot`, `9:16 vertical full body`). **`p-image-try-on`** inherits plate size when `preserve_input_size: true` — diversify person plates first.\n\n**Same character arc:** one ratio for the whole chain unless the user asks for reframes.\n\n## By model (minimum diversity)\n\n| Model | Besides ritual seed, always vary |\n|-------|-----------------------------------|\n| **`p-image`** | cast/creature + objects + action + setting + camera + **`render_category_tag`** + **aspect_ratio**; [explicit structure](#explicit-prompt-structure-required); [text hygiene](#text--typography-by-model) (no upsampling) |\n| **`p-image-edit`** | edit tag + setting/angle delta; same identity URL |\n| **`p-image-try-on`** | person plate world + garment complexity; preserve scene |\n| **`p-image-upscale`** | N/A on prompt — diversify **source** stills |\n| **`p-video`** | motion/scene tag + `video_prompt`; differ start plates per scene |\n| **`p-video-avatar`** | `video_prompt` + still world per scene; lock voice per character |\n| **`p-video-animate`** | persona still style/setting per slider ref |\n| **`p-video-replace`** | video-edit tag + full cast spread on showcase reels |\n\n## When **not** to maximize diversity\n\n- **Same character arc** — lock hero plate URL, one `voice`, cast descriptor; vary only setting/angle/motion per scene.\n- **User asked for continuity** — match their cast and approved plates.\n- **Draft → final** — same prompt; change only `draft: false`. Use `api_seed` only if user locked API reproducibility.\n\n## Anti-patterns\n\n| Wrong | Right |\n|-------|--------|\n| Copy doc example ritual strings | [Random seed ritual](./random-seed-ritual.md) — fresh string each time |\n| Pass ritual string as API `seed` | Ritual is SSoT planning only; `api_seed` when user requests |\n| White wall + MC CU on every demo | Rotate setting + camera + cast |\n| One `video_prompt` for whole reel | Unique motion per scene row |\n| New ritual string mid avatar chain on same brief | Reuse `ritual_seed` until recast or new independent output |\n| Same aspect ratio on every playground example | Rotate `1:1` · `16:9` · `9:16` · `4:3` · `3:4` · `3:2` · `2:3` per [aspect ratio rules](#aspect-ratio-multi-example-sets) |\n| Every hero same athletic body | Rotate [body type spread](#body-type-spread) |\n| Generic hospital hallway | Named ER set dressing + mixed body types in crowd |\n| `holographic` / `chrome` on photoreal cyber scenes | Worn leather, scratched cabinets, documentary skin cues |\n| Monoculture crowd in a named global city | [Location-matched crowds](#location-matched-crowds) — hero ≠ background ethnicity |\n| Magenta-cyan neon on photoreal gym | Warm overhead studio light, mirror wall, real spin bikes |\n| All-male or all-female group class | [Group classes](#group-classes--courses) — mixed-gender background cues |\n| Centered subject every frame | [Framing & camera](#framing--camera) — rotate `camera_tag` + placement |\n| Subject facing camera / at viewer | Off-lens gaze, profile, from behind, or watching crowd |\n| Random animals with no setting reason | Animals only when place implies them |\n| Every stylized panel is anime | Rotate [render categories](#render-categories) — use `cartoon_anime_fantasy` at most once per batch |\n| Vague `cool portrait, neon vibes` | [Explicit structure](#explicit-prompt-structure-required) — named subject, action, objects, setting |\n| `no text` / `without signage` in prompt | Negation invokes text — use [text rules by model](#text--typography-by-model) |\n| Dense typography on **`p-image`** | Drop copy or simplify the brief — `p-image` has no prompt upsampling |\n\n## Related\n\n- [generation-quality-checklists.md](./generation-quality-checklists.md) — core + model checklists\n- [staged-generation-gate.md](https://github.com/PrunaAI/pruna-skills/tree/main/references/policies/staged-generation-gate.md) — approval phases\n\nFile v1.0.6:references/generation-quality-checklists.md\n\n# Generation quality checklist hub\n\nUse this as the shared quality gate across models and workflows.\nRun the **Core checklist** for every generation job, then run the model-specific checklist.\n\n## Who applies these checklists?\n\n**The coding agent** — by **opening the real output files** (images, video, or audio) and reviewing them with vision. These checklists are **not** automated test scripts. There is no separate scoring service: the agent reads each item and judges pass or fail from what it sees and hears.\n\nTypical flow:\n\n1. **Generate or download** the asset to a local path (`stills/`, `clips/`, etc.).\n2. **Inspect the file** — view the image, watch the video clip, or listen to narration when the checklist covers audio.\n3. Run the **Core checklist** (below), then the **model-specific checklist** for that job.\n4. **If something fails** — note which items failed, adjust prompt / settings / seed, and regenerate **only that asset** (do not advance to expensive video steps on a bad still).\n5. **If it passes** — show the user the file paths (and previews when helpful). In workflows, still follow [staged-generation-gate.md](https://github.com/PrunaAI/pruna-skills/tree/main/references/policies/staged-generation-gate.md): agent checklist review happens **before** you ask the user to approve stills or clips.\n\nThe user's **approve plan / approve stills / approve clips** gates are separate. Agent checklists catch obvious problems early so the user is not asked to sign off on broken outputs.\n\nMaintenance rule: keep tool/workflow mapping only in this file to avoid link drift.\n\n## Match map (tool -> checklist -> workflows)\n\n| Tool/model | Checklist | Common workflows |\n|------------|-----------|---------------|\n| `p-image` | [`p-image-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/image/p-image-quality-checklist.md) · persona bar: [`realistic-persona-showcase.md`](https://github.com/PrunaAI/pruna-skills/tree/main/shared/realistic-persona-showcase.md) | [`image-to-video`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/image-to-video/skills/image-to-video/SKILL.md), [`narrated-multi-scene`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/narrated-multi-scene/skills/narrated-multi-scene/SKILL.md), [`WORKFLOW-RECIPES`](https://github.com/PrunaAI/pruna-skills/tree/main/docs/WORKFLOW-RECIPES.md) |\n| `p-image-edit` | [`p-image-edit-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/image/p-image-edit-quality-checklist.md) | [`avatar-single-scene`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/avatar-single-scene/skills/avatar-single-scene/SKILL.md), [`avatar-multi-scene`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/avatar-multi-scene/skills/avatar-multi-scene/SKILL.md), [`WORKFLOW-RECIPES`](https://github.com/PrunaAI/pruna-skills/tree/main/docs/WORKFLOW-RECIPES.md) |\n| `p-image-upscale` | [`p-image-upscale-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/image/p-image-upscale-quality-checklist.md) | [`image-to-video`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/image-to-video/skills/image-to-video/SKILL.md), [`narrated-multi-scene`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/narrated-multi-scene/skills/narrated-multi-scene/SKILL.md), [`WORKFLOW-RECIPES`](https://github.com/PrunaAI/pruna-skills/tree/main/docs/WORKFLOW-RECIPES.md), [`generate_upscale_comparison.py`](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/_shared/scripts/generate_upscale_comparison.py) |\n| `p-image-try-on` | [`p-image-try-on-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/image/p-image-try-on-quality-checklist.md) | [`WORKFLOW-RECIPES`](https://github.com/PrunaAI/pruna-skills/tree/main/docs/WORKFLOW-RECIPES.md), [`p-image-try-on`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/p-image-try-on/skills/p-image-try-on/SKILL.md), [`realistic-persona-showcase.md`](https://github.com/PrunaAI/pruna-skills/tree/main/shared/realistic-persona-showcase.md) |\n| `p-video` | [`p-video-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/video/p-video-quality-checklist.md) | [`image-to-video`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/image-to-video/skills/image-to-video/SKILL.md), [`narrated-multi-scene`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/narrated-multi-scene/skills/narrated-multi-scene/SKILL.md), [`interactive-explainer`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/interactive-explainer/skills/interactive-explainer/SKILL.md), [`visual-transition-reel`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/visual-transition-reel/skills/visual-transition-reel/SKILL.md), [`WORKFLOW-RECIPES`](https://github.com/PrunaAI/pruna-skills/tree/main/docs/WORKFLOW-RECIPES.md) |\n| `p-video-avatar` | [`p-video-avatar-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/video/p-video-avatar-quality-checklist.md) · persona bar: [`realistic-persona-showcase.md`](https://github.com/PrunaAI/pruna-skills/tree/main/shared/realistic-persona-showcase.md) | [`avatar-single-scene`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/avatar-single-scene/skills/avatar-single-scene/SKILL.md), [`avatar-multi-scene`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/avatar-multi-scene/skills/avatar-multi-scene/SKILL.md), [`interactive-explainer`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/interactive-explainer/skills/interactive-explainer/SKILL.md), [`WORKFLOW-RECIPES`](https://github.com/PrunaAI/pruna-skills/tree/main/docs/WORKFLOW-RECIPES.md) |\n| `p-video-animate` | [`p-video-animate-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/video/p-video-animate-quality-checklist.md) | [`avatar-multi-scene`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/avatar-multi-scene/skills/avatar-multi-scene/SKILL.md), [`WORKFLOW-RECIPES`](https://github.com/PrunaAI/pruna-skills/tree/main/docs/WORKFLOW-RECIPES.md) |\n| `p-video-replace` | [`p-video-replace-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/video/p-video-replace-quality-checklist.md) | [`p-video-replace`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/p-video-replace/skills/p-video-replace/SKILL.md), [`generate_video_comparison.py`](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/_shared/scripts/generate_video_comparison.py) |\n| `music-2.5` + music video assembly | [`music-video-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/music-video-quality-checklist.md) | [`music-video`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/music-video/skills/music-video/SKILL.md), [`music-2.5`](../SKILL.md) |\n\n## Core checklist (all models)\n\n- **[Generation diversity](./generation-diversity.md)** — ritual seed + rotate scenario axes on **every** model (image, video, try-on, avatar, …).\n- **[Random seed ritual](./random-seed-ritual.md) (SSoT)** — generate and state a ritual string **before** every generation; derive prompt axes via sum-mod; never copy example strings from docs.\n- Goal and acceptance criteria are explicit (what \"good\" looks like is written down).\n- Input assets are valid and licensed (URL/file reachable, rights cleared).\n- Prompt and settings match the intended output format (`aspect_ratio`, duration, resolution, style lock). **Video default:** `720p`, `24` fps unless the brief asks for final `1080p` / `48`.\n- Output contains no accidental watermarks, UI overlays, or stray text unless requested.\n- Brand, legal, and safety constraints are satisfied before handoff.\n- Manifest/log captures model, input fields, prediction id, output URL, and **`ritual_seed`** for traceability.\n\n## Model-specific checklists\n\n- [`p-image-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/image/p-image-quality-checklist.md)\n- [`p-image-edit-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/image/p-image-edit-quality-checklist.md)\n- [`p-image-upscale-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/image/p-image-upscale-quality-checklist.md)\n- [`p-image-try-on-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/image/p-image-try-on-quality-checklist.md)\n- [`p-video-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/video/p-video-quality-checklist.md)\n- [`p-video-avatar-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/video/p-video-avatar-quality-checklist.md)\n- [`p-video-animate-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/video/p-video-animate-quality-checklist.md)\n- [`p-video-replace-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/video/p-video-replace-quality-checklist.md)\n- [`music-video-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/music-video-quality-checklist.md)\n\n## Visual variety (launch reels)\n\nBefore **any** generation, run [generation-diversity.md](./generation-diversity.md). Launch reels: also [visual-variety-bible.md](https://github.com/PrunaAI/pruna-skills/tree/main/shared/visual-variety-bible.md) **Variety checklist**. Persona/playground bar: [realistic-persona-showcase.md](https://github.com/PrunaAI/pruna-skills/tree/main/shared/realistic-persona-showcase.md).\n\nFor phased human review before expensive video jobs, see [staged-generation-gate.md](https://github.com/PrunaAI/pruna-skills/tree/main/references/policies/staged-generation-gate.md) and the per-skill index [workflow-feedback-gates.md](https://github.com/PrunaAI/pruna-skills/tree/main/references/policies/workflow-feedback-gates.md).\n\n## Workflow note\n\nFor multi-scene projects, run these checks per scene and add a final continuity pass\n(style, character identity, voice, and pacing consistency across scenes).\n\n**Narrated cinematic B-roll:** validate [scene anchor triple](https://github.com/PrunaAI/pruna-skills/tree/main/video/scene-anchor-triple.md) inputs before `p-video` — start still, end still, uploaded narration URL per row.\n\nFile v1.0.6:references/random-seed-ritual.md\n\n# Random seed ritual (SSoT — mandatory before every generation)\n\nThe random seed ritual is a lean [String Seed of Thought](https://pub.sakana.ai/ssot/) (DAG) protocol. **Every** Pruna generation — every prompt, every `POST /v1/predictions`, every scene row — starts here.\n\nThis prevents copy-pasting example strings (`k7Qm2xP9`, `482901`, …) and reduces accidental duplicate outputs across sessions.\n\n## The ritual (do this first)\n\nBefore writing prompts, curl, or runner JSON:\n\n1. **Generate a random string** in-agent (8–16 chars, mixed case + digits).\n2. **Log it** as `ritual_seed` in the manifest / internal plan. Do **not** require a user-visible *\"Ritual seed: …\"* line unless the user asks for transparency.\n3. **Derive prompt choices** from the string — sum char codes, mod N — pick axes from [generation-diversity.md](./generation-diversity.md) (`aspect_ratio`, `camera_tag`, `render_category_tag`, …).\n4. **Write the prompt** using [explicit prompt structure](./generation-diversity.md#explicit-prompt-structure-required) and derived axes.\n5. **Record** axes chosen and prediction id in the manifest alongside `ritual_seed`.\n\n**Do not pass the ritual string to API `seed`.** API runs without `seed` unless the user explicitly requests reproducibility (`api_seed`).\n\n**Never** proceed to `POST /v1/predictions` without completing steps 1–2 (unless the user supplied an explicit `api_seed` — see below).\n\n## Reuse rules\n\n| Situation | Action |\n|-----------|--------|\n| **New hero / independent still / mood-board panel** | Fresh ritual string |\n| **Same-brief slop retry** | Reuse same `ritual_seed`; note `retry_ritual_seed` in manifest |\n| **Same character arc** | Lock **hero plate URL** + cast descriptor; reuse `ritual_seed` only on same-brief regen |\n| **User says \"lock seed\" / provides integer** | Pass **their** number as `api_seed` → `input.seed`; skip new ritual for that chain |\n\nCharacter continuity = approved plate URL + cast descriptor — **not** the ritual string on the API.\n\n## Anti-patterns\n\n| Wrong | Right |\n|-------|--------|\n| Copy example strings from SKILL.md | Fresh ritual string each independent generation |\n| Pass ritual string as API `seed` | Ritual is planning-only; `api_seed` only when user asks |\n| One ritual string for entire mood board | New ritual per independent **`p-image`** |\n| Skip ritual because API `seed` is optional | Ritual always; API omits `seed` by default |\n\n## Example (internal plan / optional user-visible)\n\nManifest: `\"ritual_seed\": \"k7Qm2xP9\"`. Derived: aspect_ratio 16:9, camera_tag fish-eye, render_category_tag cartoon_anime_fantasy.  \nPrompt: Disco ball reflections on an otter DJ scratching vinyl at a packed 1970s roller rink, fish-eye lens, glitter confetti mid-air, funky energy.  \n…then curl / runner **without** `\"seed\"` in `input`.\n\n## Manifest snippet\n\n```json\n{\n  \"ritual_seed_policy\": \"ssot_dag_before_every_generation\",\n  \"ritual_seed\": \"k7Qm2xP9\",\n  \"seed_log\": [\n    { \"phase\": \"hero_p_image\", \"ritual_seed\": \"k7Qm2xP9\", \"creature_tag\": \"otter_dj\", \"setting_tag\": \"1970s_roller_rink\", \"prompt_hash\": \"…\" },\n    { \"phase\": \"scene_2_avatar\", \"ritual_seed\": \"k7Qm2xP9\", \"scene_id\": 2 }\n  ]\n}\n```\n\n## Where this applies\n\nAll Pruna generation skills and workflow runners — **every invocation**:\n\n- **`p-image`**, **`p-image-edit`**, **`p-image-try-on`**, **`p-image-upscale`**\n- **`p-video`**, **`p-video-avatar`**, **`p-video-animate`**, **`p-video-replace`**\n\n## Related\n\n- [generation-diversity.md](./generation-diversity.md) — ritual + axis rotation + sum-mod derivation\n- [realistic-persona-showcase.md](https://github.com/PrunaAI/pruna-skills/tree/main/shared/realistic-persona-showcase.md) — persona planning\n- [staged-generation-gate.md](https://github.com/PrunaAI/pruna-skills/tree/main/references/policies/staged-generation-gate.md) — approval phases\n- [approval-red-flags.md](https://github.com/PrunaAI/pruna-skills/tree/main/references/policies/approval-red-flags.md) — red flags\n\nFile v1.0.6:references/replicate-api.md\n\n# Replicate API (minimal)\n\nUsed by external tool skills (e.g. [stable-audio-2.5](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/stable-audio-2.5/skills/stable-audio-2.5/SKILL.md), [gemini-3.1-flash-tts](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/gemini-3.1-flash-tts/skills/gemini-3.1-flash-tts/SKILL.md)).\n\n**Missing token:** agents must stop and point the user to [api-credentials.md](./api-credentials.md) — sign up at [replicate.com/account/api-tokens](https://replicate.com/account/api-tokens) ([sign in](https://replicate.com/signin) if needed), then `export REPLICATE_API_TOKEN=r8_...`.\n\n## Auth\n\n```bash\nexport REPLICATE_API_TOKEN=r8_...\n```\n\nHeader: `Authorization: Bearer ${REPLICATE_API_TOKEN}`\n\n## Create + poll\n\n```bash\n# POST https://api.replicate.com/v1/models/{owner}/{name}/predictions\n# Body: {\"input\": { ... }}\n\n# Poll GET on response.urls.get until status == succeeded\n# Download output URL (string or list depending on model)\n```\n\nShared client: [`workflows/_shared/scripts/replicate_api.py`](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/_shared/scripts/replicate_api.py)\n\n## Stable Audio 2.5\n\nModel: `stability-ai/stable-audio-2.5`  \nRequired input: `prompt`  \nOptional: `duration` (1–190), `steps` (4–8), `cfg_scale`, `seed`\n\n## Music 2.5 (MiniMax)\n\nModel: `minimax/music-2.5`  \nRequired input: `lyrics` (1–3,500 chars, structure tags supported)  \nOptional: `prompt` (style), `sample_rate`, `bitrate`, `audio_format` (`mp3` default)\n\nWorkflow: [music-video](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/music-video/skills/music-video/SKILL.md) · tool skill: [music-2.5](../SKILL.md)\n\n## Gemini 3.1 Flash TTS\n\nModel: `google/gemini-3.1-flash-tts`  \nRequired input: `text`  \nOptional: `voice` (default `Kore`), `prompt` (style/scene), `language_code` (default `en-US`)\n\nOutput: audio file URL. Use for narration — upload to Pruna as part of [scene anchor triple](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/scene-anchor-triple/SKILL.md) (`input.audio` + `input.image` + `input.last_frame_image` on `p-video`). Layering with beds: [audio-post-production.md](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/audio-post-production/SKILL.md)\n\nFile v1.0.6:README-INSTALL.md\n\n# music-2.5\n\n## Install\n\n**Skills CLI** (copy-paste):\n\n```bash\nnpx skills add PrunaAI/pruna-skills@music-2.5 -y\n```\n\n**Plugins CLI** (workflow bundles with deps — pick from the list):\n\n```bash\nnpx plugins add PrunaAI/pruna-skills\n# when prompted, select: music-2.5\n```\n\nDo **not** use `npx plugins add PrunaAI/pruna-skills@music-2.5` — the plugins CLI has no `@name` filter (that’s skills only) and prints “No plugins found”.\n\n**Claude Code:**\n\n```text\n/plugin marketplace add PrunaAI/pruna-skills\n/plugin install music-2.5@pruna-skills\n```\n\nList all skills:\n\n```bash\nnpx skills add PrunaAI/pruna-skills -l\n```\n\nAfter install, start a **new chat**. See the [root README](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/README/SKILL.md).\n\n## From a local clone\n\n```bash\nnpx skills add .@music-2.5 -y\n# or:\nnpx skills add ./plugins/music-2.5/skills --skill music-2.5 -y\n```\n\nFile v1.0.6:skill-card.md\n\n## Description: <br>\nUse when someone wants an original AI song with vocals - sung lyrics, a style prompt track, or source audio for a music video. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[pruna-ai](https://clawhub.ai/user/pruna-ai) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and creative workflow users use this skill to generate full-length songs with natural vocals from lyrics and style prompts, typically as source audio for music videos or related media workflows. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Lyrics, style prompts, and generated outputs are sent to Replicate and MiniMax. <br>\nMitigation: Do not use confidential, regulated, or rights-sensitive lyrics unless permission exists and the user accepts the provider data handling terms. <br>\nRisk: The skill requires paid or credentialed third-party generation through Replicate. <br>\nMitigation: Check that REPLICATE_API_TOKEN is present before creating predictions, and stop with setup guidance if the token is missing. <br>\nRisk: Generated music quality, pronunciation, and arrangement can vary between runs. <br>\nMitigation: Use the documented seed ritual, diversity guidance, and quality checklists before advancing generated audio into downstream media workflows. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/pruna-ai/skills/music-2-5) <br>\n- [Music 2.5 model page](https://replicate.com/minimax/music-2.5) <br>\n- [Replicate API reference](references/replicate-api.md) <br>\n- [API credentials](references/api-credentials.md) <br>\n- [Random seed ritual](references/random-seed-ritual.md) <br>\n- [Generation diversity](references/generation-diversity.md) <br>\n- [Generation quality checklist hub](references/generation-quality-checklists.md) <br>\n- [MiniMax privacy policy](https://www.minimax.io/platform/protocol/privacy-policy) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [Guidance, Markdown, Shell commands, Configuration, API calls] <br>\n**Output Format:** [Markdown guidance with bash and curl examples for Replicate song generation] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Guides agents to produce prompts, lyrics, API requests, polling steps, and downloaded audio output URLs; generated audio is typically MP3, with WAV and PCM options.] <br>\n\n## Skill Version(s): <br>\n1.0.6 (source: server release metadata and skill frontmatter) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.0.6:skill.manifest.json\n\n{\n  \"scripts\": {\n    \"core\": [],\n    \"shared\": []\n  },\n  \"references\": [\n    \"replicate-api.md\",\n    \"api-credentials.md\"\n  ]\n}\n\nArchive v1.0.2: 9 files, 19606 bytes\n\nFiles: README-INSTALL.md (268b), references/api-credentials.md (3138b), references/generation-diversity.md (26090b), references/random-seed-ritual.md (4088b), references/replicate-api.md (2261b), skill-card.md (2383b), skill.manifest.json (188b), SKILL.md (5016b), _meta.json (128b)\n\nFile v1.0.2:SKILL.md\n\n---\nname: music-2.5\ndescription: Use when the user wants AI song generation with vocals, sung lyrics, original tracks from a style prompt, or source audio for a music video.\nlicense: MIT\nmetadata:\n  version: \"1.0.2\"\n  provider: replicate\n  replicate_model: minimax/music-2.5\n---\n\n# Music 2.5 (MiniMax · Replicate)\n\nFull-length **songs with natural vocals** from lyrics + style description. Not a Pruna P-model — runs on [Replicate](https://replicate.com/minimax/music-2.5).\n\n**Primary workflow:** [music-video](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/music-video/skills/music-video/SKILL.md) — lyrics → song → lyric-safe cuts → `p-video-avatar` / `p-video` clips → assembly.\n\n## When to use\n\n| Goal | Use this |\n|------|----------|\n| Sung track for a music video | Yes — write lyrics with section tags first |\n| Drive **`p-video`** clip length | Yes — export MP3 → upload to Pruna → `audio` input |\n| Documentary narration | No — use [gemini-3.1-flash-tts](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/gemini-3.1-flash-tts/skills/gemini-3.1-flash-tts/SKILL.md) |\n| Instrumental bed under VO | No — use [stable-audio-2.5](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/stable-audio-2.5/skills/stable-audio-2.5/SKILL.md) |\n\n## Environment\n\n```bash\nexport REPLICATE_API_TOKEN=r8_...\n```\n\nRequires **`ffmpeg`** / **`ffprobe`** for slicing and assembly in the music-video workflow.\n\n## Model input (Replicate)\n\n| Field | Notes |\n|-------|-------|\n| `lyrics` | **Required.** 1–3,500 characters. Use [structure tags](#structure-tags) and `\\n` line breaks. |\n| `prompt` | Optional style string — genre, mood, tempo, vocal timbre, key instruments (up to ~2,000 chars). |\n| `sample_rate` | `16000` · `24000` · `32000` · **`44100`** (default) |\n| `bitrate` | `32000` · `64000` · `128000` · **`256000`** (default) |\n| `audio_format` | **`mp3`** (default) · `wav` · `pcm` |\n\nOutput: audio file URL (typically **MP3**, ~2:30–4:30 for full songs).\n\n## Structure tags\n\nControl arrangement with tags on their own lines (see [Music 2.5 readme](https://replicate.com/minimax/music-2.5)):\n\n`[Intro]` · `[Verse]` · `[Pre Chorus]` · `[Chorus]` · `[Hook]` · `[Bridge]` · `[Solo]` · `[Inst]` · `[Build Up]` · `[Drop]` · `[Interlude]` · `[Break]` · `[Transition]` · `[Outro]`\n\n- One tag per section; follow with 2–4 lyric lines per section for clean melodies.\n- `\\n` = line break (also a **safe video cut boundary** in the music-video workflow).\n- `\\n\\n` = pause between sections.\n- Parentheticals work for ad-libs and directions: `(Ooh, yeah)` · `(Guitar solo — slow, bluesy)`.\n\n## Prompt tips\n\nFollow the model’s [prompt guide](https://replicate.com/minimax/music-2.5): genre + mood + vocal description + tempo + instruments + production feel.\n\n**Example prompt:**\n\n```text\nIndie pop, uplifting, warm female vocal, 92 BPM, acoustic guitar and soft synth pads,\nwide soundstage, crisp modern production, anthemic chorus\n```\n\n**Instrumental sections:** use `[Inst]` or `[Solo]` tags with parenthetical instrument directions instead of sung lines.\n\n## HTTP (curl)\n\n```bash\ncurl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"lyrics\": \"[Verse]\\nWe built it line by line\\nEvery skill a stepping stone\\n\\n[Chorus]\\nRun the pipeline, watch it grow\\nPruna models, let them flow\",\n      \"prompt\": \"Indie pop, uplifting, warm female vocal, 92 BPM, acoustic guitar and mellow synth pads, no harsh distortion\",\n      \"sample_rate\": 44100,\n      \"bitrate\": 256000,\n      \"audio_format\": \"mp3\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/minimax/music-2.5/predictions\"\n```\n\nPoll `urls.get` until `status` is `succeeded`; download `output`.\n\n## Repo helper\n\n```bash\npython3 workflows/verticals/music-video/scripts/generate_song.py \\\n  --plan output/my-music-video/music_video_plan.json \\\n  --out-dir output/my-music-video\n```\n\n## Good to know\n\n- **English and Mandarin** have strongest pronunciation; other languages vary.\n- Each generation is unique — same lyrics + prompt produce different arrangements.\n- Max ~5 minutes per generation.\n- Data is sent to MiniMax via Replicate — see their [privacy policy](https://www.minimax.io/platform/protocol/privacy-policy).\n\n## Related\n\n- [audio-post-production.md](https://github.com/PrunaAI/pruna-skills/tree/main/references/shared/audio-post-production.md) — when to use songs vs narration vs beds\n- [music-video workflow](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/music-video/skills/music-video/SKILL.md)\n- [gemini-3.1-flash-tts](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/gemini-3.1-flash-tts/skills/gemini-3.1-flash-tts/SKILL.md) — spoken narration (not song)\n- [stable-audio-2.5](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/stable-audio-2.5/skills/stable-audio-2.5/SKILL.md) — instrumental beds only\n- [replicate-api.md](./references/replicate-api.md)\n\nFile v1.0.2:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"music-2-5\",\n  \"version\": \"1.0.2\",\n  \"publishedAt\": 1784208326386\n}\n\nFile v1.0.2:references/api-credentials.md\n\n# API credentials (Pruna + Replicate)\n\n**Agent rule:** Before any `POST /v1/predictions`, Replicate prediction, or paid runner — check env vars. If a required key is **missing or empty**, **stop** and tell the user how to sign up. Do not guess, mock, or skip with placeholder keys.\n\n## Pruna P-API\n\n| | |\n|--|--|\n| **Env var** | `PRUNA_API_KEY` |\n| **Header** | `apikey: ${PRUNA_API_KEY}` (not `Authorization: Bearer`) |\n| **Sign up / get key** | [Pruna dashboard](https://dashboard.pruna.ai/) |\n| **Docs** | [Quickstart](https://docs.api.pruna.ai/guides/quickstart) · [pruna-api.md](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/core/pruna-api/SKILL.md) |\n\n**Used by:** all `p-image*`, `p-video*` tool skills and Pruna workflow runners.\n\n### If `PRUNA_API_KEY` is missing — agent message template\n\n> Pruna generation needs an API key. Sign up or sign in at **[dashboard.pruna.ai](https://dashboard.pruna.ai/)**, create an API key, then set:\n>\n> ```bash\n> export PRUNA_API_KEY=\"your_key_here\"\n> ```\n>\n> Add that to your shell profile or project `.env` (never commit the key). Reply when it’s set and we can continue.\n\n## Replicate\n\n| | |\n|--|--|\n| **Env var** | `REPLICATE_API_TOKEN` |\n| **Header** | `Authorization: Bearer ${REPLICATE_API_TOKEN}` |\n| **Sign up / get token** | [Replicate API tokens](https://replicate.com/account/api-tokens) ([sign in](https://replicate.com/signin) first if needed) |\n| **Docs** | [replicate-api.md](./replicate-api.md) |\n\n**Used by:** `music-2.5`, `gemini-3.1-flash-tts`, `stable-audio-2.5`, `whisperx`, and workflow beds/TTS/song phases.\n\n### If `REPLICATE_API_TOKEN` is missing — agent message template\n\n> This step uses Replicate (song, TTS, transcription, or background bed). Create a token at **[replicate.com/account/api-tokens](https://replicate.com/account/api-tokens)**, then set:\n>\n> ```bash\n> export REPLICATE_API_TOKEN=\"r8_...\"\n> ```\n>\n> Reply when it’s set and we can continue.\n\n## Which key does this job need?\n\n| Task | Keys required |\n|------|----------------|\n| `p-image`, `p-image-edit`, `p-image-upscale`, `p-image-try-on` | `PRUNA_API_KEY` |\n| `p-video`, `p-video-avatar`, `p-video-animate`, `p-video-replace` | `PRUNA_API_KEY` |\n| Music 2.5 song generation | `REPLICATE_API_TOKEN` |\n| Gemini TTS narration | `REPLICATE_API_TOKEN` |\n| Stable Audio background bed | `REPLICATE_API_TOKEN` |\n| WhisperX transcription | `REPLICATE_API_TOKEN` |\n| Music video / explainer (full pipeline) | **Both** — Pruna for stills/video; Replicate for song/TTS/bed as needed |\n\nWhen only one key is missing, suggest **only** that provider’s signup link — not both.\n\n## Security\n\n- Never print full keys in chat or commit them to git.\n- `.env` is gitignored; prefer env vars over hardcoding in plans or manifests.\n- Never embed keys in prompts, manifests, plan JSON, logs, or **subagent task text**.\n- Prefer the **parent agent** to own API calls; do not fan credentials across parallel subagents unless the host documents isolated secret injection.\n- Full rules: [agent-safety.md](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/core/agent-safety/SKILL.md).\n\nFile v1.0.2:references/generation-diversity.md\n\n# Generation diversity (all models)\n\nOne checklist so **every** Pruna output — **`p-image`**, **`p-video`**, try-on, avatar, replace, animate — is as **diverse** as the brief allows. Details live in linked docs; this page is the agent shortcut.\n\nUse the **full** checklist here for every generation.\n\n## Contents\n\n- [Three steps (every job)](#three-steps-every-job)\n- [Explicit prompt structure](#explicit-prompt-structure-required)\n- [Text & typography by model](#text--typography-by-model)\n- [SSoT axis derivation](#ssot-axis-derivation-sum-mod)\n- [Scenario axes](#scenario-axes-rotate-across-outputs)\n- [Render categories](#render-categories)\n- [Crowded scenes](#crowded-scenes-p-image)\n- [Body type spread](#body-type-spread)\n- [Location-matched crowds](#location-matched-crowds)\n- [Group classes](#group-classes--courses)\n- [Framing & camera](#framing--camera)\n- [Scene spice](#scene-spice-when-it-fits)\n- [Photoreal anti-slop](#photoreal-anti-slop-neon--stylized-briefs)\n- [Aspect ratio](#aspect-ratio-multi-example-sets)\n- [By model](#by-model-minimum-diversity)\n- [When not to maximize diversity](#when-not-to-maximize-diversity)\n- [Anti-patterns](#anti-patterns)\n\n## Three steps (every job)\n\n1. **[Random seed ritual](./random-seed-ritual.md) (SSoT)** — **always first**, before the prompt. Generate a fresh random string, **state it in the turn**, derive axes via [sum-mod](#ssot-axis-derivation-sum-mod). **Do not** pass the ritual string to API `seed`. **One new ritual string per independent generation**; reuse only on same-brief slop retry.\n2. **Write an [explicit prompt](#explicit-prompt-structure-required)** — name specific people, animals, objects, actions, setting, and camera/light. Add text/typography only when the brief needs it — see [text rules by model](#text--typography-by-model).\n3. **Diversify the scenario row** — change at least **two axes** from the previous output in the same session (cast, setting, camera, **`render_category_tag`**, **aspect_ratio**, creatures, props, … — unless user asked for continuity).\n4. **Log** — `ritual_seed`, axes chosen, prediction id (manifest or turn text).\n\n## Explicit prompt structure (required)\n\n**Vague prompts produce generic AI slop.** After the ritual and axis picks, every still prompt must be **specific and dynamic** — concrete nouns, frozen actions, named places. Prefer playground/creative briefs over marketing abstractions.\n\n**Name at least four of these per prompt (log tags in manifest):**\n\n| Clause | Log as | Agent must specify |\n|--------|--------|-------------------|\n| **People** | `cast_descriptor` | Named role + age band + expression (`fearless grandmother in floral apron`, not `woman`) |\n| **Animals / creatures** | `creature_tag` | Species + attitude (`otter DJ`, `luna moth knight`, `VIP anglerfish`) |\n| **Objects** | `prop_tag` | Concrete props (`vinyl record`, `chrome rocket sled`, `velvet rope`, `tiny boombox`) |\n| **Action** | `action_tag` | Frozen mid-motion verb (`scratching vinyl`, `lassoing runaway taco truck`, `cape mid-swing`) |\n| **Duration** | `duration_tag` | When timing matters (`1970s`, `8PM`, `45-minute spin class`, `Saturday-morning cartoon`) |\n| **Setting** | `setting_tag` | Named place + era + materials (`packed 1970s roller rink`, `abyss-depth jellyfish nightclub`, `Monument Valley dust storm`) |\n| **Text / typography** | `text_spec` | Only when brief needs readable type — exact strings + surface (see [by model](#text--typography-by-model)) |\n| **Camera + light** | `camera_tag`, `lighting_tag` | `fish-eye lens`, `tilt-shift macro`, `teal-magenta cinematic`, `golden hour sparkle` |\n| **Style** | `render_category_tag` | Medium (`cel-shaded anime`, `baroque oil painting`, `ink-wash storybook`, `photoreal documentary`) |\n\n**Template:**\n\n```text\n{people and/or creatures} {action} with/at {specific objects} in {named setting},\n{style or era cues}, {camera_tag}, {lighting_tag}\n```\n\n**Good examples (dynamic / specific):**\n\n```text\nDisco ball reflections on an otter DJ scratching vinyl at a packed 1970s roller rink,\nfish-eye lens, glitter confetti mid-air, funky energy\n```\n\n```text\nBioluminescent jellyfish nightclub at abyss depth, VIP anglerfish in sunglasses at velvet rope,\nteal-magenta cinematic lighting\n```\n\n```text\nCorgi cowboy lassoing a runaway taco truck through Monument Valley dust storm,\npulp western poster energy, dynamic diagonal composition\n```\n\n**Anti-pattern:** `cool cyberpunk portrait, neon vibes` — no subject, no action, no place. **Right:** name who, what they're doing, where, with which props.\n\n## Text & typography by model\n\n**Never use negation to suppress text** — `no text`, `without signs`, `no typography` often **invoke** the thing you are trying to avoid. Describe surfaces positively when you want blank walls (`plain unmarked walls`, `matte unprinted props`).\n\n| Model | Prompt upsampling | Typography in prompt |\n|-------|-------------------|----------------------|\n| **`p-image`** | **No** effective prompt upsampling | **Avoid** dense readable-type requests unless user explicitly wants `text_rendering`. Short prompts; skip `readable`, `legible`, `headline`, multi-sign lists — they drift to gibberish. Collage triggers still apply: [interactive-explainer-prompts.md](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/interactive-explainer-prompts.md) (`flat lay`, `grid`, `collage`, …). |\n\n**`p-image` text hygiene:** prefer scenes without copy. If a screen appears: `monitor soft colorful blur glow only` — not legible UI unless the user explicitly asked for readable text (then simplify the brief or drop copy).\n\n**Collage triggers (all T2I models):** still avoid `flat lay`, `packshot`, `grid`, `collage`, `montage`, `contact sheet`, `split`, `before and after` — use `single frame`, `one camera angle` instead. Full table: [interactive-explainer-prompts.md](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/interactive-explainer-prompts.md).\n\n## SSoT axis derivation (sum-mod)\n\nAfter stating `ritual_seed` (random string), derive prompt choices — sum Unicode/ASCII char codes, mod list length:\n\n```text\nRATIOS = [\"1:1\", \"16:9\", \"9:16\", \"4:3\", \"3:4\", \"3:2\", \"2:3\"]\naspect_ratio  ← RATIOS[ sum(codes(ritual_seed)) % 7 ]\ncamera_tag    ← camera_tags[ sum(codes(ritual_seed[0:4])) % len(camera_tags) ]\nrender_tag    ← render_tags[ sum(codes(ritual_seed[4:8])) % len(render_tags) ]\n```\n\n`camera_tags` and `render_tags` — see [framing & camera](#framing--camera) and [render categories](#render-categories). State derived picks in the turn (*\"Aspect ratio: 16:9, camera: over-shoulder\"*).\n\n**User `api_seed`:** when the user supplies an integer for reproducibility, pass it as `input.seed` — separate from the ritual string.\n\n## Scenario axes (rotate across outputs)\n\n| Axis | Vary with | Applies to |\n|------|-----------|------------|\n| **Cast** | age, ethnicity, gender, archetype, **hairstyle**, **body type** (rotate — see [below](#body-type-spread)), disability aids (wheelchair, cane), visible age band twice in prompt | all person/content gens |\n| **Medium** | `render_category_tag` — rotate across [render categories](#render-categories) | `p-image`, avatar stills |\n| **Setting** | unique `setting_tag` — specific room/street/venue/era, not repeat adjacent rows | stills + video plates |\n| **Camera** | `camera_tag` — rotate across [framing ladder](#framing--camera); never default MC facing lens | stills, `video_prompt` |\n| **Lighting** | `lighting_tag` — golden hour · neon · overcast · practical | stills, video mood |\n| **Motion** | unique `video_prompt` per clip | `p-video`, `p-video-avatar`, animate |\n| **Voice** | natural `voice_script`; one `voice` preset per character | avatar, TTS-led video |\n| **Seed** | new ritual string per **independent** job; reuse only on same-brief slop retry | all generation skills |\n| **Aspect ratio** | different `aspect_ratio` per independent still in a batch — see [below](#aspect-ratio-multi-example-sets) | `p-image`, `p-image-edit` |\n| **Crowd density** | layered background population + activity cues — see [below](#crowded-scenes-p-image) | `p-image` plates with busy worlds |\n\nFull style/camera/lighting ladders: [visual-variety-bible.md](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/core/visual-variety-bible/SKILL.md). Persona + try-on bar: [realistic-persona-showcase.md](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/core/realistic-persona-showcase/SKILL.md).\n\n## Render categories\n\nRotate **`render_category_tag`** (and log it) so diversity batches cover more than photoreal portraits or anime. Category families below mirror arena leaderboards — pick a **different tag per independent output**.\n\n**Random seed ritual still applies** to every generation in [step 1](#three-steps-every-job); categories describe *what* to vary, not *when* to pick `seed`.\n\n### Text-to-image — `p-image`\n\nSources: [Arena text-to-image](https://arena.ai/leaderboard/text-to-image) · [AA text-to-image](https://artificialanalysis.ai/image/leaderboard/text-to-image)\n\n**Unified `render_category_tag`** (Arena bucket = tag — pick one per still):\n\n`product_branding_commercial` · `3d_imaging_modeling` · `cartoon_anime_fantasy` · `photoreal_cinematic` · `art` · `portraits` · `nature_environment` · `animals_creature` · `text_rendering`\n\n| Tag | Typical prompt lane |\n|-----|---------------------|\n| `product_branding_commercial` | single product on seamless studio, person + product in named setting, showroom (not `flat lay` / `packshot` words) |\n| `3d_imaging_modeling` | CG film still, clay/stop-motion, rounded 3D forms |\n| `cartoon_anime_fantasy` | cel anime, fantasy character, crowded stylized world |\n| `photoreal_cinematic` | documentary crowd scenes, film-scale wide, urban march |\n| `art` | oil, watercolor, gouache, charcoal, flat vector |\n| `portraits` | single-subject editorial or documentary portrait (crowd optional behind) |\n| `nature_environment` | landscape-wide; subject small in frame |\n| `animals_creature` | named species + handler; crowded market/park when it fits |\n| `text_rendering` | **user-requested only** — otherwise no readable text |\n\nLog `render_category_tag` in manifest. Combine with [crowded scenes](#crowded-scenes-p-image), [body type](#body-type-spread), and [scene spice](#scene-spice-when-it-fits) when the brief allows.\n\n### Image edit — `p-image-edit`\n\nSources: [Arena image edit](https://arena.ai/leaderboard/image-edit) · [AA image editing](https://artificialanalysis.ai/image/leaderboard/editing)\n\nArena modalities: `single_image_edit` · `multi_image_edit`\n\nEdit diversity tags: `background_swap` · `relight` · `wardrobe_on_plate` · `pose_or_angle_delta` · `multi_ref_composite` · `region_inpaint`\n\nVary **instruction** and **what changes** while identity URL stays fixed on character arcs.\n\n### Text-to-video — `p-video`\n\nSources: [Arena text-to-video](https://arena.ai/leaderboard/text-to-video) · [AA text-to-video](https://artificialanalysis.ai/video/leaderboard/text-to-video)\n\nMotion/scene tags: `character_performance` · `landscape_broll` · `urban_street` · `product_demo` · `abstract_mood` · `crowd_scene` · `dialogue_beat`\n\nRotate `video_prompt` grammar, start plate world, and `camera_tag` per clip.\n\n### Image-to-video — `p-video` (+ plate upload)\n\nSources: [Arena image-to-video](https://arena.ai/leaderboard/image-to-video) · [AA image-to-video](https://artificialanalysis.ai/video/leaderboard/image-to-video)\n\nPlate-driven tags: `animate_hero_still` · `camera_move_on_plate` · `environmental_parallax` · `avatar_lip_sync` · `hands_or_prop_motion`\n\nMatch motion to what the **still** already shows — do not contradict the plate.\n\n### Video edit — `p-video-replace` (and edit-style video)\n\nSource: [Arena video edit](https://arena.ai/leaderboard/video-edit)\n\nEdit tags: `face_recast` · `wardrobe_swap` · `accessory_swap` · `background_replace` · `object_in_hand_swap` · `style_transfer_on_subject`\n\nSame-gender / identity rules for talking-head beats still apply — see [visual-variety-bible.md](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/core/visual-variety-bible/SKILL.md).\n\n## Crowded scenes (`p-image`)\n\nWhen the brief asks for **busy**, **crowded**, or **lively** worlds — not a lone subject on a blank wall — stack density in the prompt:\n\n1. **Three depth layers** — sharp foreground subject · readable midground faces/hands/props · landmark bokeh (stage, temple, billboards, ferris wheel).\n2. **Named population count** — `hundreds of pedestrians`, `dozens of faces in midground`, `20+ tiny clay figures` (stylized sets need explicit counts; models under-deliver on vague \"busy\").\n3. **Activity verbs** — raised hands, umbrellas open, food steam, confetti, market haggling, commuters pressed shoulder-to-shoulder.\n4. **Shallow DOF + single subject** — `single subject one frame` keeps one identity readable while the crowd stays behind them.\n5. **Age & angle lock** — repeat age band twice (`woman in her late 50s, visibly fifty`) and use [framing & camera](#framing--camera) — models drift younger, center-frame, and front-facing without it.\n\n| Crowd family | Density cues |\n|--------------|--------------|\n| **Urban rush** | crosswalk stripes, wet reflections, umbrellas, billboard bokeh |\n| **Festival / parade** | confetti, raised hands, costume layers, smoke haze |\n| **Market / bazaar** | overflowing stalls, hanging goods, steam, price tags as color blobs |\n| **Transit crush** | strap hangers, door windows, blurred faces pressed together |\n| **Stylized miniature** | counted clay/figurine shoppers (`20+`), cramped aisle, stacked crates |\n| **Institutional / ER** | framed oil portraits on beige walls, triage number board, wall sanitizer, vending machine, scuffed linoleum, TV blur, mixed-age seated patients |\n| **Urban march / protest** | named city, local landmarks, multiracial crowd cues separate from hero — see [location-matched crowds](#location-matched-crowds) |\n| **Group fitness class** | class name + duration, mixed-gender riders, realistic warm studio light — see [group classes](#group-classes--courses) |\n\n**Anti-pattern:** one blurred smear behind a portrait — name **what** the crowd is doing and **where** layers sit. **Institutional** scenes (ER, airport, classroom) need `benches full`, `standing room only`, or `shoulder-to-shoulder` — otherwise models default to a quiet hallway. Name **set dressing** too: framed portraits on walls, triage number board, vending machine glow, scuffed linoleum — generic mint corridors read AI-empty.\n\n## Body type spread\n\nModels default to one “average fitness” body. In diversity batches, **name build on the hero and vary background bodies**:\n\n| Build tag | Prompt cue |\n|-----------|------------|\n| **Plus-size / curvy** | `plus-size`, `curvy build`, `full-figured` |\n| **Athletic / muscular** | `broad shoulders`, `muscular arms`, `athletic build` |\n| **Petite / slim** | `petite frame`, `slim build`, `narrow shoulders` |\n| **Tall / lanky** | `tall and lanky`, `6-foot frame`, `long limbs` |\n| **Stocky / heavyset** | `stocky build`, `heavyset`, `barrel chest` |\n| **Lean wiry** | `lean wiry frame`, `weathered thin face` |\n\n**Rule:** rotate build across independent panels in a session — not every hero “athletic build”. Background crowd should mix ages **and** silhouettes (`elderly thin woman`, `heavyset man`, `pregnant woman seated`, `toddler on lap`).\n\n## Location-matched crowds\n\nWhen the prompt names a **real city or country**, background faces must match that place’s **demographic mix** — not clone the hero’s ethnicity.\n\n| Wrong | Right |\n|-------|--------|\n| South Asian hero + only South Asian protesters in “New York” | Hero is one identity; crowd explicitly `multiracial NYC march — Black, Latino, white, East Asian protesters` |\n| “Dense city march” with no geography | Name city + 3–4 crowd ethnicity cues + local landmarks (yellow cabs, art deco towers, steam vent) |\n| Festival in Lagos with only Nordic faces | Match crowd to `setting_tag` region |\n\n**Prompt pattern:** lock hero cast in sentence 1; sentence 2 lists **four+ distinct background silhouettes** unrelated to hero ethnicity; sentence 3 names **local landmarks** so the plate cannot read as generic stock.\n\n**Applies to:** protests, airports, transit, street markets, sports crowds — any scene where “crowded” implies a real place.\n\n## Group classes & courses\n\nWhen the scene is a **class, workshop, or team activity**, name the **course type** and **who else is in the room** — models default to monochrome crowds (all men, all one age).\n\n| Specify | Example cues |\n|---------|----------------|\n| **Class type** | `45-minute evening spin class`, `beginner yoga flow`, `HIIT bootcamp circuit` |\n| **Room realism** | warm overhead track lights, mirror wall, rubber floor, water bottles, towels — **not** magenta-cyan neon strips unless brief is explicitly nightclub |\n| **Gender mix** | hero is one person; crowd `mixed-gender class — women with ponytails, men with beards, nonbinary cyclist` |\n| **Body + age mix** | plus-size rider, petite woman, athletic man, woman in her 50s — same as [body type spread](#body-type-spread) |\n\n**Lighting rule for fitness:** real boutique studios are **dim warm overhead** or **single spotlight on instructor** — avoid `split gel`, `neon LED strips`, `magenta-cyan` on photoreal gym plates; those read AI-fake.\n\n**Prompt pattern:** `Documentary fitness portrait` + class name + instructor on bike at front + `20+ mixed-gender cyclists` with 3–4 named background silhouettes + realistic room props.\n\n## Framing & camera\n\nModels default to **centered subject, eyes at camera**. In diversity batches, **rotate `camera_tag` and frame placement** every row — log both in manifest.\n\n**Gaze rule:** `glance off-lens`, `profile`, `back to camera`, `looking down at [prop]`, or `watching the crowd` — **not** `facing camera` or `looking at viewer` unless the user asked for a direct-address avatar plate.\n\n**Placement rule:** name where the subject sits in frame — `left third`, `right third`, `lower right corner`, `edge of frame`, `small in environmental wide` — **not** centered mugshot every time.\n\n| `camera_tag` | Prompt cue |\n|--------------|------------|\n| **Overhead / bird's eye** | `overhead aerial view`, `top-down`, `drone shot looking straight down` |\n| **High corner** | `high angle from corner`, `surveillance-style downward angle` |\n| **Worm's eye** | `ground-level worm's eye`, `camera on pavement` |\n| **Crane-down** | `slight high angle crane-down` |\n| **Over-shoulder** | `over-shoulder from behind`, `seen past someone's shoulder` |\n| **Profile / side** | `profile side angle`, `walking across frame` |\n| **From behind** | `back to camera`, `three-quarter from behind` |\n| **Dutch tilt** | `dutch tilt` — tension scenes only |\n| **Through crowd** | `subject visible through gap in crowd`, `foreground heads out of focus` |\n\n**Batch rule:** no two adjacent stills share the same `camera_tag` **and** placement corner (e.g. don't do `left third` twice in a row).\n\nAvatar / lip-sync exception: face must stay readable and mouth visible — use `slight angle from the side` or `three-quarter`, still **off-center** and **off-lens gaze** when not delivering VO to camera.\n\n## Scene spice (when it fits)\n\nDefault plates are person + crowd + place. Add **one or two specific attributes** when the setting naturally supports them — not random clutter on every row.\n\n| Spice type | When to add | Example |\n|------------|-------------|---------|\n| **Animals** | setting implies them | dog park → `golden retriever on leash`; harbor → `seagulls overhead`; rooftop → `pigeons on water tower`; parade → `police horse midground` |\n| **Held / worn props** | role or weather | `red umbrella tucked under arm`, `wire beekeeper smoker`, `chipped ceramic mug`, `sample strawberry basket` |\n| **Micro-detail** | one thumb-stopping oddity | `muddy paw prints on pavement`, `honey jar on crate`, `green parade beads on fence` |\n\nCamera and placement live in [framing & camera](#framing--camera) — not optional spice.\n\n**Rule:** pick **at most two** spice items per prompt. They must answer “what would a photographer notice here?” — not a checklist dump.\n\n**Skip spice when:** product hero, avatar MC talking head, try-on full-body (garment is the focus), or minimal studio brief.\n\n## Photoreal anti-slop (neon / stylized briefs)\n\nStylized settings still need **documentary skin discipline** o","readmeExcerpt":"Skill: music-2.5 Owner: pruna-ai Summary: Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. Tags: ai:1.0.14, generative:1.0.14, latest:1.0.14, pruna:1.0.14 Version history: v1.0.14 | 2026-09-29T15:36:02.415Z | auto - Version bump: updated metadata version to 1.0.14. - Removed skill-card.md file. - No changes to usage, features, or required/o","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"export REPLICATE_API_TOKEN=r8_..."},{"language":"bash","snippet":"curl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{"},{"language":"bash","snippet":"curl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"lyrics\": \"[Verse]\\nWe built it line by line\\nEvery skill a stepping stone\\n\\n[Chorus]\\nRun the pipeline, watch it grow\\nPruna models, let them flow\",\n      \"prompt\": \"Indie pop, uplifting, warm female vocal, 92 BPM, acoustic guitar and mellow synth pads, no harsh distortion\",\n      \"sample_rate\": 44100,\n      \"bitrate\": 256000,\n      \"audio_format\": \"mp3\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/minimax/music-2.5/predictions\""},{"language":"bash","snippet":"export REPLICATE_API_TOKEN=r8_..."},{"language":"bash","snippet":"curl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{"},{"language":"bash","snippet":"curl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"lyrics\": \"[Verse]\\nWe built it line by line\\nEvery skill a stepping stone\\n\\n[Chorus]\\nRun the pipeline, watch it grow\\nPruna models, let them flow\",\n      \"prompt\": \"Indie pop, uplifting, warm female vocal, 92 BPM, acoustic guitar and mellow synth pads, no harsh distortion\",\n      \"sample_rate\": 44100,\n      \"bitrate\": 256000,\n      \"audio_format\": \"mp3\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/minimax/music-2.5/predictions\""}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: music-2.5\ndescription: Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video.\nlicense: MIT\nmetadata:\n  version: \"1.0.14\"\n  package: pruna-skills\n  provider: replicate\n  replicate_model: minimax/music-2.5\n---\n\n## Prerequisites\n\nInstall and load these skills before generating (skip if already in context via `@pruna`):\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFollow each skill's **Before generating** / craft sections — do not restate guide content here.\n\n## Agent habit\n\nIn the **first reply**, name `` `music-2.5` `` in backticks, confirm `REPLICATE_API_TOKEN` (or stop with signup links from `pruna-api`), then ask for required inputs. Open intake → **`generation-diversity`** clarification intake before the first `POST`. Redirect when **When NOT to use** fits better.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `gemini-3.1-flash-tts` | Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. | `npx skills add PrunaAI/pruna-skills@gemini-3.1-flash-tts -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n\n## Environment\n\n```bash\nexport REPLICATE_API_TOKEN=r8_...\n```\n\nRequires **`ffmpeg`** / **`ffprobe`** for slicing and assembly in the music-video workflow.\n\n## HTTP (curl)\n\n```bash\ncurl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"lyrics\": \"[Verse]\\nWe built it line by line\\nEvery skill a stepping stone\\n\\n[Chorus]\\nRun the pipeline, watch it grow\\nPruna models, let them flow\",\n      \"prompt\": \"Indie pop, uplifting, warm female vocal, 92 BPM, acoustic guitar and mellow synth pads, no harsh distortion\",\n      \"sample_rate\": 44100,\n      \"bitrate\": 256000,\n      \"audio_format\": \"mp3\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/minimax/music-2.5/predictions\"\n```\n\nPoll `urls.get` until `status` is `succeeded`; download `output`.\n\n## Before generating\n\n1. Complete"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"music-2-5\",\n  \"version\": \"1.0.14\",\n  \"publishedAt\": 1790696162415\n}"},{"path":"skill-card.md","content":"## Description:\n\nGuides agents in creating original vocal songs from lyrics and optional style prompts using MiniMax Music 2.5 via Replicate.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[pruna-ai](https://clawhub.ai/user/pruna-ai)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nCreators and developers use this skill to guide an agent through preparing lyrics, specifying a musical style, and generating a vocal song for standalone listening or a music video.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Lyrics and prompts are sent to Replicate and MiniMax.\n\nMitigation: Do not submit confidential or regulated text; review the providers' terms before use.\n\nRisk: Unpinned skill-install examples can introduce supply-chain changes.\n\nMitigation: Prefer reviewed, pinned skill versions when available.\n\n## Reference(s):\n\n- [Replicate MiniMax Music 2.5 prediction endpoint](https://api.replicate.com/v1/models/minimax/music-2.5/predictions)\n\n## Skill Output:\n\n**Output Type(s):** [Guidance, Shell commands, JSON]\n\n**Output Format:** [Markdown instructions with shell and JSON examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Guides retrieval of generated song audio; MP3, WAV, and PCM are supported output formats.]\n\n## Skill Version(s):\n\n1.0.14 (source: skill frontmatter and server release)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."},{"path":"skill.manifest.json","content":"{\n  \"references\": []\n}"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. Skill: music-2.5 Owner: pruna-ai Summary: Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. Tags: ai:1.0.14, generative:1.0.14, latest:1.0.14, pruna:1.0.14 Version history: v1.0.14 | 2026-09-29T15:36:02.415Z | auto - Version bump: updated metadata version to 1.0.14. - Removed skill-card.md file. - No changes to usage, features, or required/o","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1171,"uniquenessScore":49,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-10T13:57:59.884Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-10T13:57:59.884Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T17:37:43.255Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}