{"id":"9f12bfd2-fb48-48c5-b4d0-51665f1ae1da","entityType":"agent","slug":"clawhub-pruna-ai-audio-prompting","name":"audio-prompting","canonicalUrl":"https://www.xpersona.co/agent/clawhub-pruna-ai-audio-prompting","canonicalPath":"/agent/clawhub-pruna-ai-audio-prompting","generatedAt":"2026-10-11T20:58:01.465Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T17:08:17.830Z","emptyReason":null},"description":"Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering.","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s173djpkkm1x4yfzg522h5vfp989mrn5:audio-prompting","sourceUrl":"https://clawhub.ai/pruna-ai/audio-prompting","homepage":"https://clawhub.ai/pruna-ai/skills/audio-prompting","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/pruna-ai/audio-prompting","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/pruna-ai/skills/audio-prompting","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":60,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"audio-prompting technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T17:08:17.830Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T17:08:17.830Z","emptyReason":null},"stars":null,"forks":null,"downloads":1021,"packageName":null,"latestVersion":"1.0.14","tractionLabel":"1K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T17:08:17.816Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T17:08:17.830Z","lastCrawledAt":"2026-10-11T17:08:17.816Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T17:08:17.816Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.14","createdAt":"2026-09-29T15:31:35.155Z","changelog":"- Updated version metadata to 1.0.14 in SKILL.md. - Removed the skill-card.md file. - No changes to user-facing instructions, features, or functionality.","fileCount":7,"zipByteSize":10910},{"version":"1.0.13","createdAt":"2026-09-17T13:52:58.793Z","changelog":"- Updated skill version to 1.0.13. - Improved p-video tool descriptions in the installation table (SKILL.md) for clarity and detailed use cases. - Removed skill-card.md from the repository. - No changes to core audio prompting workflow or guidance.","fileCount":7,"zipByteSize":10998},{"version":"1.0.12","createdAt":"2026-09-10T13:51:19.190Z","changelog":"- Updated to version 1.0.12. - Added references to the new `p-video-2` skill for higher-quality video clip generation and clarified its role versus `p-video`. - Changed guidance to prefer embedding audio in `p-video-2` for layered explainer audio. - Removed the file `skill-card.md`.","fileCount":7,"zipByteSize":11018},{"version":"1.0.11","createdAt":"2026-09-03T14:06:47.943Z","changelog":"- Version bump to 1.0.11 in SKILL.md metadata. - Removed redundant skill-card.md file. - No changes to core guidance, installation, or workflow instructions.","fileCount":7,"zipByteSize":10811},{"version":"1.0.10","createdAt":"2026-08-28T07:53:03.864Z","changelog":"- Version bump to 1.0.10. - Updated SKILL.md metadata version to 1.0.10. - Removed the file skill-card.md.","fileCount":7,"zipByteSize":10854},{"version":"1.0.9","createdAt":"2026-08-04T06:16:20.892Z","changelog":"- Version updated to 1.0.9 in metadata. - Removed the file: skill-card.md. - Documentation and usage instructions in SKILL.md remain unchanged.","fileCount":7,"zipByteSize":10829},{"version":"1.0.8","createdAt":"2026-07-28T17:18:29.690Z","changelog":"- Updated SKILL.md to version 1.0.8, reflecting intake and guidance improvements. - Guide habit revised: now opens `generation-diversity` clarification intake when audio type, locale, or embed-vs-post details are unspecified. - skill-card.md removed from the package. - No functional changes to core audio prompting use or model support.","fileCount":7,"zipByteSize":10806},{"version":"1.0.7","createdAt":"2026-07-23T12:33:06.510Z","changelog":"- Expanded documentation with clear guidance for crafting prompts for TTS, music, and beds across multiple generative audio models. - Detailed usage cases and tool selection (when to use/avoid this skill). - Added step-by-step workflow and references for TTS and music/bed prompting. - Emphasized pairing with related skills for video, editing, and production pipelines. - Clarified distinction between sung vocals and instrumental beds, linking to best practices and worked examples.","fileCount":7,"zipByteSize":10876}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s173djpkkm1x4yfzg522h5vfp989mrn5:audio-prompting","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s173djpkkm1x4yfzg522h5vfp989mrn5:audio-prompting` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/pruna-ai/audio-prompting before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-audio-prompting/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-audio-prompting/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-audio-prompting/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-audio-prompting/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-audio-prompting/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-audio-prompting/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T20:58:01.462Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-audio-prompting/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-audio-prompting/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-audio-prompting/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-audio-prompting/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T17:08:17.830Z","emptyReason":null},"readme":"Skill: audio-prompting\n\nOwner: pruna-ai\n\nSummary: Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering.\n\nTags: ai:1.0.14, generative:1.0.14, latest:1.0.14, pruna:1.0.14\n\nVersion history:\n\nv1.0.14 | 2026-09-29T15:31:35.155Z | auto\n\n- Updated version metadata to 1.0.14 in SKILL.md.\n- Removed the skill-card.md file.\n- No changes to user-facing instructions, features, or functionality.\n\nv1.0.13 | 2026-09-17T13:52:58.793Z | auto\n\n- Updated skill version to 1.0.13.\n- Improved p-video tool descriptions in the installation table (SKILL.md) for clarity and detailed use cases.\n- Removed skill-card.md from the repository.\n- No changes to core audio prompting workflow or guidance.\n\nv1.0.12 | 2026-09-10T13:51:19.190Z | auto\n\n- Updated to version 1.0.12.\n- Added references to the new `p-video-2` skill for higher-quality video clip generation and clarified its role versus `p-video`.\n- Changed guidance to prefer embedding audio in `p-video-2` for layered explainer audio.\n- Removed the file `skill-card.md`.\n\nv1.0.11 | 2026-09-03T14:06:47.943Z | auto\n\n- Version bump to 1.0.11 in SKILL.md metadata.\n- Removed redundant skill-card.md file.\n- No changes to core guidance, installation, or workflow instructions.\n\nv1.0.10 | 2026-08-28T07:53:03.864Z | auto\n\n- Version bump to 1.0.10.\n- Updated SKILL.md metadata version to 1.0.10.\n- Removed the file skill-card.md.\n\nv1.0.9 | 2026-08-04T06:16:20.892Z | auto\n\n- Version updated to 1.0.9 in metadata.\n- Removed the file: skill-card.md.\n- Documentation and usage instructions in SKILL.md remain unchanged.\n\nv1.0.8 | 2026-07-28T17:18:29.690Z | auto\n\n- Updated SKILL.md to version 1.0.8, reflecting intake and guidance improvements.\n- Guide habit revised: now opens `generation-diversity` clarification intake when audio type, locale, or embed-vs-post details are unspecified.\n- skill-card.md removed from the package.\n- No functional changes to core audio prompting use or model support.\n\nv1.0.7 | 2026-07-23T12:33:06.510Z | auto\n\n- Expanded documentation with clear guidance for crafting prompts for TTS, music, and beds across multiple generative audio models.\n- Detailed usage cases and tool selection (when to use/avoid this skill).\n- Added step-by-step workflow and references for TTS and music/bed prompting.\n- Emphasized pairing with related skills for video, editing, and production pipelines.\n- Clarified distinction between sung vocals and instrumental beds, linking to best practices and worked examples.\n\nArchive index:\n\nArchive v1.0.14: 7 files, 10910 bytes\n\nFiles: references/audio-post-production.md (7604b), references/music-and-bed-prompting.md (3709b), references/tts-style-prompting.md (2888b), skill-card.md (2021b), skill.manifest.json (121b), SKILL.md (7384b), _meta.json (135b)\n\nFile v1.0.14:SKILL.md\n\n---\nname: audio-prompting\ndescription: Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering.\nlicense: MIT\nmetadata:\n  version: \"1.0.14\"\n  package: pruna-skills\n---\n\n# Audio prompting\n\nVendor-neutral craft for **speech, music, and beds**. Works with Gemini TTS, ElevenLabs, Music 2.5, Stable Audio, Suno, and similar APIs.\n\n## Install\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n\n## When to use\n\n- Director-style TTS prompts and inline performance tags\n- Full songs with vocals vs instrumental beds\n- Choosing when to embed audio in a video model vs mix in post\n- Narration + bed layering pipelines\n\n## Works with\n\nGemini Flash TTS, ElevenLabs, Music 2.5, Stable Audio, Suno, Udio, and other audio models. Pair with `video-prompting` when uploading VO into a video model.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `music-2.5` | Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. | `npx skills add PrunaAI/pruna-skills@music-2.5 -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n| `gemini-3.1-flash-tts` | Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. | `npx skills add PrunaAI/pruna-skills@gemini-3.1-flash-tts -y` |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `image-prompting` | Use when crafting still-image prompts for any generative model — composition, identity sheets, edits, try-on, and photoreal personas. | `npx skills add PrunaAI/pruna-skills@image-prompting -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\n## Guide habit\n\nIn the **first reply**, name `` `audio-prompting` `` in backticks. When VO vs bed vs full song, locale, or embed-vs-post are open, open intake → **`generation-diversity`** clarification intake. For embed-vs-post questions, cite [audio-post-production.md](./references/audio-post-production.md) — prefer embed in the video model; post-mux only as fallback.\n\n## Before generating\n\n1. Follow `generation-diversity` first.\n2. TTS → [tts-style-prompting.md](./references/tts-style-prompting.md).\n3. Songs / beds → [music-and-bed-prompting.md](./references/music-and-bed-prompting.md).\n4. Tool picker + layering → [audio-post-production.md](./references/audio-post-production.md).\n\n## Song structure (vocals)\n\nOriginal tracks with sung vocals → **`music-2.5`**, not Stable Audio. In the **first reply**, say you will draft **lyrics** (verse / chorus / bridge as needed) plus a separate **music** style prompt per [music-and-bed-prompting.md](./references/music-and-bed-prompting.md). Stable Audio is instrumental beds only — **not** for sung vocals. See **Worked examples** in that reference for full lyrics + style samples.\n\n## Layered explainer audio\n\nVO + bed: TTS (`gemini-3.1-flash-tts`) for narration; Stable Audio for **instrumental** underscore only. When asked embed vs post: prefer **embed** in `p-video-2`; **post-mux** / **assembly** mix under VO is fallback only — [audio-post-production.md](./references/audio-post-production.md).\n\n## Related skills\n\nInstall related skills when the job needs them:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\n## Pruna / Replicate tools\n\nMatching install for every model named above. Pick what you need:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `gemini-3.1-flash-tts` | Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. | `npx skills add PrunaAI/pruna-skills@gemini-3.1-flash-tts -y` |\n| `music-2.5` | Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. | `npx skills add PrunaAI/pruna-skills@music-2.5 -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n| `whisperx` | Use when someone needs word-level timestamps from audio — lyric alignment, cut-safe line boundaries, or caption source timing before burn-in with video-editing. | `npx skills add PrunaAI/pruna-skills@whisperx -y` |\n| `p-video-2-pro` | Use when someone wants a cinematic clip from text or start/end frames — product ads, documentary shots, or dialogue with generated audio. Not for 1080p, imported audio tracks, or talking-head-only hosts. | `npx skills add PrunaAI/pruna-skills@p-video-2-pro -y` |\n| `p-video-2` | Use when someone wants a polished short clip from text, images, or imported audio — 1080p B-roll, start/end frame animation, or a motion shot with a mixed track. Not for cinematic generated-audio clips or talking-head-only hosts. | `npx skills add PrunaAI/pruna-skills@p-video-2 -y` |\n| `p-video` | Use when someone wants a simple short clip from text or images — quick B-roll, drafts, or start/end frame animation. Not when the brief needs cinematic generation, highest quality, tight lip-sync, or imported audio at 1080p. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFile v1.0.14:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"audio-prompting\",\n  \"version\": \"1.0.14\",\n  \"publishedAt\": 1790695895155\n}\n\nFile v1.0.14:references/audio-post-production.md\n\n# Audio post-production (Pruna + Replicate)\n\nHow to choose and **layer** audio when building reels, multi-scene films, and launch videos.\n\n**Prompt craft (how to write):** [tts-style-prompting.md](./tts-style-prompting.md) · [music-and-bed-prompting.md](./music-and-bed-prompting.md). For in-video audio modes and talking-head VO, install and follow `video-prompting`.\n\n**Multi-scene narrated films:** scene anchor triple lives in `video-prompting` — pass TTS to **`p-video`** as `input.audio` with `image` + `last_frame_image`; do not post-mux unless re-render is impossible. Workflow: `narrated-multi-scene`.\n\n**Visual-only transitions (no VO):** scene anchor pair in `video-prompting` — `duration` instead of `audio`. Workflow: `visual-transition-reel`.\n\n## Audio-led `p-video` (required when VO/narration exists)\n\nWhen narration, TTS, or a timed audio slice is available **before** video render:\n\n1. Upload the audio file to Pruna (`POST /v1/files`) — see `pruna-api`.\n2. Pass `urls.get` as **`input.audio`** on **`p-video`** (or **`p-video-avatar`** for human lip-sync).\n3. **Omit `duration`** — clip length follows the audio (capped at **20s** on P-API); the model syncs motion to speech.\n4. Set **`save_audio`: true** so the full line is embedded in the output clip.\n5. **Probe TTS length** before render — per-scene lines should be **≤ ~19s** or the API truncates the tail even when `audio` is set.\n6. **Concat** clips in order (narration already on each clip). Optional bed mixed **under** VO in post.\n\n**Never** generate silent `p-video` and ffmpeg-mux narration afterward unless re-render is impossible — post-mux **truncates** lines longer than the video slot (common with Gemini TTS).\n\n**Over 20s?** Shorten scene copy → tighten TTS pace in `style_prompt` → split into two scene rows (each with its own triple). See `narrated-multi-scene` duration gate.\n\n| Need | Approach | Skill |\n|------|----------|-------|\n| Lip-sync / duration locked to VO | Upload audio → `p-video` with `audio` | `p-video` |\n| Documentary / story narrator | Gemini Flash TTS → upload → video | `gemini-3.1-flash-tts` |\n| Light instrumental under dialogue | Stable Audio bed under VO | `stable-audio-2.5` |\n| Full song with sung vocals | Music 2.5 track | `music-2.5` |\n| Speaking on-camera character | Portrait + script / audio | `p-video-avatar` |\n\n**Env:** Pruna calls need `PRUNA_API_KEY`; Replicate audio tools need `REPLICATE_API_TOKEN`. Assembly steps need **`ffmpeg`** / **`ffprobe`**. Credentials: `pruna-api`. Shared ffmpeg recipes (concat, captions, bed mix, export): **`video-editing`**.\n\n## Layering matrix\n\n| Stack | Primary audio | Secondary | Mix notes |\n|-------|---------------|-----------|-----------|\n| **Silent B-roll** | — | — | Concat video only |\n| **Native `p-video` sound** | Model output | — | Keep `save_audio` default; normalize in assembly if scenes differ |\n| **Narration only (fallback)** | Gemini TTS | — | Post-mux only when audio-led `p-video` is not suitable — prefer **Pipeline A** below |\n| **Bed only** | Stable Audio bed | — | Often with `music-video` or reel beds |\n\n## Recommended pipelines\n\n### A — Narrated multi-scene B-roll (**preferred — scene anchor triple**)\n\nUse `narrated-multi-scene` + `video-prompting` (triple) + tools below.\n\n```text\nPhase 0 — intake: scene table with start/end still prompts + narration lines\nPhase 1 — hero + p-image-edit start stills + end stills (parallel)\nPhase 2 — Gemini TTS per scene (parallel) → upload each to /v1/files\nPhase 3 — p-video per scene: input.image + input.last_frame_image + input.audio (parallel; omit duration)\nPhase 4 — ffmpeg concat (VO embedded; frame chain via shared end/start URLs)\nPhase 5 — optional Stable Audio bed under narration\n```\n\n**Scene anchor triple:** same pattern as first/last frame pairing — `audio` is the third required upload per scene row. **`p-video-avatar`:** portrait + optional `last_frame_image` + uploaded `audio`.\n\n### A′ — Post-mux narration (fallback only)\n\nUse only when you already have silent clips and cannot re-render. Risk: TTS longer than clip slots → cut-off VO.\n\n```text\nPhase 3 — p-video I2V without audio → concat → mux TTS in ffmpeg\n```\n\n### C — Launch / product reel (existing pattern)\n\n```text\nPhase 1 — p-video-avatar or replace reel → concat\nPhase 2 — Stable Audio bed via stable-audio-2.5 + ffmpeg bed mix (bed under VO, not replacing it) — mix recipe in `video-editing`\n```\n\n## ffmpeg mixing (conceptual)\n\nFull mix commands and default launch bed level (~**0.20** under clear promo speech; ~0.08–0.12 under soft narration): install **`video-editing`**.\n\n**Narration onto silent concat** (single VO file):\n\n```bash\nffmpeg -y -i concat_video.mp4 -i narration.mp3 \\\n  -map 0:v -map 1:a -c:v copy -c:a aac -b:a 192k -shortest output_with_vo.mp4\n```\n\n**Bed under existing narration + video** (same pattern as `stable-audio-2.5` + ffmpeg bed mix):\n\n```text\n[1:a]volume=0.12,aloop=...[bed];\n[0:a][bed]amix=inputs=2:duration=first[aout]\n```\n\nNarration / avatar dialogue stays on stream `0:a`; bed is stream `1:a` at low volume.\n\n**Bed on silent concat** — loop a short generated clip to full video length (no per-assemble Stable Audio call):\n\n```text\n[1:a]volume=0.12,aloop=loop=-1:size=2e+09[bed]  →  map video + [bed], -shortest\n```\n\nPlan field `\"reuse_bed\": true` skips regeneration when `audio/launch_bed.mp3` exists. Delete that file (or set `reuse_bed: false`) only when you want a new prompt or seed.\n\n## Intake questions (audio)\n\nAsk before generating paid audio or video:\n\n| Topic | Questions |\n|-------|-----------|\n| **Primary voice** | Narrator (`gemini-3.1-flash-tts`), on-screen avatar (`p-video-avatar`), or native `p-video` sound only? |\n| **Narration scope** | Per-scene lines vs one continuous VO track? |\n| **Music / bed** | None, instrumental bed only (`stable-audio-2.5`), or full song (`music-2.5`)? |\n| **Sync strategy** | **Preferred:** TTS → Pruna upload → **`p-video` / `p-video-avatar` with `audio`** (clip length = audio). Post-mux only as fallback. |\n| **Levels** | Bed volume: default ~**0.20** under launch/promo speech; ~0.08–0.12 under avatar VO or soft Gemini narration — mix recipe in `video-editing` |\n\n## Manifest fields\n\n```json\n{\n  \"narration\": { \"enabled\": true, \"voice\": \"Sulafat\", \"mode\": \"per_scene\" },\n  \"background_music\": { \"enabled\": true, \"reuse_bed\": true, \"volume\": 0.10, \"prompt\": \"Instrumental ... no vocals\" },\n  \"p_video_audio\": { \"save_audio\": true }\n}\n```\n\n## Limitations (P-Video audio)\n\n- Native SFX/dialogue quality varies — for premium voice realism, prefer **`gemini-3.1-flash-tts`** or **`p-video-avatar`**, then optionally mix a bed.\n- Multi-speaker native audio can drift; dedicated TTS per role is safer for narration-heavy cuts.\n- Extreme camera motion and complex multi-scene stories are weaker than **frame-anchored chaining** + per-scene prompts — see `p-video` and `video-prompting`.\n\n## Related skills\n\n| Skill | When |\n|-------|------|\n| `video-prompting` | In-video audio modes, scene anchor pair/triple, talking-head VO craft |\n| `pruna-api` | Upload / poll / parallel batches |\n| `narrated-multi-scene` | Multi-scene B-roll + VO playbook |\n| `visual-transition-reel` | Visual-only transitions (no VO) |\n| `music-video` | Full song + lyric-synced video |\n| `p-video` / `p-video-avatar` | Video API calls that consume uploaded audio |\n| `gemini-3.1-flash-tts` / `stable-audio-2.5` / `music-2.5` | Paid audio generation |\n| `video-editing` | ffmpeg assembly, caption burn-in, bed mix under finished video |\n\nFile v1.0.14:references/music-and-bed-prompting.md\n\n# Music and bed prompting\n\nPrompt craft for `music-2.5` (songs with vocals) and `stable-audio-2.5` (instrumental beds). Mix/stack: [audio-post-production.md](./audio-post-production.md). In-video sync: install `video-prompting`.\n\n## Music 2.5 (full song)\n\nStack: **genre + mood + vocal + tempo + instruments + production feel** (≤ ~2000 chars). Pair with a **lyrics** field — verse / chorus / bridge structure for vocal tracks.\n\n```text\nIndie pop, uplifting, warm female vocal, 92 BPM, acoustic guitar and mellow synth pads, no harsh distortion\n```\n\n**Lyrics:** write singable lines per section (verse, chorus, optional bridge). Lock structure before the paid call; same lyrics + prompt still yield different arrangements — lock seeds only when the user asks.\n\n| Include | Avoid |\n|---------|-------|\n| Genre, BPM, vocal timbre, key instruments | Vague `epic cinematic masterpiece` |\n| Explicit `no harsh distortion` / energy caps when needed | Contradictions (`lo-fi quiet` + `stadium EDM drop`) |\n\n## Stable Audio 2.5 (beds under VO)\n\nInstrumental, understated, mix-friendly:\n\n```text\nInstrumental light electronic pop bed, soft groove and mellow synth pads, calm positive tech atmosphere, understated background music, no vocals, 94 BPM\n```\n\nRules:\n\n- Always **`no vocals`** when under narration  \n- Keep energy **below** dialogue — assembly mixes ~0.08–0.15 under VO  \n- Tag style works well; keep prompts short  \n\n## Which tool?\n\n| Need | Tool |\n|------|------|\n| Sung song / music video source | Music 2.5 |\n| Quiet bed under TTS or avatar | Stable Audio 2.5 |\n| Diegetic SFX inside `p-video` | Native `save_audio` / prompt cues — not these models |\n\n## Pre-send\n\n- [ ] Song vs bed chosen deliberately  \n- [ ] Bed: no vocals + BPM + understated  \n- [ ] Song: genre/mood/vocal/tempo + **lyrics** structure present  \n- [ ] Duration matches scene or assembly plan\n\n## Worked examples\n\n### Full song (Music 2.5) — indie pop, remote-work theme\n\nUser lock: warm female vocal, ~92 BPM, acoustic + mellow synth, **not** EDM drop.\n\n**Style prompt** (`prompt` field):\n\n```text\nIndie pop, warm and hopeful, female vocal, 92 BPM, acoustic guitar and mellow synth pads, intimate bedroom-production feel, no harsh distortion, no stadium drop\n```\n\n**Lyrics** (`lyrics` field — verse / chorus / bridge):\n\n```text\n[Verse 1]\nCoffee rings on the desk again\nWindow light on a second screen\nSlack pings like a metronome\nBuilding something from my home\n\n[Chorus]\nWe're still here, we're still on\nPixels bridge what miles have drawn\nHeart in the work, voice in the song\nRemote but never alone\n\n[Verse 2]\nCat walks across the keyboard line\nDeadline hums but the team's aligned\nSame sky, different time zones\nSame goal in our headphones\n\n[Bridge]\nWhen the Wi‑Fi stutters, we don't fold\nCall reconnects — the story holds\n\n[Chorus]\nWe're still here, we're still on\n...\n```\n\nConfirm lyrics + style before `POST`. For music-video cut points later → `whisperx` after the track exists.\n\n### Instrumental bed (Stable Audio 2.5) — explainer under VO\n\nUser lock: **90s**, calm tech explainer, dialogue must stay clear.\n\n```text\nInstrumental light electronic pop bed, soft groove and mellow synth pads, calm positive tech atmosphere, understated background music, no vocals, 94 BPM\n```\n\nDuration: `90` seconds. Mix target ~0.08–0.15 under narration in assembly — see [audio-post-production.md](./audio-post-production.md).\n\n### Wrong tool check\n\n| User ask | Tool |\n|----------|------|\n| \"Sing an original chorus about launch day\" | Music 2.5 + **lyrics** |\n| \"Quiet underscore while the host talks\" | Stable Audio + **no vocals** |\n| \"Replace the sung hook with spoken VO\" | Gemini TTS — not Stable Audio |\n\nFile v1.0.14:references/tts-style-prompting.md\n\n# TTS style prompting (Gemini 3.1 Flash TTS)\n\nDirector-style `prompt` craft for `gemini-3.1-flash-tts`. Upload results to Pruna for Mode B in-video audio (install `video-prompting`). Layering: [audio-post-production.md](./audio-post-production.md).\n\n## Align three channels\n\n| Channel | Role |\n|---------|------|\n| `text` | Spoken words (+ optional inline `[tags]`) |\n| `prompt` | Tone, pace, accent, character — max ~4k bytes |\n| `[tags]` in text | Momentary direction matching `prompt` |\n\nAll three must point the **same** emotional direction.\n\n**Bracket clarity:** `[tags]` live only in this TTS `text` field. Still typography uses double-quoted `\"[STRING]\"` — see `image-prompting`. Native clip dialogue (`[subject] says \"[LINE]\"`) is Mode A in `video-prompting` — not this skill.\n\n## Human narrator defaults\n\n```text\nWarm storybook narrator, gentle pace, empathetic, no announcer voice.\n```\n\n```text\nNatural documentary host, measured pacing, clear consonants, calm authority.\n```\n\nAvoid: radio-ad hype, “cinematic trailer voice”, reading the product brief into `prompt`.\n\n## Duration gate for `p-video`\n\nAudio-led clips cap at **20s** (keep TTS ≤ **~19s**). If `ffprobe` is long:\n\n1. Shorten `text`  \n2. Add pace to `prompt`: `brisk pace, ~2.3 words per second, no filler`  \n3. Split into two scene rows  \n\nNever rely on post-mux over silent video.\n\n## Avatar vs TTS\n\n| Path | Fields |\n|------|--------|\n| Narrator B-roll | Gemini TTS → `p-video` `input.audio` |\n| On-camera speaker | `p-video-avatar` `voice_script` + `voice_prompt` — **not** this TTS `prompt` |\n\nDo not paste VO into avatar `voice_prompt`.\n\n## Pre-send\n\n- [ ] `prompt` / `text` / tags aligned  \n- [ ] Length probed for `p-video`  \n- [ ] Voice + language recorded in manifest for regen consistency\n\n## Worked example — explainer narration (aligned channels)\n\nUser lock: documentary explainer, **measured** pace, ~45s script, will feed `p-video` Mode B.\n\n**`prompt`** (director style):\n\n```text\nNatural documentary host, measured pacing, clear consonants, calm authority, empathetic, no radio-ad hype\n```\n\n**`text`** (spoken words + inline tags):\n\n```text\n[warm] Most teams treat diversity as a checkbox.\n[pause] But the ritual seed is what breaks repetition before you ever hit generate.\n[emphasis] Lock the brief first — then rotate free axes.\n[measured] Same subject, fresh camera and light, every panel.\n```\n\n**Alignment check:** tags match the calm documentary `prompt` — no `[shout]` hype against a gentle director line.\n\n**Duration gate:** 45s exceeds single **~19s** `p-video` audio-led cap → split into **three** scene rows (~15s each) or shorten copy; probe with `ffprobe` after TTS. Never plan one 45s embed clip.\n\n**Avatar redirect:** on-camera host speaking to lens → `p-video-avatar` `voice_script` + `voice_prompt` — do **not** paste this TTS `prompt` into avatar fields.\n\nFile v1.0.14:skill-card.md\n\n## Description:\n\nUse when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[pruna-ai](https://clawhub.ai/user/pruna-ai)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nCreators and developers use this skill to draft speech, song, and instrumental-bed prompts and plan narration and audio layering for generated media.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Suggested unpinned installation commands can execute additional skills without review.\n\nMitigation: Review each install command and skill before running it; prefer pinned or verified versions and avoid the full-suite install unless needed.\n\nRisk: Audio uploads and generated media may expose sensitive content or incur third-party API charges.\n\nMitigation: Check content sensitivity and service terms before uploading, and confirm expected usage and costs before paid generation.\n\n## Reference(s):\n\n- [Audio Prompting on ClawHub](https://clawhub.ai/pruna-ai/skills/audio-prompting)\n- [TTS style prompting](artifact/references/tts-style-prompting.md)\n- [Music and bed prompting](artifact/references/music-and-bed-prompting.md)\n- [Audio post-production](artifact/references/audio-post-production.md)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, Guidance]\n\n**Output Format:** [Text or Markdown prompts and workflow guidance]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May include separate lyrics and music style prompts, TTS direction, or instrumental-bed and layering guidance.]\n\n## Skill Version(s):\n\n1.0.14 (source: SKILL.md frontmatter and server release)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.14:skill.manifest.json\n\n{\n  \"references\": [\n    \"tts-style-prompting.md\",\n    \"music-and-bed-prompting.md\",\n    \"audio-post-production.md\"\n  ]\n}\n\nArchive v1.0.13: 7 files, 10998 bytes\n\nFiles: references/audio-post-production.md (7604b), references/music-and-bed-prompting.md (3709b), references/tts-style-prompting.md (2888b), skill-card.md (2237b), skill.manifest.json (121b), SKILL.md (7384b), _meta.json (135b)\n\nFile v1.0.13:SKILL.md\n\n---\nname: audio-prompting\ndescription: Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering.\nlicense: MIT\nmetadata:\n  version: \"1.0.13\"\n  package: pruna-skills\n---\n\n# Audio prompting\n\nVendor-neutral craft for **speech, music, and beds**. Works with Gemini TTS, ElevenLabs, Music 2.5, Stable Audio, Suno, and similar APIs.\n\n## Install\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n\n## When to use\n\n- Director-style TTS prompts and inline performance tags\n- Full songs with vocals vs instrumental beds\n- Choosing when to embed audio in a video model vs mix in post\n- Narration + bed layering pipelines\n\n## Works with\n\nGemini Flash TTS, ElevenLabs, Music 2.5, Stable Audio, Suno, Udio, and other audio models. Pair with `video-prompting` when uploading VO into a video model.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `music-2.5` | Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. | `npx skills add PrunaAI/pruna-skills@music-2.5 -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n| `gemini-3.1-flash-tts` | Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. | `npx skills add PrunaAI/pruna-skills@gemini-3.1-flash-tts -y` |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `image-prompting` | Use when crafting still-image prompts for any generative model — composition, identity sheets, edits, try-on, and photoreal personas. | `npx skills add PrunaAI/pruna-skills@image-prompting -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\n## Guide habit\n\nIn the **first reply**, name `` `audio-prompting` `` in backticks. When VO vs bed vs full song, locale, or embed-vs-post are open, open intake → **`generation-diversity`** clarification intake. For embed-vs-post questions, cite [audio-post-production.md](./references/audio-post-production.md) — prefer embed in the video model; post-mux only as fallback.\n\n## Before generating\n\n1. Follow `generation-diversity` first.\n2. TTS → [tts-style-prompting.md](./references/tts-style-prompting.md).\n3. Songs / beds → [music-and-bed-prompting.md](./references/music-and-bed-prompting.md).\n4. Tool picker + layering → [audio-post-production.md](./references/audio-post-production.md).\n\n## Song structure (vocals)\n\nOriginal tracks with sung vocals → **`music-2.5`**, not Stable Audio. In the **first reply**, say you will draft **lyrics** (verse / chorus / bridge as needed) plus a separate **music** style prompt per [music-and-bed-prompting.md](./references/music-and-bed-prompting.md). Stable Audio is instrumental beds only — **not** for sung vocals. See **Worked examples** in that reference for full lyrics + style samples.\n\n## Layered explainer audio\n\nVO + bed: TTS (`gemini-3.1-flash-tts`) for narration; Stable Audio for **instrumental** underscore only. When asked embed vs post: prefer **embed** in `p-video-2`; **post-mux** / **assembly** mix under VO is fallback only — [audio-post-production.md](./references/audio-post-production.md).\n\n## Related skills\n\nInstall related skills when the job needs them:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\n## Pruna / Replicate tools\n\nMatching install for every model named above. Pick what you need:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `gemini-3.1-flash-tts` | Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. | `npx skills add PrunaAI/pruna-skills@gemini-3.1-flash-tts -y` |\n| `music-2.5` | Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. | `npx skills add PrunaAI/pruna-skills@music-2.5 -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n| `whisperx` | Use when someone needs word-level timestamps from audio — lyric alignment, cut-safe line boundaries, or caption source timing before burn-in with video-editing. | `npx skills add PrunaAI/pruna-skills@whisperx -y` |\n| `p-video-2-pro` | Use when someone wants a cinematic clip from text or start/end frames — product ads, documentary shots, or dialogue with generated audio. Not for 1080p, imported audio tracks, or talking-head-only hosts. | `npx skills add PrunaAI/pruna-skills@p-video-2-pro -y` |\n| `p-video-2` | Use when someone wants a polished short clip from text, images, or imported audio — 1080p B-roll, start/end frame animation, or a motion shot with a mixed track. Not for cinematic generated-audio clips or talking-head-only hosts. | `npx skills add PrunaAI/pruna-skills@p-video-2 -y` |\n| `p-video` | Use when someone wants a simple short clip from text or images — quick B-roll, drafts, or start/end frame animation. Not when the brief needs cinematic generation, highest quality, tight lip-sync, or imported audio at 1080p. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFile v1.0.13:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"audio-prompting\",\n  \"version\": \"1.0.13\",\n  \"publishedAt\": 1789653178793\n}\n\nFile v1.0.13:references/audio-post-production.md\n\n# Audio post-production (Pruna + Replicate)\n\nHow to choose and **layer** audio when building reels, multi-scene films, and launch videos.\n\n**Prompt craft (how to write):** [tts-style-prompting.md](./tts-style-prompting.md) · [music-and-bed-prompting.md](./music-and-bed-prompting.md). For in-video audio modes and talking-head VO, install and follow `video-prompting`.\n\n**Multi-scene narrated films:** scene anchor triple lives in `video-prompting` — pass TTS to **`p-video`** as `input.audio` with `image` + `last_frame_image`; do not post-mux unless re-render is impossible. Workflow: `narrated-multi-scene`.\n\n**Visual-only transitions (no VO):** scene anchor pair in `video-prompting` — `duration` instead of `audio`. Workflow: `visual-transition-reel`.\n\n## Audio-led `p-video` (required when VO/narration exists)\n\nWhen narration, TTS, or a timed audio slice is available **before** video render:\n\n1. Upload the audio file to Pruna (`POST /v1/files`) — see `pruna-api`.\n2. Pass `urls.get` as **`input.audio`** on **`p-video`** (or **`p-video-avatar`** for human lip-sync).\n3. **Omit `duration`** — clip length follows the audio (capped at **20s** on P-API); the model syncs motion to speech.\n4. Set **`save_audio`: true** so the full line is embedded in the output clip.\n5. **Probe TTS length** before render — per-scene lines should be **≤ ~19s** or the API truncates the tail even when `audio` is set.\n6. **Concat** clips in order (narration already on each clip). Optional bed mixed **under** VO in post.\n\n**Never** generate silent `p-video` and ffmpeg-mux narration afterward unless re-render is impossible — post-mux **truncates** lines longer than the video slot (common with Gemini TTS).\n\n**Over 20s?** Shorten scene copy → tighten TTS pace in `style_prompt` → split into two scene rows (each with its own triple). See `narrated-multi-scene` duration gate.\n\n| Need | Approach | Skill |\n|------|----------|-------|\n| Lip-sync / duration locked to VO | Upload audio → `p-video` with `audio` | `p-video` |\n| Documentary / story narrator | Gemini Flash TTS → upload → video | `gemini-3.1-flash-tts` |\n| Light instrumental under dialogue | Stable Audio bed under VO | `stable-audio-2.5` |\n| Full song with sung vocals | Music 2.5 track | `music-2.5` |\n| Speaking on-camera character | Portrait + script / audio | `p-video-avatar` |\n\n**Env:** Pruna calls need `PRUNA_API_KEY`; Replicate audio tools need `REPLICATE_API_TOKEN`. Assembly steps need **`ffmpeg`** / **`ffprobe`**. Credentials: `pruna-api`. Shared ffmpeg recipes (concat, captions, bed mix, export): **`video-editing`**.\n\n## Layering matrix\n\n| Stack | Primary audio | Secondary | Mix notes |\n|-------|---------------|-----------|-----------|\n| **Silent B-roll** | — | — | Concat video only |\n| **Native `p-video` sound** | Model output | — | Keep `save_audio` default; normalize in assembly if scenes differ |\n| **Narration only (fallback)** | Gemini TTS | — | Post-mux only when audio-led `p-video` is not suitable — prefer **Pipeline A** below |\n| **Bed only** | Stable Audio bed | — | Often with `music-video` or reel beds |\n\n## Recommended pipelines\n\n### A — Narrated multi-scene B-roll (**preferred — scene anchor triple**)\n\nUse `narrated-multi-scene` + `video-prompting` (triple) + tools below.\n\n```text\nPhase 0 — intake: scene table with start/end still prompts + narration lines\nPhase 1 — hero + p-image-edit start stills + end stills (parallel)\nPhase 2 — Gemini TTS per scene (parallel) → upload each to /v1/files\nPhase 3 — p-video per scene: input.image + input.last_frame_image + input.audio (parallel; omit duration)\nPhase 4 — ffmpeg concat (VO embedded; frame chain via shared end/start URLs)\nPhase 5 — optional Stable Audio bed under narration\n```\n\n**Scene anchor triple:** same pattern as first/last frame pairing — `audio` is the third required upload per scene row. **`p-video-avatar`:** portrait + optional `last_frame_image` + uploaded `audio`.\n\n### A′ — Post-mux narration (fallback only)\n\nUse only when you already have silent clips and cannot re-render. Risk: TTS longer than clip slots → cut-off VO.\n\n```text\nPhase 3 — p-video I2V without audio → concat → mux TTS in ffmpeg\n```\n\n### C — Launch / product reel (existing pattern)\n\n```text\nPhase 1 — p-video-avatar or replace reel → concat\nPhase 2 — Stable Audio bed via stable-audio-2.5 + ffmpeg bed mix (bed under VO, not replacing it) — mix recipe in `video-editing`\n```\n\n## ffmpeg mixing (conceptual)\n\nFull mix commands and default launch bed level (~**0.20** under clear promo speech; ~0.08–0.12 under soft narration): install **`video-editing`**.\n\n**Narration onto silent concat** (single VO file):\n\n```bash\nffmpeg -y -i concat_video.mp4 -i narration.mp3 \\\n  -map 0:v -map 1:a -c:v copy -c:a aac -b:a 192k -shortest output_with_vo.mp4\n```\n\n**Bed under existing narration + video** (same pattern as `stable-audio-2.5` + ffmpeg bed mix):\n\n```text\n[1:a]volume=0.12,aloop=...[bed];\n[0:a][bed]amix=inputs=2:duration=first[aout]\n```\n\nNarration / avatar dialogue stays on stream `0:a`; bed is stream `1:a` at low volume.\n\n**Bed on silent concat** — loop a short generated clip to full video length (no per-assemble Stable Audio call):\n\n```text\n[1:a]volume=0.12,aloop=loop=-1:size=2e+09[bed]  →  map video + [bed], -shortest\n```\n\nPlan field `\"reuse_bed\": true` skips regeneration when `audio/launch_bed.mp3` exists. Delete that file (or set `reuse_bed: false`) only when you want a new prompt or seed.\n\n## Intake questions (audio)\n\nAsk before generating paid audio or video:\n\n| Topic | Questions |\n|-------|-----------|\n| **Primary voice** | Narrator (`gemini-3.1-flash-tts`), on-screen avatar (`p-video-avatar`), or native `p-video` sound only? |\n| **Narration scope** | Per-scene lines vs one continuous VO track? |\n| **Music / bed** | None, instrumental bed only (`stable-audio-2.5`), or full song (`music-2.5`)? |\n| **Sync strategy** | **Preferred:** TTS → Pruna upload → **`p-video` / `p-video-avatar` with `audio`** (clip length = audio). Post-mux only as fallback. |\n| **Levels** | Bed volume: default ~**0.20** under launch/promo speech; ~0.08–0.12 under avatar VO or soft Gemini narration — mix recipe in `video-editing` |\n\n## Manifest fields\n\n```json\n{\n  \"narration\": { \"enabled\": true, \"voice\": \"Sulafat\", \"mode\": \"per_scene\" },\n  \"background_music\": { \"enabled\": true, \"reuse_bed\": true, \"volume\": 0.10, \"prompt\": \"Instrumental ... no vocals\" },\n  \"p_video_audio\": { \"save_audio\": true }\n}\n```\n\n## Limitations (P-Video audio)\n\n- Native SFX/dialogue quality varies — for premium voice realism, prefer **`gemini-3.1-flash-tts`** or **`p-video-avatar`**, then optionally mix a bed.\n- Multi-speaker native audio can drift; dedicated TTS per role is safer for narration-heavy cuts.\n- Extreme camera motion and complex multi-scene stories are weaker than **frame-anchored chaining** + per-scene prompts — see `p-video` and `video-prompting`.\n\n## Related skills\n\n| Skill | When |\n|-------|------|\n| `video-prompting` | In-video audio modes, scene anchor pair/triple, talking-head VO craft |\n| `pruna-api` | Upload / poll / parallel batches |\n| `narrated-multi-scene` | Multi-scene B-roll + VO playbook |\n| `visual-transition-reel` | Visual-only transitions (no VO) |\n| `music-video` | Full song + lyric-synced video |\n| `p-video` / `p-video-avatar` | Video API calls that consume uploaded audio |\n| `gemini-3.1-flash-tts` / `stable-audio-2.5` / `music-2.5` | Paid audio generation |\n| `video-editing` | ffmpeg assembly, caption burn-in, bed mix under finished video |\n\nFile v1.0.13:references/music-and-bed-prompting.md\n\n# Music and bed prompting\n\nPrompt craft for `music-2.5` (songs with vocals) and `stable-audio-2.5` (instrumental beds). Mix/stack: [audio-post-production.md](./audio-post-production.md). In-video sync: install `video-prompting`.\n\n## Music 2.5 (full song)\n\nStack: **genre + mood + vocal + tempo + instruments + production feel** (≤ ~2000 chars). Pair with a **lyrics** field — verse / chorus / bridge structure for vocal tracks.\n\n```text\nIndie pop, uplifting, warm female vocal, 92 BPM, acoustic guitar and mellow synth pads, no harsh distortion\n```\n\n**Lyrics:** write singable lines per section (verse, chorus, optional bridge). Lock structure before the paid call; same lyrics + prompt still yield different arrangements — lock seeds only when the user asks.\n\n| Include | Avoid |\n|---------|-------|\n| Genre, BPM, vocal timbre, key instruments | Vague `epic cinematic masterpiece` |\n| Explicit `no harsh distortion` / energy caps when needed | Contradictions (`lo-fi quiet` + `stadium EDM drop`) |\n\n## Stable Audio 2.5 (beds under VO)\n\nInstrumental, understated, mix-friendly:\n\n```text\nInstrumental light electronic pop bed, soft groove and mellow synth pads, calm positive tech atmosphere, understated background music, no vocals, 94 BPM\n```\n\nRules:\n\n- Always **`no vocals`** when under narration  \n- Keep energy **below** dialogue — assembly mixes ~0.08–0.15 under VO  \n- Tag style works well; keep prompts short  \n\n## Which tool?\n\n| Need | Tool |\n|------|------|\n| Sung song / music video source | Music 2.5 |\n| Quiet bed under TTS or avatar | Stable Audio 2.5 |\n| Diegetic SFX inside `p-video` | Native `save_audio` / prompt cues — not these models |\n\n## Pre-send\n\n- [ ] Song vs bed chosen deliberately  \n- [ ] Bed: no vocals + BPM + understated  \n- [ ] Song: genre/mood/vocal/tempo + **lyrics** structure present  \n- [ ] Duration matches scene or assembly plan\n\n## Worked examples\n\n### Full song (Music 2.5) — indie pop, remote-work theme\n\nUser lock: warm female vocal, ~92 BPM, acoustic + mellow synth, **not** EDM drop.\n\n**Style prompt** (`prompt` field):\n\n```text\nIndie pop, warm and hopeful, female vocal, 92 BPM, acoustic guitar and mellow synth pads, intimate bedroom-production feel, no harsh distortion, no stadium drop\n```\n\n**Lyrics** (`lyrics` field — verse / chorus / bridge):\n\n```text\n[Verse 1]\nCoffee rings on the desk again\nWindow light on a second screen\nSlack pings like a metronome\nBuilding something from my home\n\n[Chorus]\nWe're still here, we're still on\nPixels bridge what miles have drawn\nHeart in the work, voice in the song\nRemote but never alone\n\n[Verse 2]\nCat walks across the keyboard line\nDeadline hums but the team's aligned\nSame sky, different time zones\nSame goal in our headphones\n\n[Bridge]\nWhen the Wi‑Fi stutters, we don't fold\nCall reconnects — the story holds\n\n[Chorus]\nWe're still here, we're still on\n...\n```\n\nConfirm lyrics + style before `POST`. For music-video cut points later → `whisperx` after the track exists.\n\n### Instrumental bed (Stable Audio 2.5) — explainer under VO\n\nUser lock: **90s**, calm tech explainer, dialogue must stay clear.\n\n```text\nInstrumental light electronic pop bed, soft groove and mellow synth pads, calm positive tech atmosphere, understated background music, no vocals, 94 BPM\n```\n\nDuration: `90` seconds. Mix target ~0.08–0.15 under narration in assembly — see [audio-post-production.md](./audio-post-production.md).\n\n### Wrong tool check\n\n| User ask | Tool |\n|----------|------|\n| \"Sing an original chorus about launch day\" | Music 2.5 + **lyrics** |\n| \"Quiet underscore while the host talks\" | Stable Audio + **no vocals** |\n| \"Replace the sung hook with spoken VO\" | Gemini TTS — not Stable Audio |\n\nFile v1.0.13:references/tts-style-prompting.md\n\n# TTS style prompting (Gemini 3.1 Flash TTS)\n\nDirector-style `prompt` craft for `gemini-3.1-flash-tts`. Upload results to Pruna for Mode B in-video audio (install `video-prompting`). Layering: [audio-post-production.md](./audio-post-production.md).\n\n## Align three channels\n\n| Channel | Role |\n|---------|------|\n| `text` | Spoken words (+ optional inline `[tags]`) |\n| `prompt` | Tone, pace, accent, character — max ~4k bytes |\n| `[tags]` in text | Momentary direction matching `prompt` |\n\nAll three must point the **same** emotional direction.\n\n**Bracket clarity:** `[tags]` live only in this TTS `text` field. Still typography uses double-quoted `\"[STRING]\"` — see `image-prompting`. Native clip dialogue (`[subject] says \"[LINE]\"`) is Mode A in `video-prompting` — not this skill.\n\n## Human narrator defaults\n\n```text\nWarm storybook narrator, gentle pace, empathetic, no announcer voice.\n```\n\n```text\nNatural documentary host, measured pacing, clear consonants, calm authority.\n```\n\nAvoid: radio-ad hype, “cinematic trailer voice”, reading the product brief into `prompt`.\n\n## Duration gate for `p-video`\n\nAudio-led clips cap at **20s** (keep TTS ≤ **~19s**). If `ffprobe` is long:\n\n1. Shorten `text`  \n2. Add pace to `prompt`: `brisk pace, ~2.3 words per second, no filler`  \n3. Split into two scene rows  \n\nNever rely on post-mux over silent video.\n\n## Avatar vs TTS\n\n| Path | Fields |\n|------|--------|\n| Narrator B-roll | Gemini TTS → `p-video` `input.audio` |\n| On-camera speaker | `p-video-avatar` `voice_script` + `voice_prompt` — **not** this TTS `prompt` |\n\nDo not paste VO into avatar `voice_prompt`.\n\n## Pre-send\n\n- [ ] `prompt` / `text` / tags aligned  \n- [ ] Length probed for `p-video`  \n- [ ] Voice + language recorded in manifest for regen consistency\n\n## Worked example — explainer narration (aligned channels)\n\nUser lock: documentary explainer, **measured** pace, ~45s script, will feed `p-video` Mode B.\n\n**`prompt`** (director style):\n\n```text\nNatural documentary host, measured pacing, clear consonants, calm authority, empathetic, no radio-ad hype\n```\n\n**`text`** (spoken words + inline tags):\n\n```text\n[warm] Most teams treat diversity as a checkbox.\n[pause] But the ritual seed is what breaks repetition before you ever hit generate.\n[emphasis] Lock the brief first — then rotate free axes.\n[measured] Same subject, fresh camera and light, every panel.\n```\n\n**Alignment check:** tags match the calm documentary `prompt` — no `[shout]` hype against a gentle director line.\n\n**Duration gate:** 45s exceeds single **~19s** `p-video` audio-led cap → split into **three** scene rows (~15s each) or shorten copy; probe with `ffprobe` after TTS. Never plan one 45s embed clip.\n\n**Avatar redirect:** on-camera host speaking to lens → `p-video-avatar` `voice_script` + `voice_prompt` — do **not** paste this TTS `prompt` into avatar fields.\n\nFile v1.0.13:skill-card.md\n\n## Description:\n\nUse when crafting TTS, music, or bed prompts for any generative audio model - director style, song structure, and post-production layering.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[pruna-ai](https://clawhub.ai/user/pruna-ai)\n\n### License/Terms of Use:\n\nMIT\n\n## Use Case:\n\nDevelopers and content teams use this skill to craft prompts for TTS narration, songs, instrumental beds, and audio/video layering across generative audio workflows.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Unpinned npx installer commands can install changing upstream content.\n\nMitigation: Install only from a trusted PrunaAI source and pin or review the intended release before use.\n\nRisk: Audio, scripts, images, or videos may be uploaded to external providers during related workflows.\n\nMitigation: Do not upload confidential or unauthorized content, and check the provider data-handling terms before use.\n\nRisk: Long narration can exceed audio-led video duration limits and lead to truncated output.\n\nMitigation: Probe audio duration, shorten copy, tighten pace, or split narration into separate scene rows before generation.\n\n## Reference(s):\n\n- [TTS style prompting](references/tts-style-prompting.md)\n- [Music and bed prompting](references/music-and-bed-prompting.md)\n- [Audio post-production](references/audio-post-production.md)\n- [ClawHub skill page](https://clawhub.ai/pruna-ai/skills/audio-prompting)\n\n## Skill Output:\n\n**Output Type(s):** [guidance, markdown, code, shell commands, configuration]\n\n**Output Format:** [Markdown guidance with prompt templates, structured checklists, inline code blocks, and shell command examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [References companion Pruna and Replicate skills for API calls, generation, and post-production assembly.]\n\n## Skill Version(s):\n\n1.0.13 (source: SKILL.md metadata and server release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.13:skill.manifest.json\n\n{\n  \"references\": [\n    \"tts-style-prompting.md\",\n    \"music-and-bed-prompting.md\",\n    \"audio-post-production.md\"\n  ]\n}\n\nArchive v1.0.12: 7 files, 11018 bytes\n\nFiles: references/audio-post-production.md (7604b), references/music-and-bed-prompting.md (3709b), references/tts-style-prompting.md (2888b), skill-card.md (2452b), skill.manifest.json (121b), SKILL.md (7050b), _meta.json (135b)\n\nFile v1.0.12:SKILL.md\n\n---\nname: audio-prompting\ndescription: Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering.\nlicense: MIT\nmetadata:\n  version: \"1.0.12\"\n  package: pruna-skills\n---\n\n# Audio prompting\n\nVendor-neutral craft for **speech, music, and beds**. Works with Gemini TTS, ElevenLabs, Music 2.5, Stable Audio, Suno, and similar APIs.\n\n## Install\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n\n## When to use\n\n- Director-style TTS prompts and inline performance tags\n- Full songs with vocals vs instrumental beds\n- Choosing when to embed audio in a video model vs mix in post\n- Narration + bed layering pipelines\n\n## Works with\n\nGemini Flash TTS, ElevenLabs, Music 2.5, Stable Audio, Suno, Udio, and other audio models. Pair with `video-prompting` when uploading VO into a video model.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `music-2.5` | Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. | `npx skills add PrunaAI/pruna-skills@music-2.5 -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n| `gemini-3.1-flash-tts` | Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. | `npx skills add PrunaAI/pruna-skills@gemini-3.1-flash-tts -y` |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `image-prompting` | Use when crafting still-image prompts for any generative model — composition, identity sheets, edits, try-on, and photoreal personas. | `npx skills add PrunaAI/pruna-skills@image-prompting -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\n## Guide habit\n\nIn the **first reply**, name `` `audio-prompting` `` in backticks. When VO vs bed vs full song, locale, or embed-vs-post are open, open intake → **`generation-diversity`** clarification intake. For embed-vs-post questions, cite [audio-post-production.md](./references/audio-post-production.md) — prefer embed in the video model; post-mux only as fallback.\n\n## Before generating\n\n1. Follow `generation-diversity` first.\n2. TTS → [tts-style-prompting.md](./references/tts-style-prompting.md).\n3. Songs / beds → [music-and-bed-prompting.md](./references/music-and-bed-prompting.md).\n4. Tool picker + layering → [audio-post-production.md](./references/audio-post-production.md).\n\n## Song structure (vocals)\n\nOriginal tracks with sung vocals → **`music-2.5`**, not Stable Audio. In the **first reply**, say you will draft **lyrics** (verse / chorus / bridge as needed) plus a separate **music** style prompt per [music-and-bed-prompting.md](./references/music-and-bed-prompting.md). Stable Audio is instrumental beds only — **not** for sung vocals. See **Worked examples** in that reference for full lyrics + style samples.\n\n## Layered explainer audio\n\nVO + bed: TTS (`gemini-3.1-flash-tts`) for narration; Stable Audio for **instrumental** underscore only. When asked embed vs post: prefer **embed** in `p-video-2`; **post-mux** / **assembly** mix under VO is fallback only — [audio-post-production.md](./references/audio-post-production.md).\n\n## Related skills\n\nInstall related skills when the job needs them:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\n## Pruna / Replicate tools\n\nMatching install for every model named above. Pick what you need:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `gemini-3.1-flash-tts` | Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. | `npx skills add PrunaAI/pruna-skills@gemini-3.1-flash-tts -y` |\n| `music-2.5` | Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. | `npx skills add PrunaAI/pruna-skills@music-2.5 -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n| `whisperx` | Use when someone needs word-level timestamps from audio — lyric alignment, cut-safe line boundaries, or caption source timing before burn-in with video-editing. | `npx skills add PrunaAI/pruna-skills@whisperx -y` |\n| `p-video-2` | Use when someone wants the best-quality short clip from text, images, or audio — polished B-roll, start/end frame animation, or a motion shot with stronger lip-sync. Not for full multi-scene films or talking-head-only hosts. | `npx skills add PrunaAI/pruna-skills@p-video-2 -y` |\n| `p-video` | Use when someone wants a simple short clip from text or images — quick B-roll, drafts, or start/end frame animation. Not when the brief needs the highest quality or tight lip-sync. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFile v1.0.12:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"audio-prompting\",\n  \"version\": \"1.0.12\",\n  \"publishedAt\": 1789048279190\n}\n\nFile v1.0.12:references/audio-post-production.md\n\n# Audio post-production (Pruna + Replicate)\n\nHow to choose and **layer** audio when building reels, multi-scene films, and launch videos.\n\n**Prompt craft (how to write):** [tts-style-prompting.md](./tts-style-prompting.md) · [music-and-bed-prompting.md](./music-and-bed-prompting.md). For in-video audio modes and talking-head VO, install and follow `video-prompting`.\n\n**Multi-scene narrated films:** scene anchor triple lives in `video-prompting` — pass TTS to **`p-video`** as `input.audio` with `image` + `last_frame_image`; do not post-mux unless re-render is impossible. Workflow: `narrated-multi-scene`.\n\n**Visual-only transitions (no VO):** scene anchor pair in `video-prompting` — `duration` instead of `audio`. Workflow: `visual-transition-reel`.\n\n## Audio-led `p-video` (required when VO/narration exists)\n\nWhen narration, TTS, or a timed audio slice is available **before** video render:\n\n1. Upload the audio file to Pruna (`POST /v1/files`) — see `pruna-api`.\n2. Pass `urls.get` as **`input.audio`** on **`p-video`** (or **`p-video-avatar`** for human lip-sync).\n3. **Omit `duration`** — clip length follows the audio (capped at **20s** on P-API); the model syncs motion to speech.\n4. Set **`save_audio`: true** so the full line is embedded in the output clip.\n5. **Probe TTS length** before render — per-scene lines should be **≤ ~19s** or the API truncates the tail even when `audio` is set.\n6. **Concat** clips in order (narration already on each clip). Optional bed mixed **under** VO in post.\n\n**Never** generate silent `p-video` and ffmpeg-mux narration afterward unless re-render is impossible — post-mux **truncates** lines longer than the video slot (common with Gemini TTS).\n\n**Over 20s?** Shorten scene copy → tighten TTS pace in `style_prompt` → split into two scene rows (each with its own triple). See `narrated-multi-scene` duration gate.\n\n| Need | Approach | Skill |\n|------|----------|-------|\n| Lip-sync / duration locked to VO | Upload audio → `p-video` with `audio` | `p-video` |\n| Documentary / story narrator | Gemini Flash TTS → upload → video | `gemini-3.1-flash-tts` |\n| Light instrumental under dialogue | Stable Audio bed under VO | `stable-audio-2.5` |\n| Full song with sung vocals | Music 2.5 track | `music-2.5` |\n| Speaking on-camera character | Portrait + script / audio | `p-video-avatar` |\n\n**Env:** Pruna calls need `PRUNA_API_KEY`; Replicate audio tools need `REPLICATE_API_TOKEN`. Assembly steps need **`ffmpeg`** / **`ffprobe`**. Credentials: `pruna-api`. Shared ffmpeg recipes (concat, captions, bed mix, export): **`video-editing`**.\n\n## Layering matrix\n\n| Stack | Primary audio | Secondary | Mix notes |\n|-------|---------------|-----------|-----------|\n| **Silent B-roll** | — | — | Concat video only |\n| **Native `p-video` sound** | Model output | — | Keep `save_audio` default; normalize in assembly if scenes differ |\n| **Narration only (fallback)** | Gemini TTS | — | Post-mux only when audio-led `p-video` is not suitable — prefer **Pipeline A** below |\n| **Bed only** | Stable Audio bed | — | Often with `music-video` or reel beds |\n\n## Recommended pipelines\n\n### A — Narrated multi-scene B-roll (**preferred — scene anchor triple**)\n\nUse `narrated-multi-scene` + `video-prompting` (triple) + tools below.\n\n```text\nPhase 0 — intake: scene table with start/end still prompts + narration lines\nPhase 1 — hero + p-image-edit start stills + end stills (parallel)\nPhase 2 — Gemini TTS per scene (parallel) → upload each to /v1/files\nPhase 3 — p-video per scene: input.image + input.last_frame_image + input.audio (parallel; omit duration)\nPhase 4 — ffmpeg concat (VO embedded; frame chain via shared end/start URLs)\nPhase 5 — optional Stable Audio bed under narration\n```\n\n**Scene anchor triple:** same pattern as first/last frame pairing — `audio` is the third required upload per scene row. **`p-video-avatar`:** portrait + optional `last_frame_image` + uploaded `audio`.\n\n### A′ — Post-mux narration (fallback only)\n\nUse only when you already have silent clips and cannot re-render. Risk: TTS longer than clip slots → cut-off VO.\n\n```text\nPhase 3 — p-video I2V without audio → concat → mux TTS in ffmpeg\n```\n\n### C — Launch / product reel (existing pattern)\n\n```text\nPhase 1 — p-video-avatar or replace reel → concat\nPhase 2 — Stable Audio bed via stable-audio-2.5 + ffmpeg bed mix (bed under VO, not replacing it) — mix recipe in `video-editing`\n```\n\n## ffmpeg mixing (conceptual)\n\nFull mix commands and default launch bed level (~**0.20** under clear promo speech; ~0.08–0.12 under soft narration): install **`video-editing`**.\n\n**Narration onto silent concat** (single VO file):\n\n```bash\nffmpeg -y -i concat_video.mp4 -i narration.mp3 \\\n  -map 0:v -map 1:a -c:v copy -c:a aac -b:a 192k -shortest output_with_vo.mp4\n```\n\n**Bed under existing narration + video** (same pattern as `stable-audio-2.5` + ffmpeg bed mix):\n\n```text\n[1:a]volume=0.12,aloop=...[bed];\n[0:a][bed]amix=inputs=2:duration=first[aout]\n```\n\nNarration / avatar dialogue stays on stream `0:a`; bed is stream `1:a` at low volume.\n\n**Bed on silent concat** — loop a short generated clip to full video length (no per-assemble Stable Audio call):\n\n```text\n[1:a]volume=0.12,aloop=loop=-1:size=2e+09[bed]  →  map video + [bed], -shortest\n```\n\nPlan field `\"reuse_bed\": true` skips regeneration when `audio/launch_bed.mp3` exists. Delete that file (or set `reuse_bed: false`) only when you want a new prompt or seed.\n\n## Intake questions (audio)\n\nAsk before generating paid audio or video:\n\n| Topic | Questions |\n|-------|-----------|\n| **Primary voice** | Narrator (`gemini-3.1-flash-tts`), on-screen avatar (`p-video-avatar`), or native `p-video` sound only? |\n| **Narration scope** | Per-scene lines vs one continuous VO track? |\n| **Music / bed** | None, instrumental bed only (`stable-audio-2.5`), or full song (`music-2.5`)? |\n| **Sync strategy** | **Preferred:** TTS → Pruna upload → **`p-video` / `p-video-avatar` with `audio`** (clip length = audio). Post-mux only as fallback. |\n| **Levels** | Bed volume: default ~**0.20** under launch/promo speech; ~0.08–0.12 under avatar VO or soft Gemini narration — mix recipe in `video-editing` |\n\n## Manifest fields\n\n```json\n{\n  \"narration\": { \"enabled\": true, \"voice\": \"Sulafat\", \"mode\": \"per_scene\" },\n  \"background_music\": { \"enabled\": true, \"reuse_bed\": true, \"volume\": 0.10, \"prompt\": \"Instrumental ... no vocals\" },\n  \"p_video_audio\": { \"save_audio\": true }\n}\n```\n\n## Limitations (P-Video audio)\n\n- Native SFX/dialogue quality varies — for premium voice realism, prefer **`gemini-3.1-flash-tts`** or **`p-video-avatar`**, then optionally mix a bed.\n- Multi-speaker native audio can drift; dedicated TTS per role is safer for narration-heavy cuts.\n- Extreme camera motion and complex multi-scene stories are weaker than **frame-anchored chaining** + per-scene prompts — see `p-video` and `video-prompting`.\n\n## Related skills\n\n| Skill | When |\n|-------|------|\n| `video-prompting` | In-video audio modes, scene anchor pair/triple, talking-head VO craft |\n| `pruna-api` | Upload / poll / parallel batches |\n| `narrated-multi-scene` | Multi-scene B-roll + VO playbook |\n| `visual-transition-reel` | Visual-only transitions (no VO) |\n| `music-video` | Full song + lyric-synced video |\n| `p-video` / `p-video-avatar` | Video API calls that consume uploaded audio |\n| `gemini-3.1-flash-tts` / `stable-audio-2.5` / `music-2.5` | Paid audio generation |\n| `video-editing` | ffmpeg assembly, caption burn-in, bed mix under finished video |\n\nFile v1.0.12:references/music-and-bed-prompting.md\n\n# Music and bed prompting\n\nPrompt craft for `music-2.5` (songs with vocals) and `stable-audio-2.5` (instrumental beds). Mix/stack: [audio-post-production.md](./audio-post-production.md). In-video sync: install `video-prompting`.\n\n## Music 2.5 (full song)\n\nStack: **genre + mood + vocal + tempo + instruments + production feel** (≤ ~2000 chars). Pair with a **lyrics** field — verse / chorus / bridge structure for vocal tracks.\n\n```text\nIndie pop, uplifting, warm female vocal, 92 BPM, acoustic guitar and mellow synth pads, no harsh distortion\n```\n\n**Lyrics:** write singable lines per section (verse, chorus, optional bridge). Lock structure before the paid call; same lyrics + prompt still yield different arrangements — lock seeds only when the user asks.\n\n| Include | Avoid |\n|---------|-------|\n| Genre, BPM, vocal timbre, key instruments | Vague `epic cinematic masterpiece` |\n| Explicit `no harsh distortion` / energy caps when needed | Contradictions (`lo-fi quiet` + `stadium EDM drop`) |\n\n## Stable Audio 2.5 (beds under VO)\n\nInstrumental, understated, mix-friendly:\n\n```text\nInstrumental light electronic pop bed, soft groove and mellow synth pads, calm positive tech atmosphere, understated background music, no vocals, 94 BPM\n```\n\nRules:\n\n- Always **`no vocals`** when under narration  \n- Keep energy **below** dialogue — assembly mixes ~0.08–0.15 under VO  \n- Tag style works well; keep prompts short  \n\n## Which tool?\n\n| Need | Tool |\n|------|------|\n| Sung song / music video source | Music 2.5 |\n| Quiet bed under TTS or avatar | Stable Audio 2.5 |\n| Diegetic SFX inside `p-video` | Native `save_audio` / prompt cues — not these models |\n\n## Pre-send\n\n- [ ] Song vs bed chosen deliberately  \n- [ ] Bed: no vocals + BPM + understated  \n- [ ] Song: genre/mood/vocal/tempo + **lyrics** structure present  \n- [ ] Duration matches scene or assembly plan\n\n## Worked examples\n\n### Full song (Music 2.5) — indie pop, remote-work theme\n\nUser lock: warm female vocal, ~92 BPM, acoustic + mellow synth, **not** EDM drop.\n\n**Style prompt** (`prompt` field):\n\n```text\nIndie pop, warm and hopeful, female vocal, 92 BPM, acoustic guitar and mellow synth pads, intimate bedroom-production feel, no harsh distortion, no stadium drop\n```\n\n**Lyrics** (`lyrics` field — verse / chorus / bridge):\n\n```text\n[Verse 1]\nCoffee rings on the desk again\nWindow light on a second screen\nSlack pings like a metronome\nBuilding something from my home\n\n[Chorus]\nWe're still here, we're still on\nPixels bridge what miles have drawn\nHeart in the work, voice in the song\nRemote but never alone\n\n[Verse 2]\nCat walks across the keyboard line\nDeadline hums but the team's aligned\nSame sky, different time zones\nSame goal in our headphones\n\n[Bridge]\nWhen the Wi‑Fi stutters, we don't fold\nCall reconnects — the story holds\n\n[Chorus]\nWe're still here, we're still on\n...\n```\n\nConfirm lyrics + style before `POST`. For music-video cut points later → `whisperx` after the track exists.\n\n### Instrumental bed (Stable Audio 2.5) — explainer under VO\n\nUser lock: **90s**, calm tech explainer, dialogue must stay clear.\n\n```text\nInstrumental light electronic pop bed, soft groove and mellow synth pads, calm positive tech atmosphere, understated background music, no vocals, 94 BPM\n```\n\nDuration: `90` seconds. Mix target ~0.08–0.15 under narration in assembly — see [audio-post-production.md](./audio-post-production.md).\n\n### Wrong tool check\n\n| User ask | Tool |\n|----------|------|\n| \"Sing an original chorus about launch day\" | Music 2.5 + **lyrics** |\n| \"Quiet underscore while the host talks\" | Stable Audio + **no vocals** |\n| \"Replace the sung hook with spoken VO\" | Gemini TTS — not Stable Audio |\n\nFile v1.0.12:references/tts-style-prompting.md\n\n# TTS style prompting (Gemini 3.1 Flash TTS)\n\nDirector-style `prompt` craft for `gemini-3.1-flash-tts`. Upload results to Pruna for Mode B in-video audio (install `video-prompting`). Layering: [audio-post-production.md](./audio-post-production.md).\n\n## Align three channels\n\n| Channel | Role |\n|---------|------|\n| `text` | Spoken words (+ optional inline `[tags]`) |\n| `prompt` | Tone, pace, accent, character — max ~4k bytes |\n| `[tags]` in text | Momentary direction matching `prompt` |\n\nAll three must point the **same** emotional direction.\n\n**Bracket clarity:** `[tags]` live only in this TTS `text` field. Still typography uses double-quoted `\"[STRING]\"` — see `image-prompting`. Native clip dialogue (`[subject] says \"[LINE]\"`) is Mode A in `video-prompting` — not this skill.\n\n## Human narrator defaults\n\n```text\nWarm storybook narrator, gentle pace, empathetic, no announcer voice.\n```\n\n```text\nNatural documentary host, measured pacing, clear consonants, calm authority.\n```\n\nAvoid: radio-ad hype, “cinematic trailer voice”, reading the product brief into `prompt`.\n\n## Duration gate for `p-video`\n\nAudio-led clips cap at **20s** (keep TTS ≤ **~19s**). If `ffprobe` is long:\n\n1. Shorten `text`  \n2. Add pace to `prompt`: `brisk pace, ~2.3 words per second, no filler`  \n3. Split into two scene rows  \n\nNever rely on post-mux over silent video.\n\n## Avatar vs TTS\n\n| Path | Fields |\n|------|--------|\n| Narrator B-roll | Gemini TTS → `p-video` `input.audio` |\n| On-camera speaker | `p-video-avatar` `voice_script` + `voice_prompt` — **not** this TTS `prompt` |\n\nDo not paste VO into avatar `voice_prompt`.\n\n## Pre-send\n\n- [ ] `prompt` / `text` / tags aligned  \n- [ ] Length probed for `p-video`  \n- [ ] Voice + language recorded in manifest for regen consistency\n\n## Worked example — explainer narration (aligned channels)\n\nUser lock: documentary explainer, **measured** pace, ~45s script, will feed `p-video` Mode B.\n\n**`prompt`** (director style):\n\n```text\nNatural documentary host, measured pacing, clear consonants, calm authority, empathetic, no radio-ad hype\n```\n\n**`text`** (spoken words + inline tags):\n\n```text\n[warm] Most teams treat diversity as a checkbox.\n[pause] But the ritual seed is what breaks repetition before you ever hit generate.\n[emphasis] Lock the brief first — then rotate free axes.\n[measured] Same subject, fresh camera and light, every panel.\n```\n\n**Alignment check:** tags match the calm documentary `prompt` — no `[shout]` hype against a gentle director line.\n\n**Duration gate:** 45s exceeds single **~19s** `p-video` audio-led cap → split into **three** scene rows (~15s each) or shorten copy; probe with `ffprobe` after TTS. Never plan one 45s embed clip.\n\n**Avatar redirect:** on-camera host speaking to lens → `p-video-avatar` `voice_script` + `voice_prompt` — do **not** paste this TTS `prompt` into avatar fields.\n\nFile v1.0.12:skill-card.md\n\n## Description:\n\nUse when crafting TTS, music, or bed prompts for any generative audio model - director style, song structure, and post-production layering.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[pruna-ai](https://clawhub.ai/user/pruna-ai)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers, creators, and agent operators use this skill to draft and review prompts for speech narration, generated songs, instrumental beds, and audio/video layering workflows.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill includes repeated unpinned remote npx install commands, including a full-suite install.\n\nMitigation: Review install commands before execution, prefer pinned versions or reviewed commits, and avoid the full-suite install unless it is required.\n\nRisk: Audio, prompts, or media may be sent to Pruna or Replicate services and may incur provider handling or usage charges.\n\nMitigation: Do not upload confidential material unless the user accepts provider handling and costs; run workflows with minimal credentials.\n\nRisk: Generated audio or video guidance may be misleading if model limits, timing caps, or mixing assumptions are not checked.\n\nMitigation: Probe audio duration, review prompts and outputs before deployment, and follow the skill's guidance to split or shorten clips when limits apply.\n\n## Reference(s):\n\n- [TTS style prompting](references/tts-style-prompting.md)\n- [Music and bed prompting](references/music-and-bed-prompting.md)\n- [Audio post-production](references/audio-post-production.md)\n- [ClawHub skill page](https://clawhub.ai/pruna-ai/skills/audio-prompting)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown guidance with prompt examples, decision tables, and inline shell commands]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May include model-selection guidance, intake questions, prompt drafts, lyrics, audio-layering plans, and install commands for related skills.]\n\n## Skill Version(s):\n\n1.0.12 (source: evidence.release.version and SKILL.md metadata.version)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.12:skill.manifest.json\n\n{\n  \"references\": [\n    \"tts-style-prompting.md\",\n    \"music-and-bed-prompting.md\",\n    \"audio-post-production.md\"\n  ]\n}\n\nArchive v1.0.11: 7 files, 10811 bytes\n\nFiles: references/audio-post-production.md (7604b), references/music-and-bed-prompting.md (3709b), references/tts-style-prompting.md (2888b), skill-card.md (2168b), skill.manifest.json (121b), SKILL.md (6746b), _meta.json (135b)\n\nFile v1.0.11:SKILL.md\n\n---\nname: audio-prompting\ndescription: Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering.\nlicense: MIT\nmetadata:\n  version: \"1.0.11\"\n  package: pruna-skills\n---\n\n# Audio prompting\n\nVendor-neutral craft for **speech, music, and beds**. Works with Gemini TTS, ElevenLabs, Music 2.5, Stable Audio, Suno, and similar APIs.\n\n## Install\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n\n## When to use\n\n- Director-style TTS prompts and inline performance tags\n- Full songs with vocals vs instrumental beds\n- Choosing when to embed audio in a video model vs mix in post\n- Narration + bed layering pipelines\n\n## Works with\n\nGemini Flash TTS, ElevenLabs, Music 2.5, Stable Audio, Suno, Udio, and other audio models. Pair with `video-prompting` when uploading VO into a video model.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `music-2.5` | Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. | `npx skills add PrunaAI/pruna-skills@music-2.5 -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n| `gemini-3.1-flash-tts` | Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. | `npx skills add PrunaAI/pruna-skills@gemini-3.1-flash-tts -y` |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `image-prompting` | Use when crafting still-image prompts for any generative model — composition, identity sheets, edits, try-on, and photoreal personas. | `npx skills add PrunaAI/pruna-skills@image-prompting -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\n## Guide habit\n\nIn the **first reply**, name `` `audio-prompting` `` in backticks. When VO vs bed vs full song, locale, or embed-vs-post are open, open intake → **`generation-diversity`** clarification intake. For embed-vs-post questions, cite [audio-post-production.md](./references/audio-post-production.md) — prefer embed in the video model; post-mux only as fallback.\n\n## Before generating\n\n1. Follow `generation-diversity` first.\n2. TTS → [tts-style-prompting.md](./references/tts-style-prompting.md).\n3. Songs / beds → [music-and-bed-prompting.md](./references/music-and-bed-prompting.md).\n4. Tool picker + layering → [audio-post-production.md](./references/audio-post-production.md).\n\n## Song structure (vocals)\n\nOriginal tracks with sung vocals → **`music-2.5`**, not Stable Audio. In the **first reply**, say you will draft **lyrics** (verse / chorus / bridge as needed) plus a separate **music** style prompt per [music-and-bed-prompting.md](./references/music-and-bed-prompting.md). Stable Audio is instrumental beds only — **not** for sung vocals. See **Worked examples** in that reference for full lyrics + style samples.\n\n## Layered explainer audio\n\nVO + bed: TTS (`gemini-3.1-flash-tts`) for narration; Stable Audio for **instrumental** underscore only. When asked embed vs post: prefer **embed** in `p-video`; **post-mux** / **assembly** mix under VO is fallback only — [audio-post-production.md](./references/audio-post-production.md).\n\n## Related skills\n\nInstall related skills when the job needs them:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\n## Pruna / Replicate tools\n\nMatching install for every model named above. Pick what you need:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `gemini-3.1-flash-tts` | Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. | `npx skills add PrunaAI/pruna-skills@gemini-3.1-flash-tts -y` |\n| `music-2.5` | Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. | `npx skills add PrunaAI/pruna-skills@music-2.5 -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n| `whisperx` | Use when someone needs word-level timestamps from audio — lyric alignment, cut-safe line boundaries, or caption source timing before burn-in with video-editing. | `npx skills add PrunaAI/pruna-skills@whisperx -y` |\n| `p-video` | Use when someone wants one short video clip from text or images — B-roll, start/end frame animation, or a quick motion shot. Not for full multi-scene films or lip-synced hosts. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFile v1.0.11:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"audio-prompting\",\n  \"version\": \"1.0.11\",\n  \"publishedAt\": 1788444407943\n}\n\nFile v1.0.11:references/audio-post-production.md\n\n# Audio post-production (Pruna + Replicate)\n\nHow to choose and **layer** audio when building reels, multi-scene films, and launch videos.\n\n**Prompt craft (how to write):** [tts-style-prompting.md](./tts-style-prompting.md) · [music-and-bed-prompting.md](./music-and-bed-prompting.md). For in-video audio modes and talking-head VO, install and follow `video-prompting`.\n\n**Multi-scene narrated films:** scene anchor triple lives in `video-prompting` — pass TTS to **`p-video`** as `input.audio` with `image` + `last_frame_image`; do not post-mux unless re-render is impossible. Workflow: `narrated-multi-scene`.\n\n**Visual-only transitions (no VO):** scene anchor pair in `video-prompting` — `duration` instead of `audio`. Workflow: `visual-transition-reel`.\n\n## Audio-led `p-video` (required when VO/narration exists)\n\nWhen narration, TTS, or a timed audio slice is available **before** video render:\n\n1. Upload the audio file to Pruna (`POST /v1/files`) — see `pruna-api`.\n2. Pass `urls.get` as **`input.audio`** on **`p-video`** (or **`p-video-avatar`** for human lip-sync).\n3. **Omit `duration`** — clip length follows the audio (capped at **20s** on P-API); the model syncs motion to speech.\n4. Set **`save_audio`: true** so the full line is embedded in the output clip.\n5. **Probe TTS length** before render — per-scene lines should be **≤ ~19s** or the API truncates the tail even when `audio` is set.\n6. **Concat** clips in order (narration already on each clip). Optional bed mixed **under** VO in post.\n\n**Never** generate silent `p-video` and ffmpeg-mux narration afterward unless re-render is impossible — post-mux **truncates** lines longer than the video slot (common with Gemini TTS).\n\n**Over 20s?** Shorten scene copy → tighten TTS pace in `style_prompt` → split into two scene rows (each with its own triple). See `narrated-multi-scene` duration gate.\n\n| Need | Approach | Skill |\n|------|----------|-------|\n| Lip-sync / duration locked to VO | Upload audio → `p-video` with `audio` | `p-video` |\n| Documentary / story narrator | Gemini Flash TTS → upload → video | `gemini-3.1-flash-tts` |\n| Light instrumental under dialogue | Stable Audio bed under VO | `stable-audio-2.5` |\n| Full song with sung vocals | Music 2.5 track | `music-2.5` |\n| Speaking on-camera character | Portrait + script / audio | `p-video-avatar` |\n\n**Env:** Pruna calls need `PRUNA_API_KEY`; Replicate audio tools need `REPLICATE_API_TOKEN`. Assembly steps need **`ffmpeg`** / **`ffprobe`**. Credentials: `pruna-api`. Shared ffmpeg recipes (concat, captions, bed mix, export): **`video-editing`**.\n\n## Layering matrix\n\n| Stack | Primary audio | Secondary | Mix notes |\n|-------|---------------|-----------|-----------|\n| **Silent B-roll** | — | — | Concat video only |\n| **Native `p-video` sound** | Model output | — | Keep `save_audio` default; normalize in assembly if scenes differ |\n| **Narration only (fallback)** | Gemini TTS | — | Post-mux only when audio-led `p-video` is not suitable — prefer **Pipeline A** below |\n| **Bed only** | Stable Audio bed | — | Often with `music-video` or reel beds |\n\n## Recommended pipelines\n\n### A — Narrated multi-scene B-roll (**preferred — scene anchor triple**)\n\nUse `narrated-multi-scene` + `video-prompting` (triple) + tools below.\n\n```text\nPhase 0 — intake: scene table with start/end still prompts + narration lines\nPhase 1 — hero + p-image-edit start stills + end stills (parallel)\nPhase 2 — Gemini TTS per scene (parallel) → upload each to /v1/files\nPhase 3 — p-video per scene: input.image + input.last_frame_image + input.audio (parallel; omit duration)\nPhase 4 — ffmpeg concat (VO embedded; frame chain via shared end/start URLs)\nPhase 5 — optional Stable Audio bed under narration\n```\n\n**Scene anchor triple:** same pattern as first/last frame pairing — `audio` is the third required upload per scene row. **`p-video-avatar`:** portrait + optional `last_frame_image` + uploaded `audio`.\n\n### A′ — Post-mux narration (fallback only)\n\nUse only when you already have silent clips and cannot re-render. Risk: TTS longer than clip slots → cut-off VO.\n\n```text\nPhase 3 — p-video I2V without audio → concat → mux TTS in ffmpeg\n```\n\n### C — Launch / product reel (existing pattern)\n\n```text\nPhase 1 — p-video-avatar or replace reel → concat\nPhase 2 — Stable Audio bed via stable-audio-2.5 + ffmpeg bed mix (bed under VO, not replacing it) — mix recipe in `video-editing`\n```\n\n## ffmpeg mixing (conceptual)\n\nFull mix commands and default launch bed level (~**0.20** under clear promo speech; ~0.08–0.12 under soft narration): install **`video-editing`**.\n\n**Narration onto silent concat** (single VO file):\n\n```bash\nffmpeg -y -i concat_video.mp4 -i narration.mp3 \\\n  -map 0:v -map 1:a -c:v copy -c:a aac -b:a 192k -shortest output_with_vo.mp4\n```\n\n**Bed under existing narration + video** (same pattern as `stable-audio-2.5` + ffmpeg bed mix):\n\n```text\n[1:a]volume=0.12,aloop=...[bed];\n[0:a][bed]amix=inputs=2:duration=first[aout]\n```\n\nNarration / avatar dialogue stays on stream `0:a`; bed is stream `1:a` at low volume.\n\n**Bed on silent concat** — loop a short generated clip to full video length (no per-assemble Stable Audio call):\n\n```text\n[1:a]volume=0.12,aloop=loop=-1:size=2e+09[bed]  →  map video + [bed], -shortest\n```\n\nPlan field `\"reuse_bed\": true` skips regeneration when `audio/launch_bed.mp3` exists. Delete that file (or set `reuse_bed: false`) only when you want a new prompt or seed.\n\n## Intake questions (audio)\n\nAsk before generating paid audio or video:\n\n| Topic | Questions |\n|-------|-----------|\n| **Primary voice** | Narrator (`gemini-3.1-flash-tts`), on-screen avatar (`p-video-avatar`), or native `p-video` sound only? |\n| **Narration scope** | Per-scene lines vs one continuous VO track? |\n| **Music / bed** | None, instrumental bed only (`stable-audio-2.5`), or full song (`music-2.5`)? |\n| **Sync strategy** | **Preferred:** TTS → Pruna upload → **`p-video` / `p-video-avatar` with `audio`** (clip length = audio). Post-mux only as fallback. |\n| **Levels** | Bed volume: default ~**0.20** under launch/promo speech; ~0.08–0.12 under avatar VO or soft Gemini narration — mix recipe in `video-editing` |\n\n## Manifest fields\n\n```json\n{\n  \"narration\": { \"enabled\": true, \"voice\": \"Sulafat\", \"mode\": \"per_scene\" },\n  \"background_music\": { \"enabled\": true, \"reuse_bed\": true, \"volume\": 0.10, \"prompt\": \"Instrumental ... no vocals\" },\n  \"p_video_audio\": { \"save_audio\": true }\n}\n```\n\n## Limitations (P-Video audio)\n\n- Native SFX/dialogue quality varies — for premium voice realism, prefer **`gemini-3.1-flash-tts`** or **`p-video-avatar`**, then optionally mix a bed.\n- Multi-speaker native audio can drift; dedicated TTS per role is safer for narration-heavy cuts.\n- Extreme camera motion and complex multi-scene stories are weaker than **frame-anchored chaining** + per-scene prompts — see `p-video` and `video-prompting`.\n\n## Related skills\n\n| Skill | When |\n|-------|------|\n| `video-prompting` | In-video audio modes, scene anchor pair/triple, talking-head VO craft |\n| `pruna-api` | Upload / poll / parallel batches |\n| `narrated-multi-scene` | Multi-scene B-roll + VO playbook |\n| `visual-transition-reel` | Visual-only transitions (no VO) |\n| `music-video` | Full song + lyric-synced video |\n| `p-video` / `p-video-avatar` | Video API calls that consume uploaded audio |\n| `gemini-3.1-flash-tts` / `stable-audio-2.5` / `music-2.5` | Paid audio generation |\n| `video-editing` | ffmpeg assembly, caption burn-in, bed mix under finished video |\n\nFile v1.0.11:references/music-and-bed-prompting.md\n\n# Music and bed prompting\n\nPrompt craft for `music-2.5` (songs with vocals) and `stable-audio-2.5` (instrumental beds). Mix/stack: [audio-post-production.md](./audio-post-production.md). In-video sync: install `video-prompting`.\n\n## Music 2.5 (full song)\n\nStack: **genre + mood + vocal + tempo + instruments + production feel** (≤ ~2000 chars). Pair with a **lyrics** field — verse / chorus / bridge structure for vocal tracks.\n\n```text\nIndie pop, uplifting, warm female vocal, 92 BPM, acoustic guitar and mellow synth pads, no harsh distortion\n```\n\n**Lyrics:** write singable lines per section (verse, chorus, optional bridge). Lock structure before the paid call; same lyrics + prompt still yield different arrangements — lock seeds only when the user asks.\n\n| Include | Avoid |\n|---------|-------|\n| Genre, BPM, vocal timbre, key instruments | Vague `epic cinematic masterpiece` |\n| Explicit `no harsh distortion` / energy caps when needed | Contradictions (`lo-fi quiet` + `stadium EDM drop`) |\n\n## Stable Audio 2.5 (beds under VO)\n\nInstrumental, understated, mix-friendly:\n\n```text\nInstrumental light electronic pop bed, soft groove and mellow synth pads, calm positive tech atmosphere, understated background music, no vocals, 94 BPM\n```\n\nRules:\n\n- Always **`no vocals`** when under narration  \n- Keep energy **below** dialogue — assembly mixes ~0.08–0.15 under VO  \n- Tag style works well; keep prompts short  \n\n## Which tool?\n\n| Need | Tool |\n|------|------|\n| Sung song / music video source | Music 2.5 |\n| Quiet bed under TTS or avatar | Stable Audio 2.5 |\n| Diegetic SFX inside `p-video` | Native `save_audio` / prompt cues — not these models |\n\n## Pre-send\n\n- [ ] Song vs bed chosen deliberately  \n- [ ] Bed: no vocals + BPM + understated  \n- [ ] Song: genre/mood/vocal/tempo + **lyrics** structure present  \n- [ ] Duration matches scene or assembly plan\n\n## Worked examples\n\n### Full song (Music 2.5) — indie pop, remote-work theme\n\nUser lock: warm female vocal, ~92 BPM, acoustic + mellow synth, **not** EDM drop.\n\n**Style prompt** (`prompt` field):\n\n```text\nIndie pop, warm and hopeful, female vocal, 92 BPM, acoustic guitar and mellow synth pads, intimate bedroom-production feel, no harsh distortion, no stadium drop\n```\n\n**Lyrics** (`lyrics` field — verse / chorus / bridge):\n\n```text\n[Verse 1]\nCoffee rings on the desk again\nWindow light on a second screen\nSlack pings like a metronome\nBuilding something from my home\n\n[Chorus]\nWe're still here, we're still on\nPixels bridge what miles have drawn\nHeart in the work, voice in the song\nRemote but never alone\n\n[Verse 2]\nCat walks across the keyboard line\nDeadline hums but the team's aligned\nSame sky, different time zones\nSame goal in our headphones\n\n[Bridge]\nWhen the Wi‑Fi stutters, we don't fold\nCall reconnects — the story holds\n\n[Chorus]\nWe're still here, we're still on\n...\n```\n\nConfirm lyrics + style before `POST`. For music-video cut points later → `whisperx` after the track exists.\n\n### Instrumental bed (Stable Audio 2.5) — explainer under VO\n\nUser lock: **90s**, calm tech explainer, dialogue must stay clear.\n\n```text\nInstrumental light electronic pop bed, soft groove and mellow synth pads, calm positive tech atmosphere, understated background music, no vocals, 94 BPM\n```\n\nDuration: `90` seconds. Mix target ~0.08–0.15 under narration in assembly — see [audio-post-production.md](./audio-post-production.md).\n\n### Wrong tool check\n\n| User ask | Tool |\n|----------|------|\n| \"Sing an original chorus about launch day\" | Music 2.5 + **lyrics** |\n| \"Quiet underscore while the host talks\" | Stable Audio + **no vocals** |\n| \"Replace the sung hook with spoken VO\" | Gemini TTS — not Stable Audio |\n\nFile v1.0.11:references/tts-style-prompting.md\n\n# TTS style prompting (Gemini 3.1 Flash TTS)\n\nDirector-style `prompt` craft for `gemini-3.1-flash-tts`. Upload results to Pruna for Mode B in-video audio (install `video-prompting`). Layering: [audio-post-production.md](./audio-post-production.md).\n\n## Align three channels\n\n| Channel | Role |\n|---------|------|\n| `text` | Spoken words (+ optional inline `[tags]`) |\n| `prompt` | Tone, pace, accent, character — max ~4k bytes |\n| `[tags]` in text | Momentary direction matching `prompt` |\n\nAll three must point the **same** emotional direction.\n\n**Bracket clarity:** `[tags]` live only in this TTS `text` field. Still typography uses double-quoted `\"[STRING]\"` — see `image-prompting`. Native clip dialogue (`[subject] says \"[LINE]\"`) is Mode A in `video-prompting` — not this skill.\n\n## Human narrator defaults\n\n```text\nWarm storybook narrator, gentle pace, empathetic, no announcer voice.\n```\n\n```text\nNatural documentary host, measured pacing, clear consonants, calm authority.\n```\n\nAvoid: radio-ad hype, “cinematic trailer voice”, reading the product brief into `prompt`.\n\n## Duration gate for `p-video`\n\nAudio-led clips cap at **20s** (keep TTS ≤ **~19s**). If `ffprobe` is long:\n\n1. Shorten `text`  \n2. Add pace to `prompt`: `brisk pace, ~2.3 words per second, no filler`  \n3. Split into two scene rows  \n\nNever rely on post-mux over silent video.\n\n## Avatar vs TTS\n\n| Path | Fields |\n|------|--------|\n| Narrator B-roll | Gemini TTS → `p-video` `input.audio` |\n| On-camera speaker | `p-video-avatar` `voice_script` + `voice_prompt` — **not** this TTS `prompt` |\n\nDo not paste VO into avatar `voice_prompt`.\n\n## Pre-send\n\n- [ ] `prompt` / `text` / tags aligned  \n- [ ] Length probed for `p-video`  \n- [ ] Voice + language recorded in manifest for regen consistency\n\n## Worked example — explainer narration (aligned channels)\n\nUser lock: documentary explainer, **measured** pace, ~45s script, will feed `p-video` Mode B.\n\n**`prompt`** (director style):\n\n```text\nNatural documentary host, measured pacing, clear consonants, calm authority, empathetic, no radio-ad hype\n```\n\n**`text`** (spoken words + inline tags):\n\n```text\n[warm] Most teams treat diversity as a checkbox.\n[pause] But the ritual seed is what breaks repetition before you ever hit generate.\n[emphasis] Lock the brief first — then rotate free axes.\n[measured] Same subject, fresh camera and light, every panel.\n```\n\n**Alignment check:** tags match the calm documentary `prompt` — no `[shout]` hype against a gentle director line.\n\n**Duration gate:** 45s exceeds single **~19s** `p-video` audio-led cap → split into **three** scene rows (~15s each) or shorten copy; probe with `ffprobe` after TTS. Never plan one 45s embed clip.\n\n**Avatar redirect:** on-camera host speaking to lens → `p-video-avatar` `voice_script` + `voice_prompt` — do **not** paste this TTS `prompt` into avatar fields.\n\nFile v1.0.11:skill-card.md\n\n## Description:\n\nUse when crafting TTS, music, or bed prompts for any generative audio model - director style, song structure, and post-production layering.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[pruna-ai](https://clawhub.ai/user/pruna-ai)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and creative agents use this skill to draft TTS, song, instrumental bed, and audio-layering prompts for generative audio and video workflows.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Unpinned install commands can fetch a moving package reference.\n\nMitigation: Review install commands before use and prefer trusted or pinned package references where available.\n\nRisk: Audio workflows can send voice or audio content to Pruna or Replicate and may incur paid API usage.\n\nMitigation: Confirm before sending audio, avoid sensitive voice or audio content unless intended, and verify credential and cost expectations.\n\nRisk: Post-muxing narration over silent video can truncate voiceover that is longer than the video slot.\n\nMitigation: Prefer audio-led video generation when narration exists, then probe, shorten, or split TTS lines to fit per-clip duration limits.\n\n## Reference(s):\n\n- [TTS Style Prompting](references/tts-style-prompting.md)\n- [Music and Bed Prompting](references/music-and-bed-prompting.md)\n- [Audio Post-Production](references/audio-post-production.md)\n\n## Skill Output:\n\n**Output Type(s):** [Guidance, Text, Markdown, Shell commands, Configuration]\n\n**Output Format:** [Markdown guidance with prompt examples, inline shell commands, and configuration snippets]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May include prompts for TTS, music, instrumental beds, narration timing, and audio layering decisions.]\n\n## Skill Version(s):\n\n1.0.11 (source: release evidence and frontmatter)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.11:skill.manifest.json\n\n{\n  \"references\": [\n    \"tts-style-prompting.md\",\n    \"music-and-bed-prompting.md\",\n    \"audio-post-production.md\"\n  ]\n}\n\nArchive v1.0.10: 7 files, 10854 bytes\n\nFiles: references/audio-post-production.md (7604b), references/music-and-bed-prompting.md (3709b), references/tts-style-prompting.md (2888b), skill-card.md (2276b), skill.manifest.json (121b), SKILL.md (6746b), _meta.json (135b)\n\nFile v1.0.10:SKILL.md\n\n---\nname: audio-prompting\ndescription: Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering.\nlicense: MIT\nmetadata:\n  version: \"1.0.10\"\n  package: pruna-skills\n---\n\n# Audio prompting\n\nVendor-neutral craft for **speech, music, and beds**. Works with Gemini TTS, ElevenLabs, Music 2.5, Stable Audio, Suno, and similar APIs.\n\n## Install\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n\n## When to use\n\n- Director-style TTS prompts and inline performance tags\n- Full songs with vocals vs instrumental beds\n- Choosing when to embed audio in a video model vs mix in post\n- Narration + bed layering pipelines\n\n## Works with\n\nGemini Flash TTS, ElevenLabs, Music 2.5, Stable Audio, Suno, Udio, and other audio models. Pair with `video-prompting` when uploading VO into a video model.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `music-2.5` | Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. | `npx skills add PrunaAI/pruna-skills@music-2.5 -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n| `gemini-3.1-flash-tts` | Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. | `npx skills add PrunaAI/pruna-skills@gemini-3.1-flash-tts -y` |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `image-prompting` | Use when crafting still-image prompts for any generative model — composition, identity sheets, edits, try-on, and photoreal personas. | `npx skills add PrunaAI/pruna-skills@image-prompting -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\n## Guide habit\n\nIn the **first reply**, name `` `audio-prompting` `` in backticks. When VO vs bed vs full song, locale, or embed-vs-post are open, open intake → **`generation-diversity`** clarification intake. For embed-vs-post questions, cite [audio-post-production.md](./references/audio-post-production.md) — prefer embed in the video model; post-mux only as fallback.\n\n## Before generating\n\n1. Follow `generation-diversity` first.\n2. TTS → [tts-style-prompting.md](./references/tts-style-prompting.md).\n3. Songs / beds → [music-and-bed-prompting.md](./references/music-and-bed-prompting.md).\n4. Tool picker + layering → [audio-post-production.md](./references/audio-post-production.md).\n\n## Song structure (vocals)\n\nOriginal tracks with sung vocals → **`music-2.5`**, not Stable Audio. In the **first reply**, say you will draft **lyrics** (verse / chorus / bridge as needed) plus a separate **music** style prompt per [music-and-bed-prompting.md](./references/music-and-bed-prompting.md). Stable Audio is instrumental beds only — **not** for sung vocals. See **Worked examples** in that reference for full lyrics + style samples.\n\n## Layered explainer audio\n\nVO + bed: TTS (`gemini-3.1-flash-tts`) for narration; Stable Audio for **instrumental** underscore only. When asked embed vs post: prefer **embed** in `p-video`; **post-mux** / **assembly** mix under VO is fallback only — [audio-post-production.md](./references/audio-post-production.md).\n\n## Related skills\n\nInstall related skills when the job needs them:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\n## Pruna / Replicate tools\n\nMatching install for every model named above. Pick what you need:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `gemini-3.1-flash-tts` | Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. | `npx skills add PrunaAI/pruna-skills@gemini-3.1-flash-tts -y` |\n| `music-2.5` | Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. | `npx skills add PrunaAI/pruna-skills@music-2.5 -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n| `whisperx` | Use when someone needs word-level timestamps from audio — lyric alignment, cut-safe line boundaries, or caption source timing before burn-in with video-editing. | `npx skills add PrunaAI/pruna-skills@whisperx -y` |\n| `p-video` | Use when someone wants one short video clip from text or images — B-roll, start/end frame animation, or a quick motion shot. Not for full multi-scene films or lip-synced hosts. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFile v1.0.10:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"audio-prompting\",\n  \"version\": \"1.0.10\",\n  \"publishedAt\": 1787903583864\n}\n\nFile v1.0.10:references/audio-post-production.md\n\n# Audio post-production (Pruna + Replicate)\n\nHow to choose and **layer** audio when building reels, multi-scene films, and launch videos.\n\n**Prompt craft (how to write):** [tts-style-prompting.md](./tts-style-prompting.md) · [music-and-bed-prompting.md](./music-and-bed-prompting.md). For in-video audio modes and talking-head VO, install and follow `video-prompting`.\n\n**Multi-scene narrated films:** scene anchor triple lives in `video-prompting` — pass TTS to **`p-video`** as `input.audio` with `image` + `last_frame_image`; do not post-mux unless re-render is impossible. Workflow: `narrated-multi-scene`.\n\n**Visual-only transitions (no VO):** scene anchor pair in `video-prompting` — `duration` instead of `audio`. Workflow: `visual-transition-reel`.\n\n## Audio-led `p-video` (required when VO/narration exists)\n\nWhen narration, TTS, or a timed audio slice is available **before** video render:\n\n1. Upload the audio file to Pruna (`POST /v1/files`) — see `pruna-api`.\n2. Pass `urls.get` as **`input.audio`** on **`p-video`** (or **`p-video-avatar`** for human lip-sync).\n3. **Omit `duration`** — clip length follows the audio (capped at **20s** on P-API); the model syncs motion to speech.\n4. Set **`save_audio`: true** so the full line is embedded in the output clip.\n5. **Probe TTS length** before render — per-scene lines should be **≤ ~19s** or the API truncates the tail even when `audio` is set.\n6. **Concat** clips in order (narration already on each clip). Optional bed mixed **under** VO in post.\n\n**Never** generate silent `p-video` and ffmpeg-mux narration afterward unless re-render is impossible — post-mux **truncates** lines longer than the video slot (common with Gemini TTS).\n\n**Over 20s?** Shorten scene copy → tighten TTS pace in `style_prompt` → split into two scene rows (each with its own triple). See `narrated-multi-scene` duration gate.\n\n| Need | Approach | Skill |\n|------|----------|-------|\n| Lip-sync / duration locked to VO | Upload audio → `p-video` with `audio` | `p-video` |\n| Documentary / story narrator | Gemini Flash TTS → upload → video | `gemini-3.1-flash-tts` |\n| Light instrumental under dialogue | Stable Audio bed under VO | `stable-audio-2.5` |\n| Full song with sung vocals | Music 2.5 track | `music-2.5` |\n| Speaking on-camera character | Portrait + script / audio | `p-video-avatar` |\n\n**Env:** Pruna calls need `PRUNA_API_KEY`; Replicate audio tools need `REPLICATE_API_TOKEN`. Assembly steps need **`ffmpeg`** / **`ffprobe`**. Credentials: `pruna-api`. Shared ffmpeg recipes (concat, captions, bed mix, export): **`video-editing`**.\n\n## Layering matrix\n\n| Stack | Primary audio | Secondary | Mix notes |\n|-------|---------------|-----------|-----------|\n| **Silent B-roll** | — | — | Concat video only |\n| **Native `p-video` sound** | Model output | — | Keep `save_audio` default; normalize in assembly if scenes differ |\n| **Narration only (fallback)** | Gemini TTS | — | Post-mux only when audio-led `p-video` is not suitable — prefer **Pipeline A** below |\n| **Bed only** | Stable Audio bed | — | Often with `music-video` or reel beds |\n\n## Recommended pipelines\n\n### A — Narrated multi-scene B-roll (**preferred — scene anchor triple**)\n\nUse `narrated-multi-scene` + `video-prompting` (triple) + tools below.\n\n```text\nPhase 0 — intake: scene table with start/end still prompts + narration lines\nPhase 1 — hero + p-image-edit start stills + end stills (parallel)\nPhase 2 — Gemini TTS per scene (parallel) → upload each to /v1/files\nPhase 3 — p-video per scene: input.image + input.last_frame_image + input.audio (parallel; omit duration)\nPhase 4 — ffmpeg concat (VO embedded; frame chain via shared end/start URLs)\nPhase 5 — optional Stable Audio bed under narration\n```\n\n**Scene anchor triple:** same pattern as first/last frame pairing — `audio` is the third required upload per scene row. **`p-video-avatar`:** portrait + optional `last_frame_image` + uploaded `audio`.\n\n### A′ — Post-mux narration (fallback only)\n\nUse only when you already have silent clips and cannot re-render. Risk: TTS longer than clip slots → cut-off VO.\n\n```text\nPhase 3 — p-video I2V without audio → concat → mux TTS in ffmpeg\n```\n\n### C — Launch / product reel (existing pattern)\n\n```text\nPhase 1 — p-video-avatar or replace reel → concat\nPhase 2 — Stable Audio bed via stable-audio-2.5 + ffmpeg bed mix (bed under VO, not replacing it) — mix recipe in `video-editing`\n```\n\n## ffmpeg mixing (conceptual)\n\nFull mix commands and default launch bed level (~**0.20** under clear promo speech; ~0.08–0.12 under soft narration): install **`video-editing`**.\n\n**Narration onto silent concat** (single VO file):\n\n```bash\nffmpeg -y -i concat_video.mp4 -i narration.mp3 \\\n  -map 0:v -map 1:a -c:v copy -c:a aac -b:a 192k -shortest output_with_vo.mp4\n```\n\n**Bed under existing narration + video** (same pattern as `stable-audio-2.5` + ffmpeg bed mix):\n\n```text\n[1:a]volume=0.12,aloop=...[bed];\n[0:a][bed]amix=inputs=2:duration=first[aout]\n```\n\nNarration / avatar dialogue stays on stream `0:a`; bed is stream `1:a` at low volume.\n\n**Bed on silent concat** — loop a short generated clip to full video length (no per-assemble Stable Audio call):\n\n```text\n[1:a]volume=0.12,aloop=loop=-1:size=2e+09[bed]  →  map video + [bed], -shortest\n```\n\nPlan field `\"reuse_bed\": true` skips regeneration when `audio/launch_bed.mp3` exists. Delete that file (or set `reuse_bed: false`) only when you want a new prompt or seed.\n\n## Intake questions (audio)\n\nAsk before generating paid audio or video:\n\n| Topic | Questions |\n|-------|-----------|\n| **Primary voice** | Narrator (`gemini-3.1-flash-tts`), on-screen avatar (`p-video-avatar`), or native `p-video` sound only? |\n| **Narration scope** | Per-scene lines vs one continuous VO track? |\n| **Music / bed** | None, instrumental bed only (`stable-audio-2.5`), or full song (`music-2.5`)? |\n| **Sync strategy** | **Preferred:** TTS → Pruna upload → **`p-video` / `p-video-avatar` with `audio`** (clip length = audio). Post-mux only as fallback. |\n| **Levels** | Bed volume: default ~**0.20** under launch/promo speech; ~0.08–0.12 under avatar VO or soft Gemini narration — mix recipe in `video-editing` |\n\n## Manifest fields\n\n```json\n{\n  \"narration\": { \"enabled\": true, \"voice\": \"Sulafat\", \"mode\": \"per_scene\" },\n  \"background_music\": { \"enabled\": true, \"reuse_bed\": true, \"volume\": 0.10, \"prompt\": \"Instrumental ... no vocals\" },\n  \"p_video_audio\": { \"save_audio\": true }\n}\n```\n\n## Limitations (P-Video audio)\n\n- Native SFX/dialogue quality varies — for premium voice realism, prefer **`gemini-3.1-flash-tts`** or **`p-video-avatar`**, then optionally mix a bed.\n- Multi-speaker native audio can drift; dedicated TTS per role is safer for narration-heavy cuts.\n- Extreme camera motion and complex multi-scene stories are weaker than **frame-anchored chaining** + per-scene prompts — see `p-video` and `video-prompting`.\n\n## Related skills\n\n| Skill | When |\n|-------|------|\n| `video-prompting` | In-video audio modes, scene anchor pair/triple, talking-head VO craft |\n| `pruna-api` | Upload / poll / parallel batches |\n| `narrated-multi-scene` | Multi-scene B-roll + VO playbook |\n| `visual-transition-reel` | Visual-only transitions (no VO) |\n| `music-video` | Full song + lyric-synced video |\n| `p-video` / `p-video-avatar` | Video API calls that consume uploaded audio |\n| `gemini-3.1-flash-tts` / `stable-audio-2.5` / `music-2.5` | Paid audio generation |\n| `video-editing` | ffmpeg assembly, caption burn-in, bed mix under finished video |\n\nFile v1.0.10:references/music-and-bed-prompting.md\n\n# Music and bed prompting\n\nPrompt craft for `music-2.5` (songs with vocals) and `stable-audio-2.5` (instrumental beds). Mix/stack: [audio-post-production.md](./audio-post-production.md). In-video sync: install `video-prompting`.\n\n## Music 2.5 (full song)\n\nStack: **genre + mood + vocal + tempo + instruments + production feel** (≤ ~2000 chars). Pair with a **lyrics** field — verse / chorus / bridge structure for vocal tracks.\n\n```text\nIndie pop, uplifting, warm female vocal, 92 BPM, acoustic guitar and mellow synth pads, no harsh distortion\n```\n\n**Lyrics:** write singable lines per section (verse, chorus, optional bridge). Lock structure before the paid call; same lyrics + prompt still yield different arrangements — lock seeds only when the user asks.\n\n| Include | Avoid |\n|---------|-------|\n| Genre, BPM, vocal timbre, key instruments | Vague `epic cinematic masterpiece` |\n| Explicit `no harsh distortion` / energy caps when needed | Contradictions (`lo-fi quiet` + `stadium EDM drop`) |\n\n## Stable Audio 2.5 (beds under VO)\n\nInstrumental, understated, mix-friendly:\n\n```text\nInstrumental light electronic pop bed, soft groove and mellow synth pads, calm positive tech atmosphere, understated background music, no vocals, 94 BPM\n```\n\nRules:\n\n- Always **`no vocals`** when under narration  \n- Keep energy **below** dialogue — assembly mixes ~0.08–0.15 under VO  \n- Tag style works well; keep prompts short  \n\n## Which tool?\n\n| Need | Tool |\n|------|------|\n| Sung song / music video source | Music 2.5 |\n| Quiet bed under TTS or avatar | Stable Audio 2.5 |\n| Diegetic SFX inside `p-video` | Native `save_audio` / prompt cues — not these models |\n\n## Pre-send\n\n- [ ] Song vs bed chosen deliberately  \n- [ ] Bed: no vocals + BPM + understated  \n- [ ] Song: genre/mood/vocal/tempo + **lyrics** structure present  \n- [ ] Duration matches scene or assembly plan\n\n## Worked examples\n\n### Full song (Music 2.5) — indie pop, remote-work theme\n\nUser lock: warm female vocal, ~92 BPM, acoustic + mellow synth, **not** EDM drop.\n\n**Style prompt** (`prompt` field):\n\n```text\nIndie pop, warm and hopeful, female vocal, 92 BPM, acoustic guitar and mellow synth pads, intimate bedroom-production feel, no harsh distortion, no stadium drop\n```\n\n**Lyrics** (`lyrics` field — verse / chorus / bridge):\n\n```text\n[Verse 1]\nCoffee rings on the desk again\nWindow light on a second screen\nSlack pings like a metronome\nBuilding something from my home\n\n[Chorus]\nWe're still here, we're still on\nPixels bridge what miles have drawn\nHeart in the work, voice in the song\nRemote but never alone\n\n[Verse 2]\nCat walks across the keyboard line\nDeadline hums but the team's aligned\nSame sky, different time zones\nSame goal in our headphones\n\n[Bridge]\nWhen the Wi‑Fi stutters, we don't fold\nCall reconnects — the story holds\n\n[Chorus]\nWe're still here, we're still on\n...\n```\n\nConfirm lyrics + style before `POST`. For music-video cut points later → `whisperx` after the track exists.\n\n### Instrumental bed (Stable Audio 2.5) — explainer under VO\n\nUser lock: **90s**, calm tech explainer, dialogue must stay clear.\n\n```text\nInstrumental light electronic pop bed, soft groove and mellow synth pads, calm positive tech atmosphere, understated background music, no vocals, 94 BPM\n```\n\nDuration: `90` seconds. Mix target ~0.08–0.15 under narration in assembly — see [audio-post-production.md](./audio-post-production.md).\n\n### Wrong tool check\n\n| User ask | Tool |\n|----------|------|\n| \"Sing an original chorus about launch day\" | Music 2.5 + **lyrics** |\n| \"Quiet underscore while the host talks\" | Stable Audio + **no vocals** |\n| \"Replace the sung hook with spoken VO\" | Gemini TTS — not Stable Audio |\n\nFile v1.0.10:references/tts-style-prompting.md\n\n# TTS style prompting (Gemini 3.1 Flash TTS)\n\nDirector-style `prompt` craft for `gemini-3.1-flash-tts`. Upload results to Pruna for Mode B in-video audio (install `video-prompting`). Layering: [audio-post-production.md](./audio-post-production.md).\n\n## Align three channels\n\n| Channel | Role |\n|---------|------|\n| `text` | Spoken words (+ optional inline `[tags]`) |\n| `prompt` | Tone, pace, accent, character — max ~4k bytes |\n| `[tags]` in text | Momentary direction matching `prompt` |\n\nAll three must point the **same** emotional direction.\n\n**Bracket clarity:** `[tags]` live only in this TTS `text` field. Still typography uses double-quoted `\"[STRING]\"` — see `image-prompting`. Native clip dialogue (`[subject] says \"[LINE]\"`) is Mode A in `video-prompting` — not this skill.\n\n## Human narrator defaults\n\n```text\nWarm storybook narrator, gentle pace, empathetic, no announcer voice.\n```\n\n```text\nNatural documentary host, measured pacing, clear consonants, calm authority.\n```\n\nAvoid: radio-ad hype, “cinematic trailer voice”, reading the product brief into `prompt`.\n\n## Duration gate for `p-video`\n\nAudio-led clips cap at **20s** (keep TTS ≤ **~19s**). If `ffprobe` is long:\n\n1. Shorten `text`  \n2. Add pace to `prompt`: `brisk pace, ~2.3 words per second, no filler`  \n3. Split into two scene rows  \n\nNever rely on post-mux over silent video.\n\n## Avatar vs TTS\n\n| Path | Fields |\n|------|--------|\n| Narrator B-roll | Gemini TTS → `p-video` `input.audio` |\n| On-camera speaker | `p-video-avatar` `voice_script` + `voice_prompt` — **not** this TTS `prompt` |\n\nDo not paste VO into avatar `voice_prompt`.\n\n## Pre-send\n\n- [ ] `prompt` / `text` / tags aligned  \n- [ ] Length probed for `p-video`  \n- [ ] Voice + language recorded in manifest for regen consistency\n\n## Worked example — explainer narration (aligned channels)\n\nUser lock: documentary explainer, **measured** pace, ~45s script, will feed `p-video` Mode B.\n\n**`prompt`** (director style):\n\n```text\nNatural documentary host, measured pacing, clear consonants, calm authority, empathetic, no radio-ad hype\n```\n\n**`text`** (spoken words + inline tags):\n\n```text\n[warm] Most teams treat diversity as a checkbox.\n[pause] But the ritual seed is what breaks repetition before you ever hit generate.\n[emphasis] Lock the brief first — then rotate free axes.\n[measured] Same subject, fresh camera and light, every panel.\n```\n\n**Alignment check:** tags match the calm documentary `prompt` — no `[shout]` hype against a gentle director line.\n\n**Duration gate:** 45s exceeds single **~19s** `p-video` audio-led cap → split into **three** scene rows (~15s each) or shorten copy; probe with `ffprobe` after TTS. Never plan one 45s embed clip.\n\n**Avatar redirect:** on-camera host speaking to lens → `p-video-avatar` `voice_script` + `voice_prompt` — do **not** paste this TTS `prompt` into avatar fields.\n\nFile v1.0.10:skill-card.md\n\n## Description:\n\nUse when crafting TTS, music, or bed prompts for any generative audio model - director style, song structure, and post-production layering.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[pruna-ai](https://clawhub.ai/user/pruna-ai)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nExternal developers and creators use this skill to draft and review prompt guidance for TTS narration, sung music, instrumental beds, and audio layering decisions across generative audio and video workflows.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Related-skill install suggestions and provider workflows can lead an agent toward additional tools or paid generation services.\n\nMitigation: Review related-skill installation suggestions before accepting them, and only provide API keys or upload audio when those services are intended for the task.\n\nRisk: Audio-led video workflows can truncate narration when generated speech exceeds the target model's clip duration.\n\nMitigation: Probe TTS length before rendering, keep per-scene narration within the documented duration gate, or split long narration into multiple scene rows.\n\n## Reference(s):\n\n- [TTS Style Prompting](references/tts-style-prompting.md)\n- [Music and Bed Prompting](references/music-and-bed-prompting.md)\n- [Audio Post-Production](references/audio-post-production.md)\n- [ClawHub Skill Page](https://clawhub.ai/pruna-ai/skills/audio-prompting)\n- [Pruna AI Publisher Profile](https://clawhub.ai/user/pruna-ai)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, Shell commands, Configuration, Guidance]\n\n**Output Format:** [Markdown with prompt text, checklists, tables, and inline shell commands]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May include TTS style prompts, lyrics, music prompts, narration text, mix settings, and related-skill installation guidance.]\n\n## Skill Version(s):\n\n1.0.10 (source: release evidence and SKILL.md metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.10:skill.manifest.json\n\n{\n  \"references\": [\n    \"tts-style-prompting.md\",\n    \"music-and-bed-prompting.md\",\n    \"audio-post-production.md\"\n  ]\n}\n\nArchive v1.0.9: 7 files, 10829 bytes\n\nFiles: references/audio-post-production.md (7604b), references/music-and-bed-prompting.md (3709b), references/tts-style-prompting.md (2888b), skill-card.md (2207b), skill.manifest.json (121b), SKILL.md (6745b), _meta.json (134b)\n\nFile v1.0.9:SKILL.md\n\n---\nname: audio-prompting\ndescription: Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering.\nlicense: MIT\nmetadata:\n  version: \"1.0.9\"\n  package: pruna-skills\n---\n\n# Audio prompting\n\nVendor-neutral craft for **speech, music, and beds**. Works with Gemini TTS, ElevenLabs, Music 2.5, Stable Audio, Suno, and similar APIs.\n\n## Install\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n\n## When to use\n\n- Director-style TTS prompts and inline performance tags\n- Full songs with vocals vs instrumental beds\n- Choosing when to embed audio in a video model vs mix in post\n- Narration + bed layering pipelines\n\n## Works with\n\nGemini Flash TTS, ElevenLabs, Music 2.5, Stable Audio, Suno, Udio, and other audio models. Pair with `video-prompting` when uploading VO into a video model.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `music-2.5` | Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. | `npx skills add PrunaAI/pruna-skills@music-2.5 -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n| `gemini-3.1-flash-tts` | Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. | `npx skills add PrunaAI/pruna-skills@gemini-3.1-flash-tts -y` |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `image-prompting` | Use when crafting still-image prompts for any generative model — composition, identity sheets, edits, try-on, and photoreal personas. | `npx skills add PrunaAI/pruna-skills@image-prompting -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\n## Guide habit\n\nIn the **first reply**, name `` `audio-prompting` `` in backticks. When VO vs bed vs full song, locale, or embed-vs-post are open, open intake → **`generation-diversity`** clarification intake. For embed-vs-post questions, cite [audio-post-production.md](./references/audio-post-production.md) — prefer embed in the video model; post-mux only as fallback.\n\n## Before generating\n\n1. Follow `generation-diversity` first.\n2. TTS → [tts-style-prompting.md](./references/tts-style-prompting.md).\n3. Songs / beds → [music-and-bed-prompting.md](./references/music-and-bed-prompting.md).\n4. Tool picker + layering → [audio-post-production.md](./references/audio-post-production.md).\n\n## Song structure (vocals)\n\nOriginal tracks with sung vocals → **`music-2.5`**, not Stable Audio. In the **first reply**, say you will draft **lyrics** (verse / chorus / bridge as needed) plus a separate **music** style prompt per [music-and-bed-prompting.md](./references/music-and-bed-prompting.md). Stable Audio is instrumental beds only — **not** for sung vocals. See **Worked examples** in that reference for full lyrics + style samples.\n\n## Layered explainer audio\n\nVO + bed: TTS (`gemini-3.1-flash-tts`) for narration; Stable Audio for **instrumental** underscore only. When asked embed vs post: prefer **embed** in `p-video`; **post-mux** / **assembly** mix under VO is fallback only — [audio-post-production.md](./references/audio-post-production.md).\n\n## Related skills\n\nInstall related skills when the job needs them:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\n## Pruna / Replicate tools\n\nMatching install for every model named above. Pick what you need:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `gemini-3.1-flash-tts` | Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. | `npx skills add PrunaAI/pruna-skills@gemini-3.1-flash-tts -y` |\n| `music-2.5` | Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. | `npx skills add PrunaAI/pruna-skills@music-2.5 -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n| `whisperx` | Use when someone needs word-level timestamps from audio — lyric alignment, cut-safe line boundaries, or caption source timing before burn-in with video-editing. | `npx skills add PrunaAI/pruna-skills@whisperx -y` |\n| `p-video` | Use when someone wants one short video clip from text or images — B-roll, start/end frame animation, or a quick motion shot. Not for full multi-scene films or lip-synced hosts. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFile v1.0.9:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"audio-prompting\",\n  \"version\": \"1.0.9\",\n  \"publishedAt\": 1785824180892\n}\n\nFile v1.0.9:references/audio-post-production.md\n\n# Audio post-production (Pruna + Replicate)\n\nHow to choose and **layer** audio when building reels, multi-scene films, and launch videos.\n\n**Prompt craft (how to write):** [tts-style-prompting.md](./tts-style-prompting.md) · [music-and-bed-prompting.md](./music-and-bed-prompting.md). For in-video audio modes and talking-head VO, install and follow `video-prompting`.\n\n**Multi-scene narrated films:** scene anchor triple lives in `video-prompting` — pass TTS to **`p-video`** as `input.audio` with `image` + `last_frame_image`; do not post-mux unless re-render is impossible. Workflow: `narrated-multi-scene`.\n\n**Visual-only transitions (no VO):** scene anchor pair in `video-prompting` — `duration` instead of `audio`. Workflow: `visual-transition-reel`.\n\n## Audio-led `p-video` (required when VO/narration exists)\n\nWhen narration, TTS, or a timed audio slice is available **before** video render:\n\n1. Upload the audio file to Pruna (`POST /v1/files`) — see `pruna-api`.\n2. Pass `urls.get` as **`input.audio`** on **`p-video`** (or **`p-video-avatar`** for human lip-sync).\n3. **Omit `duration`** — clip length follows the audio (capped at **20s** on P-API); the model syncs motion to speech.\n4. Set **`save_audio`: true** so the full line is embedded in the output clip.\n5. **Probe TTS length** before render — per-scene lines should be **≤ ~19s** or the API truncates the tail even when `audio` is set.\n6. **Concat** clips in order (narration already on each clip). Optional bed mixed **under** VO in post.\n\n**Never** generate silent `p-video` and ffmpeg-mux narration afterward unless re-render is impossible — post-mux **truncates** lines longer than the video slot (common with Gemini TTS).\n\n**Over 20s?** Shorten scene copy → tighten TTS pace in `style_prompt` → split into two scene rows (each with its own triple). See `narrated-multi-scene` duration gate.\n\n| Need | Approach | Skill |\n|------|----------|-------|\n| Lip-sync / duration locked to VO | Upload audio → `p-video` with `audio` | `p-video` |\n| Documentary / story narrator | Gemini Flash TTS → upload → video | `gemini-3.1-flash-tts` |\n| Light instrumental under dialogue | Stable Audio bed under VO | `stable-audio-2.5` |\n| Full song with sung vocals | Music 2.5 track | `music-2.5` |\n| Speaking on-camera character | Portrait + script / audio | `p-video-avatar` |\n\n**Env:** Pruna calls need `PRUNA_API_KEY`; Replicate audio tools need `REPLICATE_API_TOKEN`. Assembly steps need **`ffmpeg`** / **`ffprobe`**. Credentials: `pruna-api`. Shared ffmpeg recipes (concat, captions, bed mix, export): **`video-editing`**.\n\n## Layering matrix\n\n| Stack | Primary audio | Secondary | Mix notes |\n|-------|---------------|-----------|-----------|\n| **Silent B-roll** | — | — | Concat video only |\n| **Native `p-video` sound** | Model output | — | Keep `save_audio` default; normalize in assembly if scenes differ |\n| **Narration only (fallback)** | Gemini TTS | — | Post-mux only when audio-led `p-video` is not suitable — prefer **Pipeline A** below |\n| **Bed only** | Stable Audio bed | — | Often with `music-video` or reel beds |\n\n## Recommended pipelines\n\n### A — Narrated multi-scene B-roll (**preferred — scene anchor triple**)\n\nUse `narrated-multi-scene` + `video-prompting` (triple) + tools below.\n\n```text\nPhase 0 — intake: scene table with start/end still prompts + narration lines\nPhase 1 — hero + p-image-edit start stills + end stills (parallel)\nPhase 2 — Gemini TTS per scene (parallel) → upload each to /v1/files\nPhase 3 — p-video per scene: input.image + input.last_frame_image + input.audio (parallel; omit duration)\nPhase 4 — ffmpeg concat (VO embedded; frame chain via shared end/start URLs)\nPhase 5 — optional Stable Audio bed under narration\n```\n\n**Scene anchor triple:** same pattern as first/last frame pairing — `audio` is the third required upload per scene row. **`p-video-avatar`:** portrait + optional `last_frame_image` + uploaded `audio`.\n\n### A′ — Post-mux narration (fallback only)\n\nUse only when you already have silent clips and cannot re-render. Risk: TTS longer than clip slots → cut-off VO.\n\n```text\nPhase 3 — p-video I2V without audio → concat → mux TTS in ffmpeg\n```\n\n### C — Launch / product reel (existing pattern)\n\n```text\nPhase 1 — p-video-avatar or replace reel → concat\nPhase 2 — Stable Audio bed via stable-audio-2.5 + ffmpeg bed mix (bed under VO, not replacing it) — mix recipe in `video-editing`\n```\n\n## ffmpeg mixing (conceptual)\n\nFull mix commands and default launch bed level (~**0.20** under clear promo speech; ~0.08–0.12 under soft narration): install **`video-editing`**.\n\n**Narration onto silent concat** (single VO file):\n\n```bash\nffmpeg -y -i concat_video.mp4 -i narration.mp3 \\\n  -map 0:v -map 1:a -c:v copy -c:a aac -b:a 192k -shortest output_with_vo.mp4\n```\n\n**Bed under existing narration + video** (same pattern as `stable-audio-2.5` + ffmpeg bed mix):\n\n```text\n[1:a]volume=0.12,aloop=...[bed];\n[0:a][bed]amix=inputs=2:duration=first[aout]\n```\n\nNarration / avatar dialogue stays on stream `0:a`; bed is stream `1:a` at low volume.\n\n**Bed on silent concat** — loop a short generated clip to full video length (no per-assemble Stable Audio call):\n\n```text\n[1:a]volume=0.12,aloop=loop=-1:size=2e+09[bed]  →  map video + [bed], -shortest\n```\n\nPlan field `\"reuse_bed\": true` skips regeneration when `audio/launch_bed.mp3` exists. Delete that file (or set `reuse_bed: false`) only when you want a new prompt or seed.\n\n## Intake questions (audio)\n\nAsk before generating paid audio or video:\n\n| Topic | Questions |\n|-------|-----------|\n| **Primary voice** | Narrator (`gemini-3.1-flash-tts`), on-screen avatar (`p-video-avatar`), or native `p-video` sound only? |\n| **Narration scope** | Per-scene lines vs one continuous VO track? |\n| **Music / bed** | None, instrumental bed only (`stable-audio-2.5`), or full song (`music-2.5`)? |\n| **Sync strategy** | **Preferred:** TTS → Pruna upload → **`p-video` / `p-video-avatar` with `audio`** (clip length = audio). Post-mux only as fallback. |\n| **Levels** | Bed volume: default ~**0.20** under launch/promo speech; ~0.08–0.12 under avatar VO or soft Gemini narration — mix recipe in `video-editing` |\n\n## Manifest fields\n\n```json\n{\n  \"narration\": { \"enabled\": true, \"voice\": \"Sulafat\", \"mode\": \"per_scene\" },\n  \"background_music\": { \"enabled\": true, \"reuse_bed\": true, \"volume\": 0.10, \"prompt\": \"Instrumental ... no vocals\" },\n  \"p_video_audio\": { \"save_audio\": true }\n}\n```\n\n## Limitations (P-Video audio)\n\n- Native SFX/dialogue quality varies — for premium voice realism, prefer **`gemini-3.1-flash-tts`** or **`p-video-avatar`**, then optionally mix a bed.\n- Multi-speaker native audio can drift; dedicated TTS per role is safer for narration-heavy cuts.\n- Extreme camera motion and complex multi-scene stories are weaker than **frame-anchored chaining** + per-scene prompts — see `p-video` and `video-prompting`.\n\n## Related skills\n\n| Skill | When |\n|-------|------|\n| `video-prompting` | In-video audio modes, scene anchor pair/triple, talking-head VO craft |\n| `pruna-api` | Upload / poll / parallel batches |\n| `narrated-multi-scene` | Multi-scene B-roll + VO playbook |\n| `visual-transition-reel` | Visual-only transitions (no VO) |\n| `music-video` | Full song + lyric-synced video |\n| `p-video` / `p-video-avatar` | Video API calls that consume uploaded audio |\n| `gemini-3.1-flash-tts` / `stable-audio-2.5` / `music-2.5` | Paid audio generation |\n| `video-editing` | ffmpeg assembly, caption burn-in, bed mix under finished video |\n\nFile v1.0.9:references/music-and-bed-prompting.md\n\n# Music and bed prompting\n\nPrompt craft for `music-2.5` (songs with vocals) and `stable-audio-2.5` (instrumental beds). Mix/stack: [audio-post-production.md](./audio-post-production.md). In-video sync: install `video-prompting`.\n\n## Music 2.5 (full song)\n\nStack: **genre + mood + vocal + tempo + instruments + production feel** (≤ ~2000 chars). Pair with a **lyrics** field — verse / chorus / bridge structure for vocal tracks.\n\n```text\nIndie pop, uplifting, warm female vocal, 92 BPM, acoustic guitar and mellow synth pads, no harsh distortion\n```\n\n**Lyrics:** write singable lines per section (verse, chorus, optional bridge). Lock structure before the paid call; same lyrics + prompt still yield different arrangements — lock seeds only when the user asks.\n\n| Include | Avoid |\n|---------|-------|\n| Genre, BPM, vocal timbre, key instruments | Vague `epic cinematic masterpiece` |\n| Explicit `no harsh distortion` / energy caps when needed | Contradictions (`lo-fi quiet` + `stadium EDM drop`) |\n\n## Stable Audio 2.5 (beds under VO)\n\nInstrumental, understated, mix-friendly:\n\n```text\nInstrumental light electronic pop bed, soft groove and mellow synth pads, calm positive tech atmosphere, understated background music, no vocals, 94 BPM\n```\n\nRules:\n\n- Always **`no vocals`** when under narration  \n- Keep energy **below** dialogue — assembly mixes ~0.08–0.15 under VO  \n- Tag style works well; keep prompts short  \n\n## Which tool?\n\n| Need | Tool |\n|------|------|\n| Sung song / music video source | Music 2.5 |\n| Quiet bed under TTS or avatar | Stable Audio 2.5 |\n| Diegetic SFX inside `p-video` | Native `save_audio` / prompt cues — not these models |\n\n## Pre-send\n\n- [ ] Song vs bed chosen deliberately  \n- [ ] Bed: no vocals + BPM + understated  \n- [ ] Song: genre/mood/vocal/tempo + **lyrics** structure present  \n- [ ] Duration matches scene or assembly plan\n\n## Worked examples\n\n### Full song (Music 2.5) — indie pop, remote-work theme\n\nUser lock: warm female vocal, ~92 BPM, acoustic + mellow synth, **not** EDM drop.\n\n**Style prompt** (`prompt` field):\n\n```text\nIndie pop, warm and hopeful, female vocal, 92 BPM, acoustic guitar and mellow synth pads, intimate bedroom-production feel, no harsh distortion, no stadium drop\n```\n\n**Lyrics** (`lyrics` field — verse / chorus / bridge):\n\n```text\n[Verse 1]\nCoffee rings on the desk again\nWindow light on a second screen\nSlack pings like a metronome\nBuilding something from my home\n\n[Chorus]\nWe're still here, we're still on\nPixels bridge what miles have drawn\nHeart in the work, voice in the song\nRemote but never alone\n\n[Verse 2]\nCat walks across the keyboard line\nDeadline hums but the team's aligned\nSame sky, different time zones\nSame goal in our headphones\n\n[Bridge]\nWhen the Wi‑Fi stutters, we don't fold\nCall reconnects — the story holds\n\n[Chorus]\nWe're still here, we're still on\n...\n```\n\nConfirm lyrics + style before `POST`. For music-video cut points later → `whisperx` after the track exists.\n\n### Instrumental bed (Stable Audio 2.5) — explainer under VO\n\nUser lock: **90s**, calm tech explainer, dialogue must stay clear.\n\n```text\nInstrumental light electronic pop bed, soft groove and mellow synth pads, calm positive tech atmosphere, understated background music, no vocals, 94 BPM\n```\n\nDuration: `90` seconds. Mix target ~0.08–0.15 under narration in assembly — see [audio-post-production.md](./audio-post-production.md).\n\n### Wrong tool check\n\n| User ask | Tool |\n|----------|------|\n| \"Sing an original chorus about launch day\" | Music 2.5 + **lyrics** |\n| \"Quiet underscore while the host talks\" | Stable Audio + **no vocals** |\n| \"Replace the sung hook with spoken VO\" | Gemini TTS — not Stable Audio |\n\nFile v1.0.9:references/tts-style-prompting.md\n\n# TTS style prompting (Gemini 3.1 Flash TTS)\n\nDirector-style `prompt` craft for `gemini-3.1-flash-tts`. Upload results to Pruna for Mode B in-video audio (install `video-prompting`). Layering: [audio-post-production.md](./audio-post-production.md).\n\n## Align three channels\n\n| Channel | Role |\n|---------|------|\n| `text` | Spoken words (+ optional inline `[tags]`) |\n| `prompt` | Tone, pace, accent, character — max ~4k bytes |\n| `[tags]` in text | Momentary direction matching `prompt` |\n\nAll three must point the **same** emotional direction.\n\n**Bracket clarity:** `[tags]` live only in this TTS `text` field. Still typography uses double-quoted `\"[STRING]\"` — see `image-prompting`. Native clip dialogue (`[subject] says \"[LINE]\"`) is Mode A in `video-prompting` — not this skill.\n\n## Human narrator defaults\n\n```text\nWarm storybook narrator, gentle pace, empathetic, no announcer voice.\n```\n\n```text\nNatural documentary host, measured pacing, clear consonants, calm authority.\n```\n\nAvoid: radio-ad hype, “cinematic trailer voice”, reading the product brief into `prompt`.\n\n## Duration gate for `p-video`\n\nAudio-led clips cap at **20s** (keep TTS ≤ **~19s**). If `ffprobe` is long:\n\n1. Shorten `text`  \n2. Add pace to `prompt`: `brisk pace, ~2.3 words per second, no filler`  \n3. Split into two scene rows  \n\nNever rely on post-mux over silent video.\n\n## Avatar vs TTS\n\n| Path | Fields |\n|------|--------|\n| Narrator B-roll | Gemini TTS → `p-video` `input.audio` |\n| On-camera speaker | `p-video-avatar` `voice_script` + `voice_prompt` — **not** this TTS `prompt` |\n\nDo not paste VO into avatar `voice_prompt`.\n\n## Pre-send\n\n- [ ] `prompt` / `text` / tags aligned  \n- [ ] Length probed for `p-video`  \n- [ ] Voice + language recorded in manifest for regen consistency\n\n## Worked example — explainer narration (aligned channels)\n\nUser lock: documentary explainer, **measured** pace, ~45s script, will feed `p-video` Mode B.\n\n**`prompt`** (director style):\n\n```text\nNatural documentary host, measured pacing, clear consonants, calm authority, empathetic, no radio-ad hype\n```\n\n**`text`** (spoken words + inline tags):\n\n```text\n[warm] Most teams treat diversity as a checkbox.\n[pause] But the ritual seed is what breaks repetition before you ever hit generate.\n[emphasis] Lock the brief first — then rotate free axes.\n[measured] Same subject, fresh camera and light, every panel.\n```\n\n**Alignment check:** tags match the calm documentary `prompt` — no `[shout]` hype against a gentle director line.\n\n**Duration gate:** 45s exceeds single **~19s** `p-video` audio-led cap → split into **three** scene rows (~15s each) or shorten copy; probe with `ffprobe` after TTS. Never plan one 45s embed clip.\n\n**Avatar redirect:** on-camera host speaking to lens → `p-video-avatar` `voice_script` + `voice_prompt` — do **not** paste this TTS `prompt` into avatar fields.\n\nFile v1.0.9:skill-card.md\n\n## Description: <br>\nUse when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[pruna-ai](https://clawhub.ai/user/pruna-ai) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers, creators, and agents use this skill to draft and review prompts for TTS narration, sung songs, instrumental beds, and audio/video layering workflows. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill may steer agents toward companion skill installs, paid media-generation APIs, API tokens, or audio uploads. <br>\nMitigation: Review requested installs, API-token use, uploads, and paid calls before execution; approve only steps that match the project. <br>\nRisk: Audio planning mistakes can truncate narration or make dialogue hard to hear under music beds. <br>\nMitigation: Probe TTS length, keep audio-led clips within documented limits, prefer embedding narration before video render, and review mix levels before publishing. <br>\n\n\n## Reference(s): <br>\n- [TTS style prompting](references/tts-style-prompting.md) <br>\n- [Music and bed prompting](references/music-and-bed-prompting.md) <br>\n- [Audio post-production](references/audio-post-production.md) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [Guidance, Markdown, Shell commands, C\n\nArchive v1.0.8: 7 files, 10806 bytes\n\nFiles: references/audio-post-production.md (7604b), references/music-and-bed-prompting.md (3709b), references/tts-style-prompting.md (2888b), skill-card.md (2218b), skill.manifest.json (121b), SKILL.md (6745b), _meta.json (134b)\n\nArchive v1.0.7: 7 files, 10876 bytes\n\nFiles: references/audio-post-production.md (7604b), references/music-and-bed-prompting.md (3709b), references/tts-style-prompting.md (2888b), skill-card.md (2464b), skill.manifest.json (121b), SKILL.md (6616b), _meta.json (134b)","readmeExcerpt":"Skill: audio-prompting Owner: pruna-ai Summary: Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. Tags: ai:1.0.14, generative:1.0.14, latest:1.0.14, pruna:1.0.14 Version history: v1.0.14 | 2026-09-29T15:31:35.155Z | auto - Updated version metadata to 1.0.14 in SKILL.md. - Removed the skill-card.md file. - No changes to user-faci","codeSnippets":[],"executableExamples":[{"language":"text","snippet":"Phase 0 — intake: scene table with start/end still prompts + narration lines\nPhase 1 — hero + p-image-edit start stills + end stills (parallel)\nPhase 2 — Gemini TTS per scene (parallel) → upload each to /v1/files\nPhase 3 — p-video per scene: input.image + input.last_frame_image + input.audio (parallel; omit duration)\nPhase 4 — ffmpeg concat (VO embedded; frame chain via shared end/start URLs)\nPhase 5 — optional Stable Audio bed under narration"},{"language":"text","snippet":"Phase 3 — p-video I2V without audio → concat → mux TTS in ffmpeg"},{"language":"text","snippet":"Phase 1 — p-video-avatar or replace reel → concat\nPhase 2 — Stable Audio bed via stable-audio-2.5 + ffmpeg bed mix (bed under VO, not replacing it) — mix recipe in `video-editing`"},{"language":"bash","snippet":"ffmpeg -y -i concat_video.mp4 -i narration.mp3 \\\n  -map 0:v -map 1:a -c:v copy -c:a aac -b:a 192k -shortest output_with_vo.mp4"},{"language":"text","snippet":"[1:a]volume=0.12,aloop=...[bed];\n[0:a][bed]amix=inputs=2:duration=first[aout]"},{"language":"text","snippet":"[1:a]volume=0.12,aloop=loop=-1:size=2e+09[bed]  →  map video + [bed], -shortest"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: audio-prompting\ndescription: Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering.\nlicense: MIT\nmetadata:\n  version: \"1.0.14\"\n  package: pruna-skills\n---\n\n# Audio prompting\n\nVendor-neutral craft for **speech, music, and beds**. Works with Gemini TTS, ElevenLabs, Music 2.5, Stable Audio, Suno, and similar APIs.\n\n## Install\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n\n## When to use\n\n- Director-style TTS prompts and inline performance tags\n- Full songs with vocals vs instrumental beds\n- Choosing when to embed audio in a video model vs mix in post\n- Narration + bed layering pipelines\n\n## Works with\n\nGemini Flash TTS, ElevenLabs, Music 2.5, Stable Audio, Suno, Udio, and other audio models. Pair with `video-prompting` when uploading VO into a video model.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `music-2.5` | Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. | `npx skills add PrunaAI/pruna-skills@music-2.5 -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n| `gemini-3.1-flash-tts` | Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. | `npx skills add PrunaAI/pruna-skills@gemini-3.1-flash-tts -y` |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `image-prompting` | Use when crafting still-image prompts for any generative model — composition, identity sheets, edits, try-on, and photoreal personas. | `npx skills add PrunaAI/pruna-skills@image-prompting -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\n## Guide habit\n\nIn the **first reply**, name `` `audio-prompting` `` "},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"audio-prompting\",\n  \"version\": \"1.0.14\",\n  \"publishedAt\": 1790695895155\n}"},{"path":"references/audio-post-production.md","content":"# Audio post-production (Pruna + Replicate)\n\nHow to choose and **layer** audio when building reels, multi-scene films, and launch videos.\n\n**Prompt craft (how to write):** [tts-style-prompting.md](./tts-style-prompting.md) · [music-and-bed-prompting.md](./music-and-bed-prompting.md). For in-video audio modes and talking-head VO, install and follow `video-prompting`.\n\n**Multi-scene narrated films:** scene anchor triple lives in `video-prompting` — pass TTS to **`p-video`** as `input.audio` with `image` + `last_frame_image`; do not post-mux unless re-render is impossible. Workflow: `narrated-multi-scene`.\n\n**Visual-only transitions (no VO):** scene anchor pair in `video-prompting` — `duration` instead of `audio`. Workflow: `visual-transition-reel`.\n\n## Audio-led `p-video` (required when VO/narration exists)\n\nWhen narration, TTS, or a timed audio slice is available **before** video render:\n\n1. Upload the audio file to Pruna (`POST /v1/files`) — see `pruna-api`.\n2. Pass `urls.get` as **`input.audio`** on **`p-video`** (or **`p-video-avatar`** for human lip-sync).\n3. **Omit `duration`** — clip length follows the audio (capped at **20s** on P-API); the model syncs motion to speech.\n4. Set **`save_audio`: true** so the full line is embedded in the output clip.\n5. **Probe TTS length** before render — per-scene lines should be **≤ ~19s** or the API truncates the tail even when `audio` is set.\n6. **Concat** clips in order (narration already on each clip). Optional bed mixed **under** VO in post.\n\n**Never** generate silent `p-video` and ffmpeg-mux narration afterward unless re-render is impossible — post-mux **truncates** lines longer than the video slot (common with Gemini TTS).\n\n**Over 20s?** Shorten scene copy → tighten TTS pace in `style_prompt` → split into two scene rows (each with its own triple). See `narrated-multi-scene` duration gate.\n\n| Need | Approach | Skill |\n|------|----------|-------|\n| Lip-sync / duration locked to VO | Upload audio → `p-video` with `audio` | `p-video` |\n| Documentary / story narrator | Gemini Flash TTS → upload → video | `gemini-3.1-flash-tts` |\n| Light instrumental under dialogue | Stable Audio bed under VO | `stable-audio-2.5` |\n| Full song with sung vocals | Music 2.5 track | `music-2.5` |\n| Speaking on-camera character | Portrait + script / audio | `p-video-avatar` |\n\n**Env:** Pruna calls need `PRUNA_API_KEY`; Replicate audio tools need `REPLICATE_API_TOKEN`. Assembly steps need **`ffmpeg`** / **`ffprobe`**. Credentials: `pruna-api`. Shared ffmpeg recipes (concat, captions, bed mix, export): **`video-editing`**.\n\n## Layering matrix\n\n| Stack | Primary audio | Secondary | Mix notes |\n|-------|---------------|-----------|-----------|\n| **Silent B-roll** | — | — | Concat video only |\n| **Native `p-video` sound** | Model output | — | Keep `save_audio` default; normalize in assembly if scenes differ |\n| **Narration only (fallback)** | Gemini TTS | — | Post-mux only when audio-led `p-video` is not suitable — prefer **Pipelin"},{"path":"references/music-and-bed-prompting.md","content":"# Music and bed prompting\n\nPrompt craft for `music-2.5` (songs with vocals) and `stable-audio-2.5` (instrumental beds). Mix/stack: [audio-post-production.md](./audio-post-production.md). In-video sync: install `video-prompting`.\n\n## Music 2.5 (full song)\n\nStack: **genre + mood + vocal + tempo + instruments + production feel** (≤ ~2000 chars). Pair with a **lyrics** field — verse / chorus / bridge structure for vocal tracks.\n\n```text\nIndie pop, uplifting, warm female vocal, 92 BPM, acoustic guitar and mellow synth pads, no harsh distortion\n```\n\n**Lyrics:** write singable lines per section (verse, chorus, optional bridge). Lock structure before the paid call; same lyrics + prompt still yield different arrangements — lock seeds only when the user asks.\n\n| Include | Avoid |\n|---------|-------|\n| Genre, BPM, vocal timbre, key instruments | Vague `epic cinematic masterpiece` |\n| Explicit `no harsh distortion` / energy caps when needed | Contradictions (`lo-fi quiet` + `stadium EDM drop`) |\n\n## Stable Audio 2.5 (beds under VO)\n\nInstrumental, understated, mix-friendly:\n\n```text\nInstrumental light electronic pop bed, soft groove and mellow synth pads, calm positive tech atmosphere, understated background music, no vocals, 94 BPM\n```\n\nRules:\n\n- Always **`no vocals`** when under narration  \n- Keep energy **below** dialogue — assembly mixes ~0.08–0.15 under VO  \n- Tag style works well; keep prompts short  \n\n## Which tool?\n\n| Need | Tool |\n|------|------|\n| Sung song / music video source | Music 2.5 |\n| Quiet bed under TTS or avatar | Stable Audio 2.5 |\n| Diegetic SFX inside `p-video` | Native `save_audio` / prompt cues — not these models |\n\n## Pre-send\n\n- [ ] Song vs bed chosen deliberately  \n- [ ] Bed: no vocals + BPM + understated  \n- [ ] Song: genre/mood/vocal/tempo + **lyrics** structure present  \n- [ ] Duration matches scene or assembly plan\n\n## Worked examples\n\n### Full song (Music 2.5) — indie pop, remote-work theme\n\nUser lock: warm female vocal, ~92 BPM, acoustic + mellow synth, **not** EDM drop.\n\n**Style prompt** (`prompt` field):\n\n```text\nIndie pop, warm and hopeful, female vocal, 92 BPM, acoustic guitar and mellow synth pads, intimate bedroom-production feel, no harsh distortion, no stadium drop\n```\n\n**Lyrics** (`lyrics` field — verse / chorus / bridge):\n\n```text\n[Verse 1]\nCoffee rings on the desk again\nWindow light on a second screen\nSlack pings like a metronome\nBuilding something from my home\n\n[Chorus]\nWe're still here, we're still on\nPixels bridge what miles have drawn\nHeart in the work, voice in the song\nRemote but never alone\n\n[Verse 2]\nCat walks across the keyboard line\nDeadline hums but the team's aligned\nSame sky, different time zones\nSame goal in our headphones\n\n[Bridge]\nWhen the Wi‑Fi stutters, we don't fold\nCall reconnects — the story holds\n\n[Chorus]\nWe're still here, we're still on\n...\n```\n\nConfirm lyrics + style before `POST`. For music-video cut points later → `whisperx` after the track exists.\n\n### Instrumental bed (Stable Audio 2.5"},{"path":"references/tts-style-prompting.md","content":"# TTS style prompting (Gemini 3.1 Flash TTS)\n\nDirector-style `prompt` craft for `gemini-3.1-flash-tts`. Upload results to Pruna for Mode B in-video audio (install `video-prompting`). Layering: [audio-post-production.md](./audio-post-production.md).\n\n## Align three channels\n\n| Channel | Role |\n|---------|------|\n| `text` | Spoken words (+ optional inline `[tags]`) |\n| `prompt` | Tone, pace, accent, character — max ~4k bytes |\n| `[tags]` in text | Momentary direction matching `prompt` |\n\nAll three must point the **same** emotional direction.\n\n**Bracket clarity:** `[tags]` live only in this TTS `text` field. Still typography uses double-quoted `\"[STRING]\"` — see `image-prompting`. Native clip dialogue (`[subject] says \"[LINE]\"`) is Mode A in `video-prompting` — not this skill.\n\n## Human narrator defaults\n\n```text\nWarm storybook narrator, gentle pace, empathetic, no announcer voice.\n```\n\n```text\nNatural documentary host, measured pacing, clear consonants, calm authority.\n```\n\nAvoid: radio-ad hype, “cinematic trailer voice”, reading the product brief into `prompt`.\n\n## Duration gate for `p-video`\n\nAudio-led clips cap at **20s** (keep TTS ≤ **~19s**). If `ffprobe` is long:\n\n1. Shorten `text`  \n2. Add pace to `prompt`: `brisk pace, ~2.3 words per second, no filler`  \n3. Split into two scene rows  \n\nNever rely on post-mux over silent video.\n\n## Avatar vs TTS\n\n| Path | Fields |\n|------|--------|\n| Narrator B-roll | Gemini TTS → `p-video` `input.audio` |\n| On-camera speaker | `p-video-avatar` `voice_script` + `voice_prompt` — **not** this TTS `prompt` |\n\nDo not paste VO into avatar `voice_prompt`.\n\n## Pre-send\n\n- [ ] `prompt` / `text` / tags aligned  \n- [ ] Length probed for `p-video`  \n- [ ] Voice + language recorded in manifest for regen consistency\n\n## Worked example — explainer narration (aligned channels)\n\nUser lock: documentary explainer, **measured** pace, ~45s script, will feed `p-video` Mode B.\n\n**`prompt`** (director style):\n\n```text\nNatural documentary host, measured pacing, clear consonants, calm authority, empathetic, no radio-ad hype\n```\n\n**`text`** (spoken words + inline tags):\n\n```text\n[warm] Most teams treat diversity as a checkbox.\n[pause] But the ritual seed is what breaks repetition before you ever hit generate.\n[emphasis] Lock the brief first — then rotate free axes.\n[measured] Same subject, fresh camera and light, every panel.\n```\n\n**Alignment check:** tags match the calm documentary `prompt` — no `[shout]` hype against a gentle director line.\n\n**Duration gate:** 45s exceeds single **~19s** `p-video` audio-led cap → split into **three** scene rows (~15s each) or shorten copy; probe with `ffprobe` after TTS. Never plan one 45s embed clip.\n\n**Avatar redirect:** on-camera host speaking to lens → `p-video-avatar` `voice_script` + `voice_prompt` — do **not** paste this TTS `prompt` into avatar fields."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":2057,"uniquenessScore":39,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T17:08:17.830Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T17:08:17.830Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T20:58:01.465Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}