{"id":"42c05c91-b4c2-4417-a22b-9d703390208e","entityType":"agent","slug":"clawhub-pruna-ai-gemini-3-1-flash-tts","name":"gemini-3.1-flash-tts","canonicalUrl":"https://www.xpersona.co/agent/clawhub-pruna-ai-gemini-3-1-flash-tts","canonicalPath":"/agent/clawhub-pruna-ai-gemini-3-1-flash-tts","generatedAt":"2026-10-10T14:50:28.999Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-10T12:32:12.994Z","emptyReason":null},"description":"Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. Skill: gemini-3.1-flash-tts Owner: pruna-ai Summary: Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. Tags: ai:1.0.14, generative:1.0.14, latest:1.0.14, pruna:1.0.14 Version history: v1.0.14 | 2026-09-29T15:35:50.539Z | auto - Bumped skill version to 1.0.14. - Updated version metadata in SKILL.md. - Removed the redundant skill-card.md f","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.4K downloads reported by the source. Last updated 10/10/2026.","installCommand":"clawhub skill install s173djpkkm1x4yfzg522h5vfp989mrn5:gemini-3-1-flash-tts","sourceUrl":"https://clawhub.ai/pruna-ai/gemini-3-1-flash-tts","homepage":"https://clawhub.ai/pruna-ai/skills/gemini-3-1-flash-tts","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/pruna-ai/gemini-3-1-flash-tts","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/pruna-ai/skills/gemini-3-1-flash-tts","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":63,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. Skill: gemini-3.1-flash-tts O"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-10T12:32:12.994Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T12:32:12.994Z","emptyReason":null},"stars":null,"forks":null,"downloads":1431,"packageName":null,"latestVersion":"1.0.14","tractionLabel":"1.4K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T12:32:12.993Z","emptyReason":null},"lastUpdatedAt":"2026-10-10T12:32:12.994Z","lastCrawledAt":"2026-10-10T12:32:12.993Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-11T12:32:12.993Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.14","createdAt":"2026-09-29T15:35:50.539Z","changelog":"- Bumped skill version to 1.0.14. - Updated version metadata in SKILL.md. - Removed the redundant skill-card.md file.","fileCount":4,"zipByteSize":4057},{"version":"1.0.13","createdAt":"2026-09-17T13:58:36.340Z","changelog":"- Bumped version to 1.0.13. - Updated the \"Typical next steps\" section for `p-video` to clarify use cases, adding more detail on limitations and suitable scenarios. - Removed the redundant skill-card.md file.","fileCount":4,"zipByteSize":4226},{"version":"1.0.12","createdAt":"2026-09-10T13:57:08.752Z","changelog":"- Version bump to 1.0.12. - Updated `SKILL.md` to clarify the description of the `p-video` skill in the Typical next steps section. - Removed `skill-card.md` file.","fileCount":4,"zipByteSize":4179},{"version":"1.0.11","createdAt":"2026-09-03T14:11:27.901Z","changelog":"- Bumped version to 1.0.11. - Removed redundant file: skill-card.md. - No user-facing feature changes.","fileCount":4,"zipByteSize":4252},{"version":"1.0.10","createdAt":"2026-08-28T07:57:19.145Z","changelog":"- Updated to version 1.0.10 in SKILL.md metadata. - Removed the skill-card.md file. - No functional or usage changes to guidance or API.","fileCount":4,"zipByteSize":4153},{"version":"1.0.9","createdAt":"2026-08-04T06:18:12.930Z","changelog":"- Version bump to 1.0.9. - Removed redundant file: skill-card.md. - No feature or functionality changes; documentation only.","fileCount":4,"zipByteSize":4153},{"version":"1.0.8","createdAt":"2026-07-28T17:20:17.908Z","changelog":"- Improved agent intake: now opens a clarification step (locale, voice, script) with generation-diversity before first TTS request. - Updated onboarding instructions for first reply behavior and intake flow. - Minor clarifications in agent prompts and user guidance. - Removed redundant file: skill-card.md.","fileCount":4,"zipByteSize":4289},{"version":"1.0.7","createdAt":"2026-07-23T12:35:15.821Z","changelog":"gemini-3-1-flash-tts v1.0.7 - Simplified and refocused documentation: streamlined SKILL.md to reinforce prerequisites, agent behavior, and required/optional input structure. - Added a Prerequisites section listing related skills to install, with clear install commands. - Introduced a \"When NOT to use\" section, listing alternate skills for singing, avatars, or instrumental music. - Removed detailed prompting guides, scene composition references, and most local documentation files (now refers to external or other skill docs). - Updated metadata to include \"package: pruna-skills\" and set version to 1.0.7.","fileCount":4,"zipByteSize":4166}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s173djpkkm1x4yfzg522h5vfp989mrn5:gemini-3-1-flash-tts","setupComplexity":"low","setupSteps":["Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-gemini-3-1-flash-tts/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-gemini-3-1-flash-tts/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-gemini-3-1-flash-tts/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-gemini-3-1-flash-tts/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-gemini-3-1-flash-tts/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-gemini-3-1-flash-tts/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T14:50:28.994Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-gemini-3-1-flash-tts/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-gemini-3-1-flash-tts/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-gemini-3-1-flash-tts/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-gemini-3-1-flash-tts/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-10T12:32:12.994Z","emptyReason":null},"readme":"Skill: gemini-3.1-flash-tts\n\nOwner: pruna-ai\n\nSummary: Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video.\n\nTags: ai:1.0.14, generative:1.0.14, latest:1.0.14, pruna:1.0.14\n\nVersion history:\n\nv1.0.14 | 2026-09-29T15:35:50.539Z | auto\n\n- Bumped skill version to 1.0.14.\n- Updated version metadata in SKILL.md.\n- Removed the redundant skill-card.md file.\n\nv1.0.13 | 2026-09-17T13:58:36.340Z | auto\n\n- Bumped version to 1.0.13.\n- Updated the \"Typical next steps\" section for `p-video` to clarify use cases, adding more detail on limitations and suitable scenarios.\n- Removed the redundant skill-card.md file.\n\nv1.0.12 | 2026-09-10T13:57:08.752Z | auto\n\n- Version bump to 1.0.12.\n- Updated `SKILL.md` to clarify the description of the `p-video` skill in the Typical next steps section.\n- Removed `skill-card.md` file.\n\nv1.0.11 | 2026-09-03T14:11:27.901Z | auto\n\n- Bumped version to 1.0.11.\n- Removed redundant file: skill-card.md.\n- No user-facing feature changes.\n\nv1.0.10 | 2026-08-28T07:57:19.145Z | auto\n\n- Updated to version 1.0.10 in SKILL.md metadata.\n- Removed the skill-card.md file.\n- No functional or usage changes to guidance or API.\n\nv1.0.9 | 2026-08-04T06:18:12.930Z | auto\n\n- Version bump to 1.0.9.\n- Removed redundant file: skill-card.md.\n- No feature or functionality changes; documentation only.\n\nv1.0.8 | 2026-07-28T17:20:17.908Z | auto\n\n- Improved agent intake: now opens a clarification step (locale, voice, script) with generation-diversity before first TTS request.\n- Updated onboarding instructions for first reply behavior and intake flow.\n- Minor clarifications in agent prompts and user guidance.\n- Removed redundant file: skill-card.md.\n\nv1.0.7 | 2026-07-23T12:35:15.821Z | auto\n\ngemini-3-1-flash-tts v1.0.7\n\n- Simplified and refocused documentation: streamlined SKILL.md to reinforce prerequisites, agent behavior, and required/optional input structure.\n- Added a Prerequisites section listing related skills to install, with clear install commands.\n- Introduced a \"When NOT to use\" section, listing alternate skills for singing, avatars, or instrumental music.\n- Removed detailed prompting guides, scene composition references, and most local documentation files (now refers to external or other skill docs).\n- Updated metadata to include \"package: pruna-skills\" and set version to 1.0.7.\n\nv1.0.6 | 2026-07-16T20:56:49.368Z | auto\n\nVersion 1.0.6\n\n- Introduced a shared generation policy requiring random seed ritual, diversity, and quality checklist steps before running predictions.\n- Added new reference: **generation-quality-checklists.md**.\n- Improved documentation with clearer guidelines on input handling and generation best practices.\n- Minor copyedits and clarifications for usage patterns and scene anchor workflow.\n- Removed legacy skill-card.md file.\n\nv1.0.2 | 2026-07-16T13:22:51.818Z | auto\n\n- Version bump to 1.0.2\n- Updated documentation and reference guides (SKILL.md and reference files)\n- Removed outdated or redundant file (skill-card.md)\n- No functional changes; maintenance and clarity improvements for usage and integration\n\nv1.0.1 | 2026-07-14T15:44:33.684Z | auto\n\n- Added detailed documentation for Gemini 3.1 Flash TTS (Replicate) covering usage, prompting, workflow integration, and voice/style options.\n- Clarified model inputs, supported voices, and API usage with examples.\n- Provided step-by-step guidance for multi-scene narration and integration with Pruna video workflows.\n- Documented duration and style limitations, and best practices for scene-based narration.\n- Included references and examples for audio post-production and related tools.\n\nv0.0.1 | 2026-06-30T13:54:02.176Z | auto\n\nInitial release of gemini-3.1-flash-tts skill:\n\n- Provides natural text-to-speech (TTS) with style control via prompts and inline tags.\n- Supports multiple preset voices and rich delivery modifiers for narration, explainers, and scene voice lines.\n- Integrates with Pruna workflows for voiceover in video generation.\n- Offers guidance on narration patterns, style prompting, and scene structuring.\n- Includes usage instructions for HTTP API and environment setup.\n\nArchive index:\n\nArchive v1.0.14: 4 files, 4057 bytes\n\nFiles: skill-card.md (1797b), skill.manifest.json (23b), SKILL.md (6110b), _meta.json (140b)\n\nFile v1.0.14:SKILL.md\n\n---\nname: gemini-3.1-flash-tts\ndescription: Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video.\nlicense: MIT\nmetadata:\n  version: \"1.0.14\"\n  package: pruna-skills\n  provider: replicate\n  replicate_model: google/gemini-3.1-flash-tts\n---\n\n## Prerequisites\n\nInstall and load these skills before generating (skip if already in context via `@pruna`):\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFollow each skill's **Before generating** / craft sections — do not restate guide content here.\n\n## Agent habit\n\nIn the **first reply**, name `` `gemini-3.1-flash-tts` `` in backticks, confirm `REPLICATE_API_TOKEN` (or stop with signup links from `pruna-api`), then ask for required inputs. Open intake → **`generation-diversity`** clarification intake (locale, voice, script) before the first `POST`. Redirect when **When NOT to use** fits better.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `p-video-avatar` | Use when someone wants a person on camera speaking a script — lip-synced host, spokesperson, or narrated avatar from a portrait photo. | `npx skills add PrunaAI/pruna-skills@p-video-avatar -y` |\n| `music-2.5` | Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. | `npx skills add PrunaAI/pruna-skills@music-2.5 -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n\n## Environment\n\n```bash\nexport REPLICATE_API_TOKEN=r8_...\n```\n\nRequires **`ffmpeg`** / **`ffprobe`** when trimming, concatenating scene VO, or mixing with a bed.\n\n## HTTP (curl)\n\n```bash\ncurl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"text\": \"[warmly] The plush went flying. [short pause] And then it was gone.\",\n      \"voice\": \"Sulafat\",\n      \"prompt\": \"Warm storybook narrator, gentle pace, empathetic, no announcer voice.\",\n      \"language_code\": \"en-US\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/google/gemini-3.1-flash-tts/predictions\"\n```\n\nPoll `urls.get` until `status` is `succeeded`; download `output` (audio URL). Shared client: follow `pruna-api` (Replicate HTTP in the tool skill).\n\n## Before generating\n\n1. Complete Prerequisites guide reading order.\n2. Confirm **`text`**, **`voice`**, **`prompt`**, and **`language_code`** with the user. **`text`**, **`prompt`**, and inline `[tags]` must **align** — same emotional direction (see `audio-prompting` tts-style-prompting). When listing fields, name **`REPLICATE_API_TOKEN`** (Replicate — not `PRUNA_API_KEY`).\n3. **Model notes:** combined `text` + `prompt` ≤ ~8,000 bytes; output capped ~655s. When TTS feeds **`p-video`** as `input.audio`, keep each line **≤ ~19s** (`ffprobe`) — P-API clips audio at **20s**. Common voices: `Kore`, `Aoede`, `Sulafat`, `Achird`, `Charon`, `Puck`, `Vindemiatrix` — full list on the [Replicate readme](https://replicate.com/google/gemini-3.1-flash-tts/readme).\n\n## Required input\n\n- `text` (string) — spoken copy; supports inline `[tags]`. Max ~4,000 bytes.\n\n## Common optional fields\n\n- `voice` (default `Kore`)\n- `prompt` — style / director notes (max ~4,000 bytes)\n- `language_code` — BCP-47 (default `en-US`)\n\nInline tags (examples): `[sigh]` `[laughing]` `[whispering]` `[short pause]` `[medium pause]` `[long pause]` `[excitedly]`.\n\n## Typical next steps\n\nCommon follow-ons after this skill:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `p-video` | Use when someone wants a simple short clip from text or images — quick B-roll, drafts, or start/end frame animation. Not when the brief needs cinematic generation, highest quality, tight lip-sync, or imported audio at 1080p. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n| `p-video-avatar` | Use when someone wants a person on camera speaking a script — lip-synced host, spokesperson, or narrated avatar from a portrait photo. | `npx skills add PrunaAI/pruna-skills@p-video-avatar -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n| `narrated-multi-scene` | Use when someone wants a multi-part story with voiceover — episodic B-roll, chaptered promo, or several linked video scenes without on-camera dialogue. | `npx skills add PrunaAI/pruna-skills@narrated-multi-scene -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\nFile v1.0.14:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"gemini-3-1-flash-tts\",\n  \"version\": \"1.0.14\",\n  \"publishedAt\": 1790696150539\n}\n\nFile v1.0.14:skill-card.md\n\n## Description:\n\nGuides agents in generating spoken narration and voiceovers from scripts using Gemini 3.1 Flash TTS through Replicate.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[pruna-ai](https://clawhub.ai/user/pruna-ai)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nCreators and developers use this skill to prepare scripts and generate spoken narration or voiceovers for explainers, documentaries, and videos.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Narration text and style prompts are sent to Replicate with an API token.\n\nMitigation: Only submit content you are comfortable sharing with Replicate and keep the token private.\n\nRisk: Suggested skill installation commands may fetch additional third-party code.\n\nMitigation: Review or pin the suggested npx skill installs before running them, especially the full-suite install.\n\n## Reference(s):\n\n- [ClawHub skill release](https://clawhub.ai/pruna-ai/skills/gemini-3-1-flash-tts)\n- [Gemini 3.1 Flash TTS model documentation on Replicate](https://replicate.com/google/gemini-3.1-flash-tts/readme)\n\n## Skill Output:\n\n**Output Type(s):** [Guidance, Shell commands, Audio]\n\n**Output Format:** [Text guidance and commands; generated audio URL or downloaded audio file]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Requires a Replicate API token; narration text and style prompt are sent to Replicate.]\n\n## Skill Version(s):\n\n1.0.14 (source: server release and skill frontmatter)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.14:skill.manifest.json\n\n{\n  \"references\": []\n}\n\nArchive v1.0.13: 4 files, 4226 bytes\n\nFiles: skill-card.md (2154b), skill.manifest.json (23b), SKILL.md (6110b), _meta.json (140b)\n\nFile v1.0.13:SKILL.md\n\n---\nname: gemini-3.1-flash-tts\ndescription: Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video.\nlicense: MIT\nmetadata:\n  version: \"1.0.13\"\n  package: pruna-skills\n  provider: replicate\n  replicate_model: google/gemini-3.1-flash-tts\n---\n\n## Prerequisites\n\nInstall and load these skills before generating (skip if already in context via `@pruna`):\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFollow each skill's **Before generating** / craft sections — do not restate guide content here.\n\n## Agent habit\n\nIn the **first reply**, name `` `gemini-3.1-flash-tts` `` in backticks, confirm `REPLICATE_API_TOKEN` (or stop with signup links from `pruna-api`), then ask for required inputs. Open intake → **`generation-diversity`** clarification intake (locale, voice, script) before the first `POST`. Redirect when **When NOT to use** fits better.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `p-video-avatar` | Use when someone wants a person on camera speaking a script — lip-synced host, spokesperson, or narrated avatar from a portrait photo. | `npx skills add PrunaAI/pruna-skills@p-video-avatar -y` |\n| `music-2.5` | Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. | `npx skills add PrunaAI/pruna-skills@music-2.5 -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n\n## Environment\n\n```bash\nexport REPLICATE_API_TOKEN=r8_...\n```\n\nRequires **`ffmpeg`** / **`ffprobe`** when trimming, concatenating scene VO, or mixing with a bed.\n\n## HTTP (curl)\n\n```bash\ncurl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"text\": \"[warmly] The plush went flying. [short pause] And then it was gone.\",\n      \"voice\": \"Sulafat\",\n      \"prompt\": \"Warm storybook narrator, gentle pace, empathetic, no announcer voice.\",\n      \"language_code\": \"en-US\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/google/gemini-3.1-flash-tts/predictions\"\n```\n\nPoll `urls.get` until `status` is `succeeded`; download `output` (audio URL). Shared client: follow `pruna-api` (Replicate HTTP in the tool skill).\n\n## Before generating\n\n1. Complete Prerequisites guide reading order.\n2. Confirm **`text`**, **`voice`**, **`prompt`**, and **`language_code`** with the user. **`text`**, **`prompt`**, and inline `[tags]` must **align** — same emotional direction (see `audio-prompting` tts-style-prompting). When listing fields, name **`REPLICATE_API_TOKEN`** (Replicate — not `PRUNA_API_KEY`).\n3. **Model notes:** combined `text` + `prompt` ≤ ~8,000 bytes; output capped ~655s. When TTS feeds **`p-video`** as `input.audio`, keep each line **≤ ~19s** (`ffprobe`) — P-API clips audio at **20s**. Common voices: `Kore`, `Aoede`, `Sulafat`, `Achird`, `Charon`, `Puck`, `Vindemiatrix` — full list on the [Replicate readme](https://replicate.com/google/gemini-3.1-flash-tts/readme).\n\n## Required input\n\n- `text` (string) — spoken copy; supports inline `[tags]`. Max ~4,000 bytes.\n\n## Common optional fields\n\n- `voice` (default `Kore`)\n- `prompt` — style / director notes (max ~4,000 bytes)\n- `language_code` — BCP-47 (default `en-US`)\n\nInline tags (examples): `[sigh]` `[laughing]` `[whispering]` `[short pause]` `[medium pause]` `[long pause]` `[excitedly]`.\n\n## Typical next steps\n\nCommon follow-ons after this skill:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `p-video` | Use when someone wants a simple short clip from text or images — quick B-roll, drafts, or start/end frame animation. Not when the brief needs cinematic generation, highest quality, tight lip-sync, or imported audio at 1080p. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n| `p-video-avatar` | Use when someone wants a person on camera speaking a script — lip-synced host, spokesperson, or narrated avatar from a portrait photo. | `npx skills add PrunaAI/pruna-skills@p-video-avatar -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n| `narrated-multi-scene` | Use when someone wants a multi-part story with voiceover — episodic B-roll, chaptered promo, or several linked video scenes without on-camera dialogue. | `npx skills add PrunaAI/pruna-skills@narrated-multi-scene -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\nFile v1.0.13:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"gemini-3-1-flash-tts\",\n  \"version\": \"1.0.13\",\n  \"publishedAt\": 1789653516340\n}\n\nFile v1.0.13:skill-card.md\n\n## Description:\n\nUse when someone needs spoken narration or voiceover - explainer tracks, documentary lines, or voice to pair with generated video.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[pruna-ai](https://clawhub.ai/user/pruna-ai)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and creators use this skill to prepare text-to-speech narration or voiceover requests for the Replicate-hosted google/gemini-3.1-flash-tts model, including voice, language, and style-prompt choices.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill recommends installing unpinned external skills that may change after review.\n\nMitigation: Review the PrunaAI skill source and prefer pinned or otherwise verified skill versions before running the listed npx skills add commands, especially the full-suite install.\n\nRisk: Text submitted for narration is sent to Replicate using the user's API token.\n\nMitigation: Avoid sensitive or private script text unless the user is comfortable sending it to Replicate under their account and token.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/pruna-ai/skills/gemini-3-1-flash-tts)\n- [Replicate model readme](https://replicate.com/google/gemini-3.1-flash-tts/readme)\n- [Replicate predictions API endpoint](https://api.replicate.com/v1/models/google/gemini-3.1-flash-tts/predictions)\n\n## Skill Output:\n\n**Output Type(s):** [guidance, text, markdown, shell commands, configuration]\n\n**Output Format:** [Markdown with inline shell commands and API request examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Guides the agent to confirm text, voice, style prompt, language code, Replicate API token availability, and optional ffmpeg-based audio checks.]\n\n## Skill Version(s):\n\n1.0.13 (source: release evidence and skill frontmatter metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.13:skill.manifest.json\n\n{\n  \"references\": []\n}\n\nArchive v1.0.12: 4 files, 4179 bytes\n\nFiles: skill-card.md (2046b), skill.manifest.json (23b), SKILL.md (6066b), _meta.json (140b)\n\nFile v1.0.12:SKILL.md\n\n---\nname: gemini-3.1-flash-tts\ndescription: Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video.\nlicense: MIT\nmetadata:\n  version: \"1.0.12\"\n  package: pruna-skills\n  provider: replicate\n  replicate_model: google/gemini-3.1-flash-tts\n---\n\n## Prerequisites\n\nInstall and load these skills before generating (skip if already in context via `@pruna`):\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFollow each skill's **Before generating** / craft sections — do not restate guide content here.\n\n## Agent habit\n\nIn the **first reply**, name `` `gemini-3.1-flash-tts` `` in backticks, confirm `REPLICATE_API_TOKEN` (or stop with signup links from `pruna-api`), then ask for required inputs. Open intake → **`generation-diversity`** clarification intake (locale, voice, script) before the first `POST`. Redirect when **When NOT to use** fits better.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `p-video-avatar` | Use when someone wants a person on camera speaking a script — lip-synced host, spokesperson, or narrated avatar from a portrait photo. | `npx skills add PrunaAI/pruna-skills@p-video-avatar -y` |\n| `music-2.5` | Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. | `npx skills add PrunaAI/pruna-skills@music-2.5 -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n\n## Environment\n\n```bash\nexport REPLICATE_API_TOKEN=r8_...\n```\n\nRequires **`ffmpeg`** / **`ffprobe`** when trimming, concatenating scene VO, or mixing with a bed.\n\n## HTTP (curl)\n\n```bash\ncurl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"text\": \"[warmly] The plush went flying. [short pause] And then it was gone.\",\n      \"voice\": \"Sulafat\",\n      \"prompt\": \"Warm storybook narrator, gentle pace, empathetic, no announcer voice.\",\n      \"language_code\": \"en-US\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/google/gemini-3.1-flash-tts/predictions\"\n```\n\nPoll `urls.get` until `status` is `succeeded`; download `output` (audio URL). Shared client: follow `pruna-api` (Replicate HTTP in the tool skill).\n\n## Before generating\n\n1. Complete Prerequisites guide reading order.\n2. Confirm **`text`**, **`voice`**, **`prompt`**, and **`language_code`** with the user. **`text`**, **`prompt`**, and inline `[tags]` must **align** — same emotional direction (see `audio-prompting` tts-style-prompting). When listing fields, name **`REPLICATE_API_TOKEN`** (Replicate — not `PRUNA_API_KEY`).\n3. **Model notes:** combined `text` + `prompt` ≤ ~8,000 bytes; output capped ~655s. When TTS feeds **`p-video`** as `input.audio`, keep each line **≤ ~19s** (`ffprobe`) — P-API clips audio at **20s**. Common voices: `Kore`, `Aoede`, `Sulafat`, `Achird`, `Charon`, `Puck`, `Vindemiatrix` — full list on the [Replicate readme](https://replicate.com/google/gemini-3.1-flash-tts/readme).\n\n## Required input\n\n- `text` (string) — spoken copy; supports inline `[tags]`. Max ~4,000 bytes.\n\n## Common optional fields\n\n- `voice` (default `Kore`)\n- `prompt` — style / director notes (max ~4,000 bytes)\n- `language_code` — BCP-47 (default `en-US`)\n\nInline tags (examples): `[sigh]` `[laughing]` `[whispering]` `[short pause]` `[medium pause]` `[long pause]` `[excitedly]`.\n\n## Typical next steps\n\nCommon follow-ons after this skill:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `p-video` | Use when someone wants a simple short clip from text or images — quick B-roll, drafts, or start/end frame animation. Not when the brief needs the highest quality or tight lip-sync. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n| `p-video-avatar` | Use when someone wants a person on camera speaking a script — lip-synced host, spokesperson, or narrated avatar from a portrait photo. | `npx skills add PrunaAI/pruna-skills@p-video-avatar -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n| `narrated-multi-scene` | Use when someone wants a multi-part story with voiceover — episodic B-roll, chaptered promo, or several linked video scenes without on-camera dialogue. | `npx skills add PrunaAI/pruna-skills@narrated-multi-scene -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\nFile v1.0.12:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"gemini-3-1-flash-tts\",\n  \"version\": \"1.0.12\",\n  \"publishedAt\": 1789048628752\n}\n\nFile v1.0.12:skill-card.md\n\n## Description:\n\nUse when someone needs spoken narration or voiceover for explainer tracks, documentary lines, or voice paired with generated video.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[pruna-ai](https://clawhub.ai/user/pruna-ai)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nExternal creators, developers, and media teams use this skill to prepare Replicate TTS requests for narration or voiceover, including spoken text, voice, language, and style prompt.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Unpinned external skill installer commands with auto-confirmation can change what code or instructions are loaded before generation.\n\nMitigation: Review and pin installer targets and referenced skill revisions before running the documented npx commands, especially in environments with sensitive files or API tokens.\n\nRisk: Text and style prompts submitted for audio generation are sent to Replicate.\n\nMitigation: Avoid submitting confidential, regulated, or user-sensitive text unless the user has confirmed that Replicate processing is acceptable.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/pruna-ai/skills/gemini-3-1-flash-tts)\n- [Replicate model readme](https://replicate.com/google/gemini-3.1-flash-tts/readme)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, Shell commands, Configuration, Guidance]\n\n**Output Format:** [Markdown with inline shell commands and API request examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Requires a Replicate API token for generation; ffmpeg and ffprobe are needed for trimming, concatenation, or mixing workflows.]\n\n## Skill Version(s):\n\n1.0.12 (source: server release metadata and artifact metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.12:skill.manifest.json\n\n{\n  \"references\": []\n}\n\nArchive v1.0.11: 4 files, 4252 bytes\n\nFiles: skill-card.md (2339b), skill.manifest.json (23b), SKILL.md (6062b), _meta.json (140b)\n\nFile v1.0.11:SKILL.md\n\n---\nname: gemini-3.1-flash-tts\ndescription: Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video.\nlicense: MIT\nmetadata:\n  version: \"1.0.11\"\n  package: pruna-skills\n  provider: replicate\n  replicate_model: google/gemini-3.1-flash-tts\n---\n\n## Prerequisites\n\nInstall and load these skills before generating (skip if already in context via `@pruna`):\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFollow each skill's **Before generating** / craft sections — do not restate guide content here.\n\n## Agent habit\n\nIn the **first reply**, name `` `gemini-3.1-flash-tts` `` in backticks, confirm `REPLICATE_API_TOKEN` (or stop with signup links from `pruna-api`), then ask for required inputs. Open intake → **`generation-diversity`** clarification intake (locale, voice, script) before the first `POST`. Redirect when **When NOT to use** fits better.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `p-video-avatar` | Use when someone wants a person on camera speaking a script — lip-synced host, spokesperson, or narrated avatar from a portrait photo. | `npx skills add PrunaAI/pruna-skills@p-video-avatar -y` |\n| `music-2.5` | Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. | `npx skills add PrunaAI/pruna-skills@music-2.5 -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n\n## Environment\n\n```bash\nexport REPLICATE_API_TOKEN=r8_...\n```\n\nRequires **`ffmpeg`** / **`ffprobe`** when trimming, concatenating scene VO, or mixing with a bed.\n\n## HTTP (curl)\n\n```bash\ncurl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"text\": \"[warmly] The plush went flying. [short pause] And then it was gone.\",\n      \"voice\": \"Sulafat\",\n      \"prompt\": \"Warm storybook narrator, gentle pace, empathetic, no announcer voice.\",\n      \"language_code\": \"en-US\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/google/gemini-3.1-flash-tts/predictions\"\n```\n\nPoll `urls.get` until `status` is `succeeded`; download `output` (audio URL). Shared client: follow `pruna-api` (Replicate HTTP in the tool skill).\n\n## Before generating\n\n1. Complete Prerequisites guide reading order.\n2. Confirm **`text`**, **`voice`**, **`prompt`**, and **`language_code`** with the user. **`text`**, **`prompt`**, and inline `[tags]` must **align** — same emotional direction (see `audio-prompting` tts-style-prompting). When listing fields, name **`REPLICATE_API_TOKEN`** (Replicate — not `PRUNA_API_KEY`).\n3. **Model notes:** combined `text` + `prompt` ≤ ~8,000 bytes; output capped ~655s. When TTS feeds **`p-video`** as `input.audio`, keep each line **≤ ~19s** (`ffprobe`) — P-API clips audio at **20s**. Common voices: `Kore`, `Aoede`, `Sulafat`, `Achird`, `Charon`, `Puck`, `Vindemiatrix` — full list on the [Replicate readme](https://replicate.com/google/gemini-3.1-flash-tts/readme).\n\n## Required input\n\n- `text` (string) — spoken copy; supports inline `[tags]`. Max ~4,000 bytes.\n\n## Common optional fields\n\n- `voice` (default `Kore`)\n- `prompt` — style / director notes (max ~4,000 bytes)\n- `language_code` — BCP-47 (default `en-US`)\n\nInline tags (examples): `[sigh]` `[laughing]` `[whispering]` `[short pause]` `[medium pause]` `[long pause]` `[excitedly]`.\n\n## Typical next steps\n\nCommon follow-ons after this skill:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `p-video` | Use when someone wants one short video clip from text or images — B-roll, start/end frame animation, or a quick motion shot. Not for full multi-scene films or lip-synced hosts. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n| `p-video-avatar` | Use when someone wants a person on camera speaking a script — lip-synced host, spokesperson, or narrated avatar from a portrait photo. | `npx skills add PrunaAI/pruna-skills@p-video-avatar -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n| `narrated-multi-scene` | Use when someone wants a multi-part story with voiceover — episodic B-roll, chaptered promo, or several linked video scenes without on-camera dialogue. | `npx skills add PrunaAI/pruna-skills@narrated-multi-scene -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\nFile v1.0.11:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"gemini-3-1-flash-tts\",\n  \"version\": \"1.0.11\",\n  \"publishedAt\": 1788444687901\n}\n\nFile v1.0.11:skill-card.md\n\n## Description:\n\nUse when someone needs spoken narration or voiceover - explainer tracks, documentary lines, or voice to pair with generated video.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[pruna-ai](https://clawhub.ai/user/pruna-ai)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nExternal users and developers use this skill to prepare and run Replicate text-to-speech requests for narration, voiceover, documentary lines, and generated-video audio tracks.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Scripts, style prompts, selected voices, language codes, and token-authenticated requests are sent to Replicate for audio generation.\n\nMitigation: Confirm the user is comfortable sending this content to Replicate and avoid submitting sensitive or confidential scripts unless approved.\n\nRisk: The skill recommends installing companion skills from a remote source.\n\nMitigation: Review and trust the PrunaAI companion skills before installing or executing their guidance.\n\nRisk: Generated narration can exceed downstream timing constraints for video-avatar workflows.\n\nMitigation: Use ffprobe to check line duration and keep audio segments within the documented downstream limits before passing them into video workflows.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/pruna-ai/skills/gemini-3-1-flash-tts)\n- [Replicate Gemini 3.1 Flash TTS readme](https://replicate.com/google/gemini-3.1-flash-tts/readme)\n- [Replicate Gemini 3.1 Flash TTS predictions endpoint](https://api.replicate.com/v1/models/google/gemini-3.1-flash-tts/predictions)\n\n## Skill Output:\n\n**Output Type(s):** [Guidance, Markdown, Shell commands, API calls, Configuration]\n\n**Output Format:** [Markdown with inline bash and JSON examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Guides token-authenticated Replicate TTS requests and may reference ffmpeg or ffprobe checks for downstream audio handling.]\n\n## Skill Version(s):\n\n1.0.11 (source: server release metadata and skill metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.11:skill.manifest.json\n\n{\n  \"references\": []\n}\n\nArchive v1.0.10: 4 files, 4153 bytes\n\nFiles: skill-card.md (2037b), skill.manifest.json (23b), SKILL.md (6062b), _meta.json (140b)\n\nFile v1.0.10:SKILL.md\n\n---\nname: gemini-3.1-flash-tts\ndescription: Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video.\nlicense: MIT\nmetadata:\n  version: \"1.0.10\"\n  package: pruna-skills\n  provider: replicate\n  replicate_model: google/gemini-3.1-flash-tts\n---\n\n## Prerequisites\n\nInstall and load these skills before generating (skip if already in context via `@pruna`):\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFollow each skill's **Before generating** / craft sections — do not restate guide content here.\n\n## Agent habit\n\nIn the **first reply**, name `` `gemini-3.1-flash-tts` `` in backticks, confirm `REPLICATE_API_TOKEN` (or stop with signup links from `pruna-api`), then ask for required inputs. Open intake → **`generation-diversity`** clarification intake (locale, voice, script) before the first `POST`. Redirect when **When NOT to use** fits better.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `p-video-avatar` | Use when someone wants a person on camera speaking a script — lip-synced host, spokesperson, or narrated avatar from a portrait photo. | `npx skills add PrunaAI/pruna-skills@p-video-avatar -y` |\n| `music-2.5` | Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. | `npx skills add PrunaAI/pruna-skills@music-2.5 -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n\n## Environment\n\n```bash\nexport REPLICATE_API_TOKEN=r8_...\n```\n\nRequires **`ffmpeg`** / **`ffprobe`** when trimming, concatenating scene VO, or mixing with a bed.\n\n## HTTP (curl)\n\n```bash\ncurl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"text\": \"[warmly] The plush went flying. [short pause] And then it was gone.\",\n      \"voice\": \"Sulafat\",\n      \"prompt\": \"Warm storybook narrator, gentle pace, empathetic, no announcer voice.\",\n      \"language_code\": \"en-US\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/google/gemini-3.1-flash-tts/predictions\"\n```\n\nPoll `urls.get` until `status` is `succeeded`; download `output` (audio URL). Shared client: follow `pruna-api` (Replicate HTTP in the tool skill).\n\n## Before generating\n\n1. Complete Prerequisites guide reading order.\n2. Confirm **`text`**, **`voice`**, **`prompt`**, and **`language_code`** with the user. **`text`**, **`prompt`**, and inline `[tags]` must **align** — same emotional direction (see `audio-prompting` tts-style-prompting). When listing fields, name **`REPLICATE_API_TOKEN`** (Replicate — not `PRUNA_API_KEY`).\n3. **Model notes:** combined `text` + `prompt` ≤ ~8,000 bytes; output capped ~655s. When TTS feeds **`p-video`** as `input.audio`, keep each line **≤ ~19s** (`ffprobe`) — P-API clips audio at **20s**. Common voices: `Kore`, `Aoede`, `Sulafat`, `Achird`, `Charon`, `Puck`, `Vindemiatrix` — full list on the [Replicate readme](https://replicate.com/google/gemini-3.1-flash-tts/readme).\n\n## Required input\n\n- `text` (string) — spoken copy; supports inline `[tags]`. Max ~4,000 bytes.\n\n## Common optional fields\n\n- `voice` (default `Kore`)\n- `prompt` — style / director notes (max ~4,000 bytes)\n- `language_code` — BCP-47 (default `en-US`)\n\nInline tags (examples): `[sigh]` `[laughing]` `[whispering]` `[short pause]` `[medium pause]` `[long pause]` `[excitedly]`.\n\n## Typical next steps\n\nCommon follow-ons after this skill:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `p-video` | Use when someone wants one short video clip from text or images — B-roll, start/end frame animation, or a quick motion shot. Not for full multi-scene films or lip-synced hosts. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n| `p-video-avatar` | Use when someone wants a person on camera speaking a script — lip-synced host, spokesperson, or narrated avatar from a portrait photo. | `npx skills add PrunaAI/pruna-skills@p-video-avatar -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n| `narrated-multi-scene` | Use when someone wants a multi-part story with voiceover — episodic B-roll, chaptered promo, or several linked video scenes without on-camera dialogue. | `npx skills add PrunaAI/pruna-skills@narrated-multi-scene -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\nFile v1.0.10:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"gemini-3-1-flash-tts\",\n  \"version\": \"1.0.10\",\n  \"publishedAt\": 1787903839145\n}\n\nFile v1.0.10:skill-card.md\n\n## Description:\n\nUse when someone needs spoken narration or voiceover - explainer tracks, documentary lines, or voice to pair with generated video.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[pruna-ai](https://clawhub.ai/user/pruna-ai)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nExternal users, developers, and content teams use this skill to guide an agent through Replicate-based text-to-speech generation for narration, voiceover, and audio tracks paired with video.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill sends user-provided text and style prompts to Replicate using the user's API token.\n\nMitigation: Confirm the user intends to submit the content to Replicate and avoid sending sensitive text or prompts unless the user has approved that use.\n\nRisk: The skill suggests installing related prerequisite skills with npx before generation.\n\nMitigation: Review the referenced Pruna prerequisite skills and install commands before allowing those commands to run.\n\n## Reference(s):\n\n- [Replicate Gemini 3.1 Flash TTS readme](https://replicate.com/google/gemini-3.1-flash-tts/readme)\n- [Replicate Gemini 3.1 Flash TTS prediction endpoint](https://api.replicate.com/v1/models/google/gemini-3.1-flash-tts/predictions)\n\n## Skill Output:\n\n**Output Type(s):** [Guidance, Shell commands, API Calls, Configuration]\n\n**Output Format:** [Markdown with inline bash and curl examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Requires user-provided text; optional voice, prompt, and language_code; uses REPLICATE_API_TOKEN and may require ffmpeg or ffprobe for media post-processing.]\n\n## Skill Version(s):\n\n1.0.10 (source: release evidence and SKILL.md metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.10:skill.manifest.json\n\n{\n  \"references\": []\n}\n\nArchive v1.0.9: 4 files, 4153 bytes\n\nFiles: skill-card.md (2204b), skill.manifest.json (23b), SKILL.md (6061b), _meta.json (139b)\n\nFile v1.0.9:SKILL.md\n\n---\nname: gemini-3.1-flash-tts\ndescription: Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video.\nlicense: MIT\nmetadata:\n  version: \"1.0.9\"\n  package: pruna-skills\n  provider: replicate\n  replicate_model: google/gemini-3.1-flash-tts\n---\n\n## Prerequisites\n\nInstall and load these skills before generating (skip if already in context via `@pruna`):\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFollow each skill's **Before generating** / craft sections — do not restate guide content here.\n\n## Agent habit\n\nIn the **first reply**, name `` `gemini-3.1-flash-tts` `` in backticks, confirm `REPLICATE_API_TOKEN` (or stop with signup links from `pruna-api`), then ask for required inputs. Open intake → **`generation-diversity`** clarification intake (locale, voice, script) before the first `POST`. Redirect when **When NOT to use** fits better.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `p-video-avatar` | Use when someone wants a person on camera speaking a script — lip-synced host, spokesperson, or narrated avatar from a portrait photo. | `npx skills add PrunaAI/pruna-skills@p-video-avatar -y` |\n| `music-2.5` | Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. | `npx skills add PrunaAI/pruna-skills@music-2.5 -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n\n## Environment\n\n```bash\nexport REPLICATE_API_TOKEN=r8_...\n```\n\nRequires **`ffmpeg`** / **`ffprobe`** when trimming, concatenating scene VO, or mixing with a bed.\n\n## HTTP (curl)\n\n```bash\ncurl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"text\": \"[warmly] The plush went flying. [short pause] And then it was gone.\",\n      \"voice\": \"Sulafat\",\n      \"prompt\": \"Warm storybook narrator, gentle pace, empathetic, no announcer voice.\",\n      \"language_code\": \"en-US\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/google/gemini-3.1-flash-tts/predictions\"\n```\n\nPoll `urls.get` until `status` is `succeeded`; download `output` (audio URL). Shared client: follow `pruna-api` (Replicate HTTP in the tool skill).\n\n## Before generating\n\n1. Complete Prerequisites guide reading order.\n2. Confirm **`text`**, **`voice`**, **`prompt`**, and **`language_code`** with the user. **`text`**, **`prompt`**, and inline `[tags]` must **align** — same emotional direction (see `audio-prompting` tts-style-prompting). When listing fields, name **`REPLICATE_API_TOKEN`** (Replicate — not `PRUNA_API_KEY`).\n3. **Model notes:** combined `text` + `prompt` ≤ ~8,000 bytes; output capped ~655s. When TTS feeds **`p-video`** as `input.audio`, keep each line **≤ ~19s** (`ffprobe`) — P-API clips audio at **20s**. Common voices: `Kore`, `Aoede`, `Sulafat`, `Achird`, `Charon`, `Puck`, `Vindemiatrix` — full list on the [Replicate readme](https://replicate.com/google/gemini-3.1-flash-tts/readme).\n\n## Required input\n\n- `text` (string) — spoken copy; supports inline `[tags]`. Max ~4,000 bytes.\n\n## Common optional fields\n\n- `voice` (default `Kore`)\n- `prompt` — style / director notes (max ~4,000 bytes)\n- `language_code` — BCP-47 (default `en-US`)\n\nInline tags (examples): `[sigh]` `[laughing]` `[whispering]` `[short pause]` `[medium pause]` `[long pause]` `[excitedly]`.\n\n## Typical next steps\n\nCommon follow-ons after this skill:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `p-video` | Use when someone wants one short video clip from text or images — B-roll, start/end frame animation, or a quick motion shot. Not for full multi-scene films or lip-synced hosts. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n| `p-video-avatar` | Use when someone wants a person on camera speaking a script — lip-synced host, spokesperson, or narrated avatar from a portrait photo. | `npx skills add PrunaAI/pruna-skills@p-video-avatar -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n| `narrated-multi-scene` | Use when someone wants a multi-part story with voiceover — episodic B-roll, chaptered promo, or several linked video scenes without on-camera dialogue. | `npx skills add PrunaAI/pruna-skills@narrated-multi-scene -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\nFile v1.0.9:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"gemini-3-1-flash-tts\",\n  \"version\": \"1.0.9\",\n  \"publishedAt\": 1785824292930\n}\n\nFile v1.0.9:skill-card.md\n\n## Description: <br>\nUse when someone needs spoken narration or voiceover - explainer tracks, documentary lines, or voice to pair with generated video. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[pruna-ai](https://clawhub.ai/user/pruna-ai) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and creators use this skill to prepare Replicate Gemini Flash TTS requests for narration, explainer tracks, documentary lines, and voiceover audio paired with generated video. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill sends the user's script and style prompt to Replicate for text-to-speech generation. <br>\nMitigation: Confirm the script, voice, language, and style prompt with the user before making the Replicate request. <br>\nRisk: Generated audio may need length checks or editing before it is used in downstream video or narration workflows. <br>\nMitigation: Use ffmpeg or ffprobe when trimming, concatenating, mixing, or verifying clip length for downstream use. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/pruna-ai/skills/gemini-3-1-flash-tts) <br>\n- [Replicate Gemini 3.1 Flash TTS readme](https://replicate.com/google/gemini-3.1-flash-tts/readme) <br>\n- [Replicate Gemini 3.1 Flash TTS predictions endpoint](https://api.replicate.com/v1/models/google/gemini-3.1-flash-tts/predictions) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [Guidance, Shell commands, Configuration, API Calls] <br>\n**Output Format:** [Markdown with inline shell commands and JSON request examples] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Guides the agent through Replicate TTS request setup, polling, download, and optional audio post-processing.] <br>\n\n## Skill Version(s): <br>\n1.0.9 (source: server release metadata and skill metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.0.9:skill.manifest.json\n\n{\n  \"references\": []\n}\n\nArchive v1.0.8: 4 files, 4289 bytes\n\nFiles: skill-card.md (2489b), skill.manifest.json (23b), SKILL.md (6061b), _meta.json (139b)\n\nFile v1.0.8:SKILL.md\n\n---\nname: gemini-3.1-flash-tts\ndescription: Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video.\nlicense: MIT\nmetadata:\n  version: \"1.0.8\"\n  package: pruna-skills\n  provider: replicate\n  replicate_model: google/gemini-3.1-flash-tts\n---\n\n## Prerequisites\n\nInstall and load these skills before generating (skip if already in context via `@pruna`):\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFollow each skill's **Before generating** / craft sections — do not restate guide content here.\n\n## Agent habit\n\nIn the **first reply**, name `` `gemini-3.1-flash-tts` `` in backticks, confirm `REPLICATE_API_TOKEN` (or stop with signup links from `pruna-api`), then ask for required inputs. Open intake → **`generation-diversity`** clarification intake (locale, voice, script) before the first `POST`. Redirect when **When NOT to use** fits better.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `p-video-avatar` | Use when someone wants a person on camera speaking a script — lip-synced host, spokesperson, or narrated avatar from a portrait photo. | `npx skills add PrunaAI/pruna-skills@p-video-avatar -y` |\n| `music-2.5` | Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. | `npx skills add PrunaAI/pruna-skills@music-2.5 -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n\n## Environment\n\n```bash\nexport REPLICATE_API_TOKEN=r8_...\n```\n\nRequires **`ffmpeg`** / **`ffprobe`** when trimming, concatenating scene VO, or mixing with a bed.\n\n## HTTP (curl)\n\n```bash\ncurl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"text\": \"[warmly] The plush went flying. [short pause] And then it was gone.\",\n      \"voice\": \"Sulafat\",\n      \"prompt\": \"Warm storybook narrator, gentle pace, empathetic, no announcer voice.\",\n      \"language_code\": \"en-US\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/google/gemini-3.1-flash-tts/predictions\"\n```\n\nPoll `urls.get` until `status` is `succeeded`; download `output` (audio URL). Shared client: follow `pruna-api` (Replicate HTTP in the tool skill).\n\n## Before generating\n\n1. Complete Prerequisites guide reading order.\n2. Confirm **`text`**, **`voice`**, **`prompt`**, and **`language_code`** with the user. **`text`**, **`prompt`**, and inline `[tags]` must **align** — same emotional direction (see `audio-prompting` tts-style-prompting). When listing fields, name **`REPLICATE_API_TOKEN`** (Replicate — not `PRUNA_API_KEY`).\n3. **Model notes:** combined `text` + `prompt` ≤ ~8,000 bytes; output capped ~655s. When TTS feeds **`p-video`** as `input.audio`, keep each line **≤ ~19s** (`ffprobe`) — P-API clips audio at **20s**. Common voices: `Kore`, `Aoede`, `Sulafat`, `Achird`, `Charon`, `Puck`, `Vindemiatrix` — full list on the [Replicate readme](https://replicate.com/google/gemini-3.1-flash-tts/readme).\n\n## Required input\n\n- `text` (string) — spoken copy; supports inline `[tags]`. Max ~4,000 bytes.\n\n## Common optional fields\n\n- `voice` (default `Kore`)\n- `prompt` — style / director notes (max ~4,000 bytes)\n- `language_code` — BCP-47 (default `en-US`)\n\nInline tags (examples): `[sigh]` `[laughing]` `[whispering]` `[short pause]` `[medium pause]` `[long pause]` `[excitedly]`.\n\n## Typical next steps\n\nCommon follow-ons after this skill:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `p-video` | Use when someone wants one short video clip from text or images — B-roll, start/end frame animation, or a quick motion shot. Not for full multi-scene films or lip-synced hosts. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n| `p-video-avatar` | Use when someone wants a person on camera speaking a script — lip-synced host, spokesperson, or narrated avatar from a portrait photo. | `npx skills add PrunaAI/pruna-skills@p-video-avatar -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n| `narrated-multi-scene` | Use when someone wants a multi-part story with voiceover — episodic B-roll, chaptered promo, or several linked video scenes without on-camera dialogue. | `npx skills add PrunaAI/pruna-skills@narrated-multi-scene -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\nFile v1.0.8:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"gemini-3-1-flash-tts\",\n  \"version\": \"1.0.8\",\n  \"publishedAt\": 1785259217908\n}\n\nFile v1.0.8:skill-card.md\n\n## Description: <br>\nUse when someone needs spoken narration or voiceover for explainer tracks, documentary lines, or voice to pair with generated video. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[pruna-ai](https://clawhub.ai/user/pruna-ai) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and content creators use this skill to prepare Replicate Gemini Flash TTS requests for narrated explainers, documentary lines, voiceover tracks, and generated-video audio. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill sends narration text and related prompt details to Replicate for text-to-speech generation. <br>\nMitigation: Use it only when sharing that text with Replicate is acceptable, and provide a Replicate API token only in an environment intended for this workflow. <br>\nRisk: Additional prerequisite skills may be installed before generation, including the full Pruna suite option. <br>\nMitigation: Review the referenced PrunaAI prerequisite skills before installing them, especially when choosing the full suite. <br>\nRisk: Long audio or video-bound narration can exceed model or downstream clip limits. <br>\nMitigation: Keep combined text and prompt within the documented byte limit, use ffmpeg or ffprobe when trimming or validating duration, and keep p-video-bound lines near 19 seconds. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/pruna-ai/skills/gemini-3-1-flash-tts) <br>\n- [Replicate Gemini 3.1 Flash TTS readme](https://replicate.com/google/gemini-3.1-flash-tts/readme) <br>\n- [Replicate prediction API endpoint](https://api.replicate.com/v1/models/google/gemini-3.1-flash-tts/predictions) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [guidance, shell commands, configuration] <br>\n**Output Format:** [Markdown with inline bash and curl examples] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Guides collection of text, voice, prompt, language_code, and Replicate API token before generating audio.] <br>\n\n## Skill Version(s): <br>\n1.0.8 (source: server release metadata and SKILL.md metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.0.8:skill.manifest.json\n\n{\n  \"references\": []\n}\n\nArchive v1.0.7: 4 files, 4166 bytes\n\nFiles: skill-card.md (2248b), skill.manifest.json (23b), SKILL.md (5948b), _meta.json (139b)\n\nFile v1.0.7:SKILL.md\n\n---\nname: gemini-3.1-flash-tts\ndescription: Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video.\nlicense: MIT\nmetadata:\n  version: \"1.0.7\"\n  package: pruna-skills\n  provider: replicate\n  replicate_model: google/gemini-3.1-flash-tts\n---\n\n## Prerequisites\n\nInstall and load these skills before generating (skip if already in context via `@pruna`):\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFollow each skill's **Before generating** / craft sections — do not restate guide content here.\n\n## Agent habit\n\nIn the **first reply**, name `` `gemini-3.1-flash-tts` `` in backticks, confirm `REPLICATE_API_TOKEN` (or stop with signup links from `pruna-api`), then ask for required inputs. Redirect when **When NOT to use** fits better.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `p-video-avatar` | Use when someone wants a person on camera speaking a script — lip-synced host, spokesperson, or narrated avatar from a portrait photo. | `npx skills add PrunaAI/pruna-skills@p-video-avatar -y` |\n| `music-2.5` | Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. | `npx skills add PrunaAI/pruna-skills@music-2.5 -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n\n## Environment\n\n```bash\nexport REPLICATE_API_TOKEN=r8_...\n```\n\nRequires **`ffmpeg`** / **`ffprobe`** when trimming, concatenating scene VO, or mixing with a bed.\n\n## HTTP (curl)\n\n```bash\ncurl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"text\": \"[warmly] The plush went flying. [short pause] And then it was gone.\",\n      \"voice\": \"Sulafat\",\n      \"prompt\": \"Warm storybook narrator, gentle pace, empathetic, no announcer voice.\",\n      \"language_code\": \"en-US\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/google/gemini-3.1-flash-tts/predictions\"\n```\n\nPoll `urls.get` until `status` is `succeeded`; download `output` (audio URL). Shared client: follow `pruna-api` (Replicate HTTP in the tool skill).\n\n## Before generating\n\n1. Complete Prerequisites guide reading order.\n2. Confirm **`text`**, **`voice`**, **`prompt`**, and **`language_code`** with the user. **`text`**, **`prompt`**, and inline `[tags]` must **align** — same emotional direction (see `audio-prompting` tts-style-prompting). When listing fields, name **`REPLICATE_API_TOKEN`** (Replicate — not `PRUNA_API_KEY`).\n3. **Model notes:** combined `text` + `prompt` ≤ ~8,000 bytes; output capped ~655s. When TTS feeds **`p-video`** as `input.audio`, keep each line **≤ ~19s** (`ffprobe`) — P-API clips audio at **20s**. Common voices: `Kore`, `Aoede`, `Sulafat`, `Achird`, `Charon`, `Puck`, `Vindemiatrix` — full list on the [Replicate readme](https://replicate.com/google/gemini-3.1-flash-tts/readme).\n\n## Required input\n\n- `text` (string) — spoken copy; supports inline `[tags]`. Max ~4,000 bytes.\n\n## Common optional fields\n\n- `voice` (default `Kore`)\n- `prompt` — style / director notes (max ~4,000 bytes)\n- `language_code` — BCP-47 (default `en-US`)\n\nInline tags (examples): `[sigh]` `[laughing]` `[whispering]` `[short pause]` `[medium pause]` `[long pause]` `[excitedly]`.\n\n## Typical next steps\n\nCommon follow-ons after this skill:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `p-video` | Use when someone wants one short video clip from text or images — B-roll, start/end frame animation, or a quick motion shot. Not for full multi-scene films or lip-synced hosts. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n| `p-video-avatar` | Use when someone wants a person on camera speaking a script — lip-synced host, spokesperson, or narrated avatar from a portrait photo. | `npx skills add PrunaAI/pruna-skills@p-video-avatar -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n| `narrated-multi-scene` | Use when someone wants a multi-part story with voiceover — episodic B-roll, chaptered promo, or several linked video scenes without on-camera dialogue. | `npx skills add PrunaAI/pruna-skills@narrated-multi-scene -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\nFile v1.0.7:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"gemini-3-1-flash-tts\",\n  \"version\": \"1.0.7\",\n  \"publishedAt\": 1784810115821\n}\n\nFile v1.0.7:skill-card.md\n\n## Description: <br>\nHelps an agent prepare and run spoken narration or voiceover generation with Gemini 3.1 Flash TTS through Replicate. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[pruna-ai](https://clawhub.ai/user/pruna-ai) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and creators use this skill to collect text, voice, style prompt, and language inputs, then guide a Replicate text-to-speech request for narration, documentary lines, explainer voiceover, or audio to pair with generated video. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: A Replicate API token is required to generate audio. <br>\nMitigation: Keep REPLICATE_API_TOKEN in the local environment and do not include it in prompts, generated files, or shared logs. <br>\nRisk: Optional related PrunaAI skills may broaden the installed skill set. <br>\nMitigation: Install only the companion skills needed for the workflow and review them before adding the full suite. <br>\nRisk: Generated narration may not match the intended text, tone, voice, or language. <br>\nMitigation: Confirm text, voice, prompt, and language_code before generation, then review the downloaded audio before using it. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/pruna-ai/skills/gemini-3-1-flash-tts) <br>\n- [Replicate model readme](https://replicate.com/google/gemini-3.1-flash-tts/readme) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [guidance, shell commands, configuration, API calls] <br>\n**Output Format:** [Markdown guidance with bash and curl examples] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Requires REPLICATE_API_TOKEN; ffmpeg and ffprobe are needed for trimming, concatenating scene voiceover, or mixing with a music bed.] <br>\n\n## Skill Version(s): <br>\n1.0.7 (source: server release metadata and skill frontmatter) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.0.7:skill.manifest.json\n\n{\n  \"references\": []\n}\n\nArchive v1.0.6: 12 files, 31898 bytes\n\nFiles: README-INSTALL.md (982b), references/api-credentials.md (3128b), references/audio-post-production.md (9168b), references/generation-diversity.md (25978b), references/generation-quality-checklists.md (10290b), references/random-seed-ritual.md (4011b), references/replicate-api.md (2101b), references/scene-anchor-triple.md (10559b), skill-card.md (2899b), skill.manifest.json (190b), SKILL.md (9092b), _meta.json (139b)\n\nFile v1.0.6:SKILL.md\n\n---\nname: gemini-3.1-flash-tts\ndescription: Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video.\nlicense: MIT\nmetadata:\n  version: \"1.0.6\"\n  provider: replicate\n  replicate_model: google/gemini-3.1-flash-tts\n---\n\n## Shared generation policy\n\n<!-- shared-generation-policy -->\n\nBefore any paid `POST /v1/predictions`:\n\n1. **[Random seed ritual](./references/random-seed-ritual.md)** — always first; derive axes via sum-mod.\n2. **[Generation diversity](./references/generation-diversity.md)** — explicit prompts; rotate ≥2 scenario axes per session.\n3. **[Quality checklists](./references/generation-quality-checklists.md)** — open output files and judge pass/fail before advancing.\n\n# Gemini 3.1 Flash TTS (Replicate)\n\nNatural **text-to-speech** with style control via a director **`prompt`** and inline **`[tags]`** in the spoken text. Not a Pruna P-model — runs on [Replicate](https://replicate.com/google/gemini-3.1-flash-tts).\n\n**Typical downstream (preferred):** upload MP3/WAV to Pruna `/v1/files` → pass as **`input.audio`** with **`input.image`** and **`input.last_frame_image`** on [`p-video`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/p-video/skills/p-video/SKILL.md) ([scene anchor triple](./references/scene-anchor-triple.md)). Same upload pattern on [`p-video-avatar`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/p-video-avatar/skills/p-video-avatar/SKILL.md) for lip-sync narration.\n\nFor **narration + instrumental bed**, render video with embedded VO first, then mix bed in post — [audio-post-production.md](./references/audio-post-production.md).\n\n## When to use\n\n| Goal | Use this |\n|------|----------|\n| Documentary / story narrator over B-roll | Yes — per-scene or full-reel script |\n| Character dialogue in a talking head | No — use [`p-video-avatar`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/p-video-avatar/skills/p-video-avatar/SKILL.md) native voice |\n| Full sung song | No — use [music-2.5](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/music-2.5/skills/music-2.5/SKILL.md) |\n| Instrumental mood bed only | No — use [stable-audio-2.5](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/stable-audio-2.5/skills/stable-audio-2.5/SKILL.md) |\n| Lip-sync from uploaded VO | Upload TTS → [`p-video`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/p-video/skills/p-video/SKILL.md) with `audio` (duration follows audio) |\n\n## Environment\n\n```bash\nexport REPLICATE_API_TOKEN=r8_...\n```\n\nRequires **`ffmpeg`** / **`ffprobe`** when trimming, concatenating scene VO, or mixing with a bed.\n\n## Model input (Replicate)\n\n| Field | Notes |\n|-------|-------|\n| `text` | **Required.** Spoken copy; supports inline `[tags]`. Max ~4,000 bytes. |\n| `voice` | One of 30 preset voices (default `Kore`). See [voice table](#voices). |\n| `prompt` | Style / scene / director notes — tone, pace, accent, character. Max ~4,000 bytes. |\n| `language_code` | BCP-47 (default `en-US`). Set explicitly for non-English. |\n\nCombined `text` + `prompt` ≤ ~8,000 bytes. Output is capped at ~655 seconds.\n\n## Style prompting\n\nAlign **`prompt`**, **`text`**, and any **`[tags]`** — all should point the same emotional direction.\n\n**Example prompt:**\n\n```text\nAUDIO PROFILE: Warm documentary narrator, gentle and empathetic.\n\nTHE SCENE: A short nature film about a dog who loses a favorite toy.\n\nDIRECTOR'S NOTES:\n- Style: Soft, curious, slightly playful — like a children's storybook read aloud.\n- Pace: Unhurried; leave room for visuals to breathe.\n- Do not sound like a hard-sell announcer.\n```\n\n**Example text:**\n\n```text\n[warmly] Every afternoon, the meadow was theirs.\n[short pause] But today, something small went missing.\n[concerned] And for the first time, the world felt a little too big.\n```\n\n## Inline tags\n\n| Tag | Effect |\n|-----|--------|\n| `[sigh]` `[laughing]` `[uhm]` | Non-speech vocalizations |\n| `[whispering]` `[shouting]` `[sarcasm]` `[robotic]` `[extremely fast]` | Delivery modifiers for following text |\n| `[short pause]` `[medium pause]` `[long pause]` | Silence (~250ms / ~500ms / ~1000ms+) |\n| `[excitedly]` `[bored]` `[reluctantly]` etc. | Descriptive tags — test before production |\n\n## Voices (common picks)\n\n| Voice | Gender | Character | Good for |\n|-------|--------|-----------|----------|\n| `Kore` | Female | Firm | Default narrator |\n| `Aoede` | Female | Breezy | Light documentary |\n| `Sulafat` | Female | Warm | Storybook / emotional beats |\n| `Achird` | Male | Friendly | Casual explainer |\n| `Charon` | Male | Informative | Product / tech VO |\n| `Puck` | Male | Upbeat | Short social hooks |\n| `Vindemiatrix` | Female | Gentle | Soft narration under music |\n\nFull list: [Replicate readme](https://replicate.com/google/gemini-3.1-flash-tts/readme).\n\n## HTTP (curl)\n\n```bash\ncurl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"text\": \"[warmly] The plush went flying. [short pause] And then it was gone.\",\n      \"voice\": \"Sulafat\",\n      \"prompt\": \"Warm storybook narrator, gentle pace, empathetic, no announcer voice.\",\n      \"language_code\": \"en-US\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/google/gemini-3.1-flash-tts/predictions\"\n```\n\nPoll `urls.get` until `status` is `succeeded`; download `output` (audio URL).\n\nShared client: [`replicate_api.py`](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/_shared/scripts/replicate_api.py).\n\n## Multi-scene narration patterns\n\n| Pattern | When | Steps |\n|---------|------|-------|\n| **Per-scene VO → scene anchor triple** (**preferred**) | Story B-roll | TTS → upload → `p-video` with `image` + `last_frame_image` + `audio` — [scene-anchor-triple.md](./references/scene-anchor-triple.md) |\n| **Per-scene VO → `p-video-avatar`** | Talking-head narration | TTS or script → upload → `p-video-avatar` with portrait + `audio` |\n| **Per-scene VO → post mux** | Fallback: silent clips already rendered | TTS per scene → concat → ffmpeg mux (may truncate long lines) |\n| **One continuous narrator track** | Single voice-over bed for whole reel | TTS full script once → mux under concat with [`launch_background_music.py`](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/_shared/scripts/launch_background_music.py)-style `amix` (narration = primary stream) |\n| **Narration + instrumental bed** | Story film with music | TTS + [stable-audio-2.5](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/stable-audio-2.5/skills/stable-audio-2.5/SKILL.md) bed → mix narration loud, bed quiet (~0.08–0.15) — see [audio-post-production.md](./references/audio-post-production.md) |\n\nRecord **`voice`**, **`prompt`**, and **`language_code`** in the project manifest for consistency across scene regens.\n\n## Duration limit (scene anchor triple)\n\nWhen TTS feeds **`p-video`** as `input.audio`, clip length follows the MP3 but **cannot exceed 20 seconds** on P-API. After each scene file is downloaded:\n\n1. Run `ffprobe` (or [`probe_media_duration_seconds`](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/_shared/scripts/p_video_payload.py)) on the MP3.\n2. Keep each line **≤ ~19 seconds** — shorten copy or split into two scenes if over.\n3. Upload to Pruna and pass as `input.audio` with `image` + `last_frame_image`; omit `duration`.\n\nTruncated narration mid-sentence usually means the line exceeded the cap, not that audio was omitted from the prediction.\n\n**If `ffprobe` > ~19s:** shorten the `text` first; if still long, add *brisk pace, ~2.3 words per second, no filler* to `style_prompt` and regenerate; if two beats remain, split into two scenes in the plan (separate TTS files).\n\n## Plan JSON (`narration`)\n\n```json\n\"narration\": {\n  \"enabled\": true,\n  \"voice\": \"Sulafat\",\n  \"language_code\": \"en-US\",\n  \"style_prompt\": \"Warm storybook narrator, gentle pace, empathetic.\",\n  \"mode\": \"per_scene\",\n  \"scenes\": {\n    \"01_playtime\": {\n      \"text\": \"[warmly] This was their favorite game.\"\n    },\n    \"02_toss\": {\n      \"text\": \"[excitedly] Up it went — higher than ever.\"\n    }\n  }\n}\n```\n\n`mode`: `p_video_audio` (**preferred** — scene anchor triple) | `per_scene` | `full_reel`\n\n## Related\n\n- [scene-anchor-triple.md](./references/scene-anchor-triple.md) — **`image` + `last_frame_image` + `audio`** per scene\n- [audio-post-production.md](./references/audio-post-production.md) — narration + bed layering\n- [narrated-multi-scene](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/narrated-multi-scene/skills/narrated-multi-scene/SKILL.md) — scene table + assembly\n- [stable-audio-2.5](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/stable-audio-2.5/skills/stable-audio-2.5/SKILL.md) — instrumental beds\n- [music-2.5](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/music-2.5/skills/music-2.5/SKILL.md) — full songs with vocals\n- [replicate-api.md](https://github.com/PrunaAI/pruna-skills/tree/main/references/policies/replicate-api.md)\n\nFile v1.0.6:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"gemini-3-1-flash-tts\",\n  \"version\": \"1.0.6\",\n  \"publishedAt\": 1784235409368\n}\n\nFile v1.0.6:references/api-credentials.md\n\n# API credentials (Pruna + Replicate)\n\n**Agent rule:** Before any `POST /v1/predictions`, Replicate prediction, or paid runner — check env vars. If a required key is **missing or empty**, **stop** and tell the user how to sign up. Do not guess, mock, or skip with placeholder keys.\n\n## Pruna P-API\n\n| | |\n|--|--|\n| **Env var** | `PRUNA_API_KEY` |\n| **Header** | `apikey: ${PRUNA_API_KEY}` (not `Authorization: Bearer`) |\n| **Sign up / get key** | [Pruna dashboard](https://dashboard.pruna.ai/) |\n| **Docs** | [Quickstart](https://docs.api.pruna.ai/guides/quickstart) · [pruna-api.md](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/pruna-api/SKILL.md) |\n\n**Used by:** all `p-image*`, `p-video*` tool skills and Pruna workflow runners.\n\n### If `PRUNA_API_KEY` is missing — agent message template\n\n> Pruna generation needs an API key. Sign up or sign in at **[dashboard.pruna.ai](https://dashboard.pruna.ai/)**, create an API key, then set:\n>\n> ```bash\n> export PRUNA_API_KEY=\"your_key_here\"\n> ```\n>\n> Add that to your shell profile or project `.env` (never commit the key). Reply when it’s set and we can continue.\n\n## Replicate\n\n| | |\n|--|--|\n| **Env var** | `REPLICATE_API_TOKEN` |\n| **Header** | `Authorization: Bearer ${REPLICATE_API_TOKEN}` |\n| **Sign up / get token** | [Replicate API tokens](https://replicate.com/account/api-tokens) ([sign in](https://replicate.com/signin) first if needed) |\n| **Docs** | [replicate-api.md](./replicate-api.md) |\n\n**Used by:** `music-2.5`, `gemini-3.1-flash-tts`, `stable-audio-2.5`, `whisperx`, and workflow beds/TTS/song phases.\n\n### If `REPLICATE_API_TOKEN` is missing — agent message template\n\n> This step uses Replicate (song, TTS, transcription, or background bed). Create a token at **[replicate.com/account/api-tokens](https://replicate.com/account/api-tokens)**, then set:\n>\n> ```bash\n> export REPLICATE_API_TOKEN=\"r8_...\"\n> ```\n>\n> Reply when it’s set and we can continue.\n\n## Which key does this job need?\n\n| Task | Keys required |\n|------|----------------|\n| `p-image`, `p-image-edit`, `p-image-upscale`, `p-image-try-on` | `PRUNA_API_KEY` |\n| `p-video`, `p-video-avatar`, `p-video-animate`, `p-video-replace` | `PRUNA_API_KEY` |\n| Music 2.5 song generation | `REPLICATE_API_TOKEN` |\n| Gemini TTS narration | `REPLICATE_API_TOKEN` |\n| Stable Audio background bed | `REPLICATE_API_TOKEN` |\n| WhisperX transcription | `REPLICATE_API_TOKEN` |\n| Music video / explainer (full pipeline) | **Both** — Pruna for stills/video; Replicate for song/TTS/bed as needed |\n\nWhen only one key is missing, suggest **only** that provider’s signup link — not both.\n\n## Security\n\n- Never print full keys in chat or commit them to git.\n- `.env` is gitignored; prefer env vars over hardcoding in plans or manifests.\n- Never embed keys in prompts, manifests, plan JSON, logs, or **subagent task text**.\n- Prefer the **parent agent** to own API calls; do not fan credentials across parallel subagents unless the host documents isolated secret injection.\n- Full rules: [agent-safety.md](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/agent-safety/SKILL.md).\n\nFile v1.0.6:references/audio-post-production.md\n\n# Audio post-production (Pruna + Replicate)\n\nHow to choose and **layer** audio when building reels, multi-scene films, and launch videos.\n\n**Multi-scene narrated films:** use the [scene anchor triple](./scene-anchor-triple.md) — pass TTS to **`p-video`** as `input.audio` with `image` + `last_frame_image`; do not post-mux unless re-render is impossible.\n\n**Visual-only transitions (no VO):** use the [scene anchor pair](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/scene-anchor-pair/SKILL.md) — `duration` instead of `audio`; see [visual-transition-reel](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/visual-transition-reel/skills/visual-transition-reel/SKILL.md).\n\n## Audio-led `p-video` (required when VO/narration exists)\n\nWhen narration, TTS, or a timed audio slice is available **before** video render:\n\n1. Upload the audio file to Pruna (`POST /v1/files`).\n2. Pass `urls.get` as **`input.audio`** on **`p-video`** (or **`p-video-avatar`** for human lip-sync).\n3. **Omit `duration`** — clip length follows the audio (capped at **20s** on P-API); the model syncs motion to speech.\n4. Set **`save_audio`: true** so the full line is embedded in the output clip.\n5. **Probe TTS length** before render — per-scene lines should be **≤ ~19s** or the API truncates the tail even when `audio` is set.\n5. **Concat** clips in order (narration already on each clip). Optional bed mixed **under** VO in post.\n\n**Never** generate silent `p-video` and ffmpeg-mux narration afterward unless re-render is impossible — post-mux **truncates** lines longer than the video slot (common with Gemini TTS).\n\n**Over 20s?** Shorten scene copy → tighten TTS pace in `style_prompt` → split into two scene rows (each with its own triple). See [narrated-multi-scene](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/narrated-multi-scene/skills/narrated-multi-scene/SKILL.md) duration gate.\n\nHelper: [`p_video_payload.py`](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/_shared/scripts/p_video_payload.py) — `build_p_video_payload(...)` enforces omitting `duration` when `audio_url` is set.\n\n| Workflow | Audio source | `p-video` fields |\n|----------|--------------|------------------|\n| Dog plush / story film | Gemini TTS per scene | `image` + `last_frame_image` + `audio` |\n| Music video performance | Song slice per cut | `image` + `audio` |\n| Music video B-roll | Song slice (optional) | `image` + `audio` or `duration` only |\n| Viking narrator beats | Gemini TTS | `image` + `last_frame_image` + `audio` |\n\n## Tool picker\n\n| Need | Tool | Skill |\n|------|------|-------|\n| Cinematic clip with model-generated sound | `p-video` (`save_audio`, optional uploaded `audio`) | [p-video](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/p-video/skills/p-video/SKILL.md) |\n| Lip-sync / duration locked to VO | Upload audio → `p-video` with `audio` | [p-video](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/p-video/skills/p-video/SKILL.md) |\n| Documentary / story narrator | [Gemini 3.1 Flash TTS](https://replicate.com/google/gemini-3.1-flash-tts) | [gemini-3.1-flash-tts](../SKILL.md) |\n| Light instrumental under dialogue | [Stable Audio 2.5](https://replicate.com/stability-ai/stable-audio-2.5) | [stable-audio-2.5](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/stable-audio-2.5/skills/stable-audio-2.5/SKILL.md) |\n| Full song with sung vocals | [Music 2.5](https://replicate.com/minimax/music-2.5) | [music-2.5](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/music-2.5/skills/music-2.5/SKILL.md) |\n| Speaking on-camera character | `p-video-avatar` | [p-video-avatar](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/p-video-avatar/skills/p-video-avatar/SKILL.md) |\n\n**Env:** Pruna calls need `PRUNA_API_KEY`; Replicate audio tools need `REPLICATE_API_TOKEN`. Assembly steps need **`ffmpeg`** / **`ffprobe`**.\n\n## Layering matrix\n\n| Stack | Primary audio | Secondary | Mix notes |\n|-------|---------------|-----------|-----------|\n| **Silent B-roll** | — | — | Concat video only |\n| **Native `p-video` sound** | Model output | — | Keep `save_audio` default; normalize in assembly if scenes differ |\n| **Narration only (fallback)** | Gemini TTS | — | Post-mux only when audio-led `p-video` is not suitable — prefer **Pipeline B** below |\n| **Bed only** | Stable Audio bed | — | [`launch_background_music.py`](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/_shared/scripts/launch_background_music.py) |\n| **Narration + bed (preferred)** | Gemini TTS → **`p-video` `audio`** | Stable Audio (quiet) | TTS uploaded to Pruna drives clip length + sync; bed mixed in post under narration (~0.08–0.15) |\n| **Avatar VO + bed** | `p-video-avatar` dialogue | Stable Audio bed | Same bed pattern as replace/launch reels — bed **under** existing speech |\n| **Music video** | Music 2.5 full song | — | [music-video](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/music-video/skills/music-video/SKILL.md) |\n\n## Recommended pipelines\n\n### A — Narrated multi-scene B-roll (**preferred — scene anchor triple**)\n\n```text\nPhase 0 — intake: scene table with start/end still prompts + narration lines\nPhase 1 — hero + p-image-edit start stills + end stills (parallel)\nPhase 2 — Gemini TTS per scene (parallel) → upload each to /v1/files\nPhase 3 — p-video per scene: input.image + input.last_frame_image + input.audio (parallel; omit duration)\nPhase 4 — ffmpeg concat (VO embedded; frame chain via shared end/start URLs)\nPhase 5 — optional Stable Audio bed under narration\n```\n\n**Scene anchor triple:** same pattern as first/last frame pairing — `audio` is the third required upload per scene row. **`p-video-avatar`:** portrait + optional `last_frame_image` + uploaded `audio`.\n\n### A′ — Post-mux narration (fallback only)\n\nUse only when you already have silent clips and cannot re-render. Risk: TTS longer than clip slots → cut-off VO.\n\n```text\nPhase 3 — p-video I2V without audio → concat → mux TTS in ffmpeg\n```\n\n### C — Launch / product reel (existing pattern)\n\n```text\nPhase 1 — p-video-avatar or replace reel → concat\nPhase 2 — Stable Audio bed via launch_background_music.py (bed under VO, not replacing it)\n```\n\n## ffmpeg mixing (conceptual)\n\n**Narration onto silent concat** (single VO file):\n\n```bash\nffmpeg -y -i concat_video.mp4 -i narration.mp3 \\\n  -map 0:v -map 1:a -c:v copy -c:a aac -b:a 192k -shortest output_with_vo.mp4\n```\n\n**Bed under existing narration + video** (same pattern as `launch_background_music.py`):\n\n```text\n[1:a]volume=0.12,aloop=...[bed];\n[0:a][bed]amix=inputs=2:duration=first[aout]\n```\n\nNarration / avatar dialogue stays on stream `0:a`; bed is stream `1:a` at low volume.\n\n**Bed on silent concat** — loop a short generated clip to full video length (no per-assemble Stable Audio call):\n\n```text\n[1:a]volume=0.12,aloop=loop=-1:size=2e+09[bed]  →  map video + [bed], -shortest\n```\n\nPlan field `\"reuse_bed\": true` skips regeneration when `audio/launch_bed.mp3` exists. Delete that file (or set `reuse_bed: false`) only when you want a new prompt or seed.\n\n## Intake questions (audio)\n\nAsk before generating paid audio or video:\n\n| Topic | Questions |\n|-------|-----------|\n| **Primary voice** | Narrator (Gemini TTS), on-screen avatar (`p-video-avatar`), or native `p-video` sound only? |\n| **Narration scope** | Per-scene lines vs one continuous VO track? |\n| **Music / bed** | None, instrumental bed only, or full song ([music-2.5](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/music-2.5/skills/music-2.5/SKILL.md))? |\n| **Sync strategy** | **Preferred:** TTS → Pruna upload → **`p-video` / `p-video-avatar` with `audio`** (clip length = audio). Post-mux only as fallback. |\n| **Levels** | Bed volume target (default ~0.12 under avatar VO; ~0.08–0.12 under Gemini narration)? |\n\n## Manifest fields\n\n```json\n{\n  \"narration\": { \"enabled\": true, \"voice\": \"Sulafat\", \"mode\": \"per_scene\" },\n  \"background_music\": { \"enabled\": true, \"reuse_bed\": true, \"volume\": 0.10, \"prompt\": \"Instrumental ... no vocals\" },\n  \"p_video_audio\": { \"save_audio\": true }\n}\n```\n\n## Limitations (from [P-Video on Replicate](https://replicate.com/prunaai/p-video))\n\n- Native SFX/dialogue quality varies — for premium voice realism, prefer **Gemini TTS** or **`p-video-avatar`**, then optionally mix a bed.\n- Multi-speaker native audio can drift; dedicated TTS per role is safer for narration-heavy cuts.\n- Extreme camera motion and complex multi-scene stories are weaker than **frame-anchored chaining** + per-scene prompts — see [p-video](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/p-video/skills/p-video/SKILL.md) **First / last frame chaining**.\n\n## Related\n\n- [parallel-execution.md](https://github.com/PrunaAI/pruna-skills/tree/main/policies/parallel-execution.md) — phased vs parallel when frames chain\n- [narrated-multi-scene](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/narrated-multi-scene/skills/narrated-multi-scene/SKILL.md)\n- [pruna-generative-pipeline](https://github.com/PrunaAI/pruna-skills/tree/main/docs/WORKFLOW-RECIPES.md)\n\nFile v1.0.6:references/generation-diversity.md\n\n# Generation diversity (all models)\n\nOne checklist so **every** Pruna output — **`p-image`**, **`p-video`**, try-on, avatar, replace, animate — is as **diverse** as the brief allows. Details live in linked docs; this page is the agent shortcut.\n\nUse the **full** checklist here for every generation.\n\n## Contents\n\n- [Three steps (every job)](#three-steps-every-job)\n- [Explicit prompt structure](#explicit-prompt-structure-required)\n- [Text & typography by model](#text--typography-by-model)\n- [SSoT axis derivation](#ssot-axis-derivation-sum-mod)\n- [Scenario axes](#scenario-axes-rotate-across-outputs)\n- [Render categories](#render-categories)\n- [Crowded scenes](#crowded-scenes-p-image)\n- [Body type spread](#body-type-spread)\n- [Location-matched crowds](#location-matched-crowds)\n- [Group classes](#group-classes--courses)\n- [Framing & camera](#framing--camera)\n- [Scene spice](#scene-spice-when-it-fits)\n- [Photoreal anti-slop](#photoreal-anti-slop-neon--stylized-briefs)\n- [Aspect ratio](#aspect-ratio-multi-example-sets)\n- [By model](#by-model-minimum-diversity)\n- [When not to maximize diversity](#when-not-to-maximize-diversity)\n- [Anti-patterns](#anti-patterns)\n\n## Three steps (every job)\n\n1. **[Random seed ritual](./random-seed-ritual.md) (SSoT)** — **always first**, before the prompt. Generate a fresh random string, **state it in the turn**, derive axes via [sum-mod](#ssot-axis-derivation-sum-mod). **Do not** pass the ritual string to API `seed`. **One new ritual string per independent generation**; reuse only on same-brief slop retry.\n2. **Write an [explicit prompt](#explicit-prompt-structure-required)** — name specific people, animals, objects, actions, setting, and camera/light. Add text/typography only when the brief needs it — see [text rules by model](#text--typography-by-model).\n3. **Diversify the scenario row** — change at least **two axes** from the previous output in the same session (cast, setting, camera, **`render_category_tag`**, **aspect_ratio**, creatures, props, … — unless user asked for continuity).\n4. **Log** — `ritual_seed`, axes chosen, prediction id (manifest or turn text).\n\n## Explicit prompt structure (required)\n\n**Vague prompts produce generic AI slop.** After the ritual and axis picks, every still prompt must be **specific and dynamic** — concrete nouns, frozen actions, named places. Prefer playground/creative briefs over marketing abstractions.\n\n**Name at least four of these per prompt (log tags in manifest):**\n\n| Clause | Log as | Agent must specify |\n|--------|--------|-------------------|\n| **People** | `cast_descriptor` | Named role + age band + expression (`fearless grandmother in floral apron`, not `woman`) |\n| **Animals / creatures** | `creature_tag` | Species + attitude (`otter DJ`, `luna moth knight`, `VIP anglerfish`) |\n| **Objects** | `prop_tag` | Concrete props (`vinyl record`, `chrome rocket sled`, `velvet rope`, `tiny boombox`) |\n| **Action** | `action_tag` | Frozen mid-motion verb (`scratching vinyl`, `lassoing runaway taco truck`, `cape mid-swing`) |\n| **Duration** | `duration_tag` | When timing matters (`1970s`, `8PM`, `45-minute spin class`, `Saturday-morning cartoon`) |\n| **Setting** | `setting_tag` | Named place + era + materials (`packed 1970s roller rink`, `abyss-depth jellyfish nightclub`, `Monument Valley dust storm`) |\n| **Text / typography** | `text_spec` | Only when brief needs readable type — exact strings + surface (see [by model](#text--typography-by-model)) |\n| **Camera + light** | `camera_tag`, `lighting_tag` | `fish-eye lens`, `tilt-shift macro`, `teal-magenta cinematic`, `golden hour sparkle` |\n| **Style** | `render_category_tag` | Medium (`cel-shaded anime`, `baroque oil painting`, `ink-wash storybook`, `photoreal documentary`) |\n\n**Template:**\n\n```text\n{people and/or creatures} {action} with/at {specific objects} in {named setting},\n{style or era cues}, {camera_tag}, {lighting_tag}\n```\n\n**Good examples (dynamic / specific):**\n\n```text\nDisco ball reflections on an otter DJ scratching vinyl at a packed 1970s roller rink,\nfish-eye lens, glitter confetti mid-air, funky energy\n```\n\n```text\nBioluminescent jellyfish nightclub at abyss depth, VIP anglerfish in sunglasses at velvet rope,\nteal-magenta cinematic lighting\n```\n\n```text\nCorgi cowboy lassoing a runaway taco truck through Monument Valley dust storm,\npulp western poster energy, dynamic diagonal composition\n```\n\n**Anti-pattern:** `cool cyberpunk portrait, neon vibes` — no subject, no action, no place. **Right:** name who, what they're doing, where, with which props.\n\n## Text & typography by model\n\n**Never use negation to suppress text** — `no text`, `without signs`, `no typography` often **invoke** the thing you are trying to avoid. Describe surfaces positively when you want blank walls (`plain unmarked walls`, `matte unprinted props`).\n\n| Model | Prompt upsampling | Typography in prompt |\n|-------|-------------------|----------------------|\n| **`p-image`** | **No** effective prompt upsampling | **Avoid** dense readable-type requests unless user explicitly wants `text_rendering`. Short prompts; skip `readable`, `legible`, `headline`, multi-sign lists — they drift to gibberish. Collage triggers still apply: [interactive-explainer-prompts.md](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/interactive-explainer-prompts.md) (`flat lay`, `grid`, `collage`, …). |\n\n**`p-image` text hygiene:** prefer scenes without copy. If a screen appears: `monitor soft colorful blur glow only` — not legible UI unless the user explicitly asked for readable text (then simplify the brief or drop copy).\n\n**Collage triggers (all T2I models):** still avoid `flat lay`, `packshot`, `grid`, `collage`, `montage`, `contact sheet`, `split`, `before and after` — use `single frame`, `one camera angle` instead. Full table: [interactive-explainer-prompts.md](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/interactive-explainer-prompts.md).\n\n## SSoT axis derivation (sum-mod)\n\nAfter stating `ritual_seed` (random string), derive prompt choices — sum Unicode/ASCII char codes, mod list length:\n\n```text\nRATIOS = [\"1:1\", \"16:9\", \"9:16\", \"4:3\", \"3:4\", \"3:2\", \"2:3\"]\naspect_ratio  ← RATIOS[ sum(codes(ritual_seed)) % 7 ]\ncamera_tag    ← camera_tags[ sum(codes(ritual_seed[0:4])) % len(camera_tags) ]\nrender_tag    ← render_tags[ sum(codes(ritual_seed[4:8])) % len(render_tags) ]\n```\n\n`camera_tags` and `render_tags` — see [framing & camera](#framing--camera) and [render categories](#render-categories). State derived picks in the turn (*\"Aspect ratio: 16:9, camera: over-shoulder\"*).\n\n**User `api_seed`:** when the user supplies an integer for reproducibility, pass it as `input.seed` — separate from the ritual string.\n\n## Scenario axes (rotate across outputs)\n\n| Axis | Vary with | Applies to |\n|------|-----------|------------|\n| **Cast** | age, ethnicity, gender, archetype, **hairstyle**, **body type** (rotate — see [below](#body-type-spread)), disability aids (wheelchair, cane), visible age band twice in prompt | all person/content gens |\n| **Medium** | `render_category_tag` — rotate across [render categories](#render-categories) | `p-image`, avatar stills |\n| **Setting** | unique `setting_tag` — specific room/street/venue/era, not repeat adjacent rows | stills + video plates |\n| **Camera** | `camera_tag` — rotate across [framing ladder](#framing--camera); never default MC facing lens | stills, `video_prompt` |\n| **Lighting** | `lighting_tag` — golden hour · neon · overcast · practical | stills, video mood |\n| **Motion** | unique `video_prompt` per clip | `p-video`, `p-video-avatar`, animate |\n| **Voice** | natural `voice_script`; one `voice` preset per character | avatar, TTS-led video |\n| **Seed** | new ritual string per **independent** job; reuse only on same-brief slop retry | all generation skills |\n| **Aspect ratio** | different `aspect_ratio` per independent still in a batch — see [below](#aspect-ratio-multi-example-sets) | `p-image`, `p-image-edit` |\n| **Crowd density** | layered background population + activity cues — see [below](#crowded-scenes-p-image) | `p-image` plates with busy worlds |\n\nFull style/camera/lighting ladders: [visual-variety-bible.md](https://github.com/PrunaAI/pruna-skills/tree/main/shared/visual-variety-bible.md). Persona + try-on bar: [realistic-persona-showcase.md](https://github.com/PrunaAI/pruna-skills/tree/main/shared/realistic-persona-showcase.md).\n\n## Render categories\n\nRotate **`render_category_tag`** (and log it) so diversity batches cover more than photoreal portraits or anime. Category families below mirror arena leaderboards — pick a **different tag per independent output**.\n\n**Random seed ritual still applies** to every generation in [step 1](#three-steps-every-job); categories describe *what* to vary, not *when* to pick `seed`.\n\n### Text-to-image — `p-image`\n\nSources: [Arena text-to-image](https://arena.ai/leaderboard/text-to-image) · [AA text-to-image](https://artificialanalysis.ai/image/leaderboard/text-to-image)\n\n**Unified `render_category_tag`** (Arena bucket = tag — pick one per still):\n\n`product_branding_commercial` · `3d_imaging_modeling` · `cartoon_anime_fantasy` · `photoreal_cinematic` · `art` · `portraits` · `nature_environment` · `animals_creature` · `text_rendering`\n\n| Tag | Typical prompt lane |\n|-----|---------------------|\n| `product_branding_commercial` | single product on seamless studio, person + product in named setting, showroom (not `flat lay` / `packshot` words) |\n| `3d_imaging_modeling` | CG film still, clay/stop-motion, rounded 3D forms |\n| `cartoon_anime_fantasy` | cel anime, fantasy character, crowded stylized world |\n| `photoreal_cinematic` | documentary crowd scenes, film-scale wide, urban march |\n| `art` | oil, watercolor, gouache, charcoal, flat vector |\n| `portraits` | single-subject editorial or documentary portrait (crowd optional behind) |\n| `nature_environment` | landscape-wide; subject small in frame |\n| `animals_creature` | named species + handler; crowded market/park when it fits |\n| `text_rendering` | **user-requested only** — otherwise no readable text |\n\nLog `render_category_tag` in manifest. Combine with [crowded scenes](#crowded-scenes-p-image), [body type](#body-type-spread), and [scene spice](#scene-spice-when-it-fits) when the brief allows.\n\n### Image edit — `p-image-edit`\n\nSources: [Arena image edit](https://arena.ai/leaderboard/image-edit) · [AA image editing](https://artificialanalysis.ai/image/leaderboard/editing)\n\nArena modalities: `single_image_edit` · `multi_image_edit`\n\nEdit diversity tags: `background_swap` · `relight` · `wardrobe_on_plate` · `pose_or_angle_delta` · `multi_ref_composite` · `region_inpaint`\n\nVary **instruction** and **what changes** while identity URL stays fixed on character arcs.\n\n### Text-to-video — `p-video`\n\nSources: [Arena text-to-video](https://arena.ai/leaderboard/text-to-video) · [AA text-to-video](https://artificialanalysis.ai/video/leaderboard/text-to-video)\n\nMotion/scene tags: `character_performance` · `landscape_broll` · `urban_street` · `product_demo` · `abstract_mood` · `crowd_scene` · `dialogue_beat`\n\nRotate `video_prompt` grammar, start plate world, and `camera_tag` per clip.\n\n### Image-to-video — `p-video` (+ plate upload)\n\nSources: [Arena image-to-video](https://arena.ai/leaderboard/image-to-video) · [AA image-to-video](https://artificialanalysis.ai/video/leaderboard/image-to-video)\n\nPlate-driven tags: `animate_hero_still` · `camera_move_on_plate` · `environmental_parallax` · `avatar_lip_sync` · `hands_or_prop_motion`\n\nMatch motion to what the **still** already shows — do not contradict the plate.\n\n### Video edit — `p-video-replace` (and edit-style video)\n\nSource: [Arena video edit](https://arena.ai/leaderboard/video-edit)\n\nEdit tags: `face_recast` · `wardrobe_swap` · `accessory_swap` · `background_replace` · `object_in_hand_swap` · `style_transfer_on_subject`\n\nSame-gender / identity rules for talking-head beats still apply — see [visual-variety-bible.md](https://github.com/PrunaAI/pruna-skills/tree/main/shared/visual-variety-bible.md).\n\n## Crowded scenes (`p-image`)\n\nWhen the brief asks for **busy**, **crowded**, or **lively** worlds — not a lone subject on a blank wall — stack density in the prompt:\n\n1. **Three depth layers** — sharp foreground subject · readable midground faces/hands/props · landmark bokeh (stage, temple, billboards, ferris wheel).\n2. **Named population count** — `hundreds of pedestrians`, `dozens of faces in midground`, `20+ tiny clay figures` (stylized sets need explicit counts; models under-deliver on vague \"busy\").\n3. **Activity verbs** — raised hands, umbrellas open, food steam, confetti, market haggling, commuters pressed shoulder-to-shoulder.\n4. **Shallow DOF + single subject** — `single subject one frame` keeps one identity readable while the crowd stays behind them.\n5. **Age & angle lock** — repeat age band twice (`woman in her late 50s, visibly fifty`) and use [framing & camera](#framing--camera) — models drift younger, center-frame, and front-facing without it.\n\n| Crowd family | Density cues |\n|--------------|--------------|\n| **Urban rush** | crosswalk stripes, wet reflections, umbrellas, billboard bokeh |\n| **Festival / parade** | confetti, raised hands, costume layers, smoke haze |\n| **Market / bazaar** | overflowing stalls, hanging goods, steam, price tags as color blobs |\n| **Transit crush** | strap hangers, door windows, blurred faces pressed together |\n| **Stylized miniature** | counted clay/figurine shoppers (`20+`), cramped aisle, stacked crates |\n| **Institutional / ER** | framed oil portraits on beige walls, triage number board, wall sanitizer, vending machine, scuffed linoleum, TV blur, mixed-age seated patients |\n| **Urban march / protest** | named city, local landmarks, multiracial crowd cues separate from hero — see [location-matched crowds](#location-matched-crowds) |\n| **Group fitness class** | class name + duration, mixed-gender riders, realistic warm studio light — see [group classes](#group-classes--courses) |\n\n**Anti-pattern:** one blurred smear behind a portrait — name **what** the crowd is doing and **where** layers sit. **Institutional** scenes (ER, airport, classroom) need `benches full`, `standing room only`, or `shoulder-to-shoulder` — otherwise models default to a quiet hallway. Name **set dressing** too: framed portraits on walls, triage number board, vending machine glow, scuffed linoleum — generic mint corridors read AI-empty.\n\n## Body type spread\n\nModels default to one “average fitness” body. In diversity batches, **name build on the hero and vary background bodies**:\n\n| Build tag | Prompt cue |\n|-----------|------------|\n| **Plus-size / curvy** | `plus-size`, `curvy build`, `full-figured` |\n| **Athletic / muscular** | `broad shoulders`, `muscular arms`, `athletic build` |\n| **Petite / slim** | `petite frame`, `slim build`, `narrow shoulders` |\n| **Tall / lanky** | `tall and lanky`, `6-foot frame`, `long limbs` |\n| **Stocky / heavyset** | `stocky build`, `heavyset`, `barrel chest` |\n| **Lean wiry** | `lean wiry frame`, `weathered thin face` |\n\n**Rule:** rotate build across independent panels in a session — not every hero “athletic build”. Background crowd should mix ages **and** silhouettes (`elderly thin woman`, `heavyset man`, `pregnant woman seated`, `toddler on lap`).\n\n## Location-matched crowds\n\nWhen the prompt names a **real city or country**, background faces must match that place’s **demographic mix** — not clone the hero’s ethnicity.\n\n| Wrong | Right |\n|-------|--------|\n| South Asian hero + only South Asian protesters in “New York” | Hero is one identity; crowd explicitly `multiracial NYC march — Black, Latino, white, East Asian protesters` |\n| “Dense city march” with no geography | Name city + 3–4 crowd ethnicity cues + local landmarks (yellow cabs, art deco towers, steam vent) |\n| Festival in Lagos with only Nordic faces | Match crowd to `setting_tag` region |\n\n**Prompt pattern:** lock hero cast in sentence 1; sentence 2 lists **four+ distinct background silhouettes** unrelated to hero ethnicity; sentence 3 names **local landmarks** so the plate cannot read as generic stock.\n\n**Applies to:** protests, airports, transit, street markets, sports crowds — any scene where “crowded” implies a real place.\n\n## Group classes & courses\n\nWhen the scene is a **class, workshop, or team activity**, name the **course type** and **who else is in the room** — models default to monochrome crowds (all men, all one age).\n\n| Specify | Example cues |\n|---------|----------------|\n| **Class type** | `45-minute evening spin class`, `beginner yoga flow`, `HIIT bootcamp circuit` |\n| **Room realism** | warm overhead track lights, mirror wall, rubber floor, water bottles, towels — **not** magenta-cyan neon strips unless brief is explicitly nightclub |\n| **Gender mix** | hero is one person; crowd `mixed-gender class — women with ponytails, men with beards, nonbinary cyclist` |\n| **Body + age mix** | plus-size rider, petite woman, athletic man, woman in her 50s — same as [body type spread](#body-type-spread) |\n\n**Lighting rule for fitness:** real boutique studios are **dim warm overhead** or **single spotlight on instructor** — avoid `split gel`, `neon LED strips`, `magenta-cyan` on photoreal gym plates; those read AI-fake.\n\n**Prompt pattern:** `Documentary fitness portrait` + class name + instructor on bike at front + `20+ mixed-gender cyclists` with 3–4 named background silhouettes + realistic room props.\n\n## Framing & camera\n\nModels default to **centered subject, eyes at camera**. In diversity batches, **rotate `camera_tag` and frame placement** every row — log both in manifest.\n\n**Gaze rule:** `glance off-lens`, `profile`, `back to camera`, `looking down at [prop]`, or `watching the crowd` — **not** `facing camera` or `looking at viewer` unless the user asked for a direct-address avatar plate.\n\n**Placement rule:** name where the subject sits in frame — `left third`, `right third`, `lower right corner`, `edge of frame`, `small in environmental wide` — **not** centered mugshot every time.\n\n| `camera_tag` | Prompt cue |\n|--------------|------------|\n| **Overhead / bird's eye** | `overhead aerial view`, `top-down`, `drone shot looking straight down` |\n| **High corner** | `high angle from corner`, `surveillance-style downward angle` |\n| **Worm's eye** | `ground-level worm's eye`, `camera on pavement` |\n| **Crane-down** | `slight high angle crane-down` |\n| **Over-shoulder** | `over-shoulder from behind`, `seen past someone's shoulder` |\n| **Profile / side** | `profile side angle`, `walking across frame` |\n| **From behind** | `back to camera`, `three-quarter from behind` |\n| **Dutch tilt** | `dutch tilt` — tension scenes only |\n| **Through crowd** | `subject visible through gap in crowd`, `foreground heads out of focus` |\n\n**Batch rule:** no two adjacent stills share the same `camera_tag` **and** placement corner (e.g. don't do `left third` twice in a row).\n\nAvatar / lip-sync exception: face must stay readable and mouth visible — use `slight angle from the side` or `three-quarter`, still **off-center** and **off-lens gaze** when not delivering VO to camera.\n\n## Scene spice (when it fits)\n\nDefault plates are person + crowd + place. Add **one or two specific attributes** when the setting naturally supports them — not random clutter on every row.\n\n| Spice type | When to add | Example |\n|------------|-------------|---------|\n| **Animals** | setting implies them | dog park → `golden retriever on leash`; harbor → `seagulls overhead`; rooftop → `pigeons on water tower`; parade → `police horse midground` |\n| **Held / worn props** | role or weather | `red umbrella tucked under arm`, `wire beekeeper smoker`, `chipped ceramic mug`, `sample strawberry basket` |\n| **Micro-detail** | one thumb-stopping oddity | `muddy paw prints on pavement`, `honey jar on crate`, `green parade beads on fence` |\n\nCamera and placement live in [framing & camera](#framing--camera) — not optional spice.\n\n**Rule:** pick **at most two** spice items per prompt. They must answer “what would a photographer notice here?” — not a checklist dump.\n\n**Skip spice when:** product hero, avatar MC talking head, try-on full-body (garment is the focus), or minimal studio brief.\n\n## Photoreal anti-slop (neon / stylized briefs)\n\nStylized settings still need **documentary skin discipline** or outputs go waxy:\n\n- Lead with `documentary portrait, natural skin pores, not CGI, not illustration` even for neon/cyberpunk worlds.\n- Prefer **worn real materials** — matte leather, faded denim, scratched CRT bezels, sticky carpet — over `holographic puffer`, `chrome armor`, `HUD`.\n- Name **gritty location cues** — basement arcade, wet alley, scuffed linoleum — not abstract `neon corridor`.\n- Background crowd faces need **imperfect texture**; blur is fine, plastic skin in midground is not.\n\n## Aspect ratio (multi-example sets)\n\nWhen generating **two or more** stills in one session (playground grid, demo batch, mood board), give each independent output a **different** `aspect_ratio` unless the user locked a format.\n\n**Allowed `p-image` values:** `1:1` · `16:9` · `9:16` · `4:3` · `3:4` · `3:2` · `2:3`\n\n**How to pick:** after the [random seed ritual](./random-seed-ritual.md), use [sum-mod](#ssot-axis-derivation-sum-mod) on `ritual_seed` — state it in the turn (*\"Aspect ratio: 16:9\"*). Do **not** default every example to `9:16` or `1:1`.\n\n| Ratio | Typical use |\n|-------|-------------|\n| `9:16` | vertical UGC, full-body fashion, avatar talking head |\n| `16:9` | environmental wide, cinematic landscape plate |\n| `3:4` | editorial portrait, try-on full-body |\n| `4:3` | classic portrait, product + person |\n| `1:1` | packshot grid, social tile |\n| `3:2` · `2:3` | magazine / poster crops |\n\nMatch prompt framing to ratio (e.g. `16:9 horizontal wide shot`, `9:16 vertical full body`). **`p-image-try-on`** inherits plate size when `preserve_input_size: true` — diversify person plates first.\n\n**Same character arc:** one ratio for the whole chain unless the user asks for reframes.\n\n## By model (minimum diversity)\n\n| Model | Besides ritual seed, always vary |\n|-------|-----------------------------------|\n| **`p-image`** | cast/creature + objects + action + setting + camera + **`render_category_tag`** + **aspect_ratio**; [explicit structure](#explicit-prompt-structure-required); [text hygiene](#text--typography-by-model) (no upsampling) |\n| **`p-image-edit`** | edit tag + setting/angle delta; same identity URL |\n| **`p-image-try-on`** | person plate world + garment complexity; preserve scene |\n| **`p-image-upscale`** | N/A on prompt — diversify **source** stills |\n| **`p-video`** | motion/scene tag + `video_prompt`; differ start plates per scene |\n| **`p-video-avatar`** | `video_prompt` + still world per scene; lock voice per character |\n| **`p-video-animate`** | persona still style/setting per slider ref |\n| **`p-video-replace`** | video-edit tag + full cast spread on showcase reels |\n\n## When **not** to maximize diversity\n\n- **Same character arc** — lock hero plate URL, one `voice`, cast descriptor; vary only setting/angle/motion per scene.\n- **User asked for continuity** — match their cast and approved plates.\n- **Draft → final** — same prompt; change only `draft: false`. Use `api_seed` only if user locked API reproducibility.\n\n## Anti-patterns\n\n| Wrong | Right |\n|-------|--------|\n| Copy doc example ritual strings | [Random seed ritual](./random-seed-ritual.md) — fresh string each time |\n| Pass ritual string as API `seed` | Ritual is SSoT planning only; `api_seed` when user requests |\n| White wall + MC CU on every demo | Rotate setting + camera + cast |\n| One `video_prompt` for whole reel | Unique motion per scene row |\n| New ritual string mid avatar chain on same brief | Reuse `ritual_seed` until recast or new independent output |\n| Same aspect ratio on every playground example | Rotate `1:1` · `16:9` · `9:16` · `4:3` · `3:4` · `3:2` · `2:3` per [aspect ratio rules](#aspect-ratio-multi-example-sets) |\n| Every hero same athletic body | Rotate [body type spread](#body-type-spread) |\n| Generic hospital hallway | Named ER set dressing + mixed body types in crowd |\n| `holographic` / `chrome` on photoreal cyber scenes | Worn leather, scratched cabinets, documentary skin cues |\n| Monoculture crowd in a named global city | [Location-matched crowds](#location-matched-crowds) — hero ≠ background ethnicity |\n| Magenta-cyan neon on photoreal gym | Warm overhead studio light, mirror wall, real spin bikes |\n| All-male or all-female group class | [Group classes](#group-classes--courses) — mixed-gender background cues |\n| Centered subject every frame | [Framing & camera](#framing--camera) — rotate `camera_tag` + placement |\n| Subject facing camera / at viewer | Off-lens gaze, profile, from behind, or watching crowd |\n| Random animals with no setting reason | Animals only when place implies them |\n| Every stylized panel is anime | Rotate [render categories](#render-categories) — use `cartoon_anime_fantasy` at most once per batch |\n| Vague `cool portrait, neon vibes` | [Explicit structure](#explicit-prompt-structure-required) — named subject, action, objects, setting |\n| `no text` / `without signage` in prompt | Negation invokes text — use [text rules by model](#text--typography-by-model) |\n| Dense typography on **`p-image`** | Drop copy or simplify the brief — `p-image` has no prompt upsampling |\n\n## Related\n\n- [generation-quality-checklists.md](./generation-quality-checklists.md) — core + model checklists\n- [staged-generation-gate.md](https://github.com/PrunaAI/pruna-skills/tree/main/references/policies/staged-generation-gate.md) — approval phases\n\nFile v1.0.6:references/generation-quality-checklists.md\n\n# Generation quality checklist hub\n\nUse this as the shared quality gate across models and workflows.\nRun the **Core checklist** for every generation job, then run the model-specific checklist.\n\n## Who applies these checklists?\n\n**The coding agent** — by **opening the real output files** (images, video, or audio) and reviewing them with vision. These checklists are **not** automated test scripts. There is no separate scoring service: the agent reads each item and judges pass or fail from what it sees and hears.\n\nTypical flow:\n\n1. **Generate or download** the asset to a local path (`stills/`, `clips/`, etc.).\n2. **Inspect the file** — view the image, watch the video clip, or listen to narration when the checklist covers audio.\n3. Run the **Core checklist** (below), then the **model-specific checklist** for that job.\n4. **If something fails** — note which items failed, adjust prompt / settings / seed, and regenerate **only that asset** (do not advance to expensive video steps on a bad still).\n5. **If it passes** — show the user the file paths (and previews when helpful). In workflows, still follow [staged-generation-gate.md](https://github.com/PrunaAI/pruna-skills/tree/main/references/policies/staged-generation-gate.md): agent checklist review happens **before** you ask the user to approve stills or clips.\n\nThe user's **approve plan / approve stills / approve clips** gates are separate. Agent checklists catch obvious problems early so the user is not asked to sign off on broken outputs.\n\nMaintenance rule: keep tool/workflow mapping only in this file to avoid link drift.\n\n## Match map (tool -> checklist -> workflows)\n\n| Tool/model | Checklist | Common workflows |\n|------------|-----------|---------------|\n| `p-image` | [`p-image-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/image/p-image-quality-checklist.md) · persona bar: [`realistic-persona-showcase.md`](https://github.com/PrunaAI/pruna-skills/tree/main/shared/realistic-persona-showcase.md) | [`image-to-video`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/image-to-video/skills/image-to-video/SKILL.md), [`narrated-multi-scene`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/narrated-multi-scene/skills/narrated-multi-scene/SKILL.md), [`WORKFLOW-RECIPES`](https://github.com/PrunaAI/pruna-skills/tree/main/docs/WORKFLOW-RECIPES.md) |\n| `p-image-edit` | [`p-image-edit-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/image/p-image-edit-quality-checklist.md) | [`avatar-single-scene`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/avatar-single-scene/skills/avatar-single-scene/SKILL.md), [`avatar-multi-scene`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/avatar-multi-scene/skills/avatar-multi-scene/SKILL.md), [`WORKFLOW-RECIPES`](https://github.com/PrunaAI/pruna-skills/tree/main/docs/WORKFLOW-RECIPES.md) |\n| `p-image-upscale` | [`p-image-upscale-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/image/p-image-upscale-quality-checklist.md) | [`image-to-video`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/image-to-video/skills/image-to-video/SKILL.md), [`narrated-multi-scene`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/narrated-multi-scene/skills/narrated-multi-scene/SKILL.md), [`WORKFLOW-RECIPES`](https://github.com/PrunaAI/pruna-skills/tree/main/docs/WORKFLOW-RECIPES.md), [`generate_upscale_comparison.py`](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/_shared/scripts/generate_upscale_comparison.py) |\n| `p-image-try-on` | [`p-image-try-on-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/image/p-image-try-on-quality-checklist.md) | [`WORKFLOW-RECIPES`](https://github.com/PrunaAI/pruna-skills/tree/main/docs/WORKFLOW-RECIPES.md), [`p-image-try-on`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/p-image-try-on/skills/p-image-try-on/SKILL.md), [`realistic-persona-showcase.md`](https://github.com/PrunaAI/pruna-skills/tree/main/shared/realistic-persona-showcase.md) |\n| `p-video` | [`p-video-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/video/p-video-quality-checklist.md) | [`image-to-video`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/image-to-video/skills/image-to-video/SKILL.md), [`narrated-multi-scene`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/narrated-multi-scene/skills/narrated-multi-scene/SKILL.md), [`interactive-explainer`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/interactive-explainer/skills/interactive-explainer/SKILL.md), [`visual-transition-reel`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/visual-transition-reel/skills/visual-transition-reel/SKILL.md), [`WORKFLOW-RECIPES`](https://github.com/PrunaAI/pruna-skills/tree/main/docs/WORKFLOW-RECIPES.md) |\n| `p-video-avatar` | [`p-video-avatar-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/video/p-video-avatar-quality-checklist.md) · persona bar: [`realistic-persona-showcase.md`](https://github.com/PrunaAI/pruna-skills/tree/main/shared/realistic-persona-showcase.md) | [`avatar-single-scene`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/avatar-single-scene/skills/avatar-single-scene/SKILL.md), [`avatar-multi-scene`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/avatar-multi-scene/skills/avatar-multi-scene/SKILL.md), [`interactive-explainer`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/interactive-explainer/skills/interactive-explainer/SKILL.md), [`WORKFLOW-RECIPES`](https://github.com/PrunaAI/pruna-skills/tree/main/docs/WORKFLOW-RECIPES.md) |\n| `p-video-animate` | [`p-video-animate-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/video/p-video-animate-quality-checklist.md) | [`avatar-multi-scene`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/avatar-multi-scene/skills/avatar-multi-scene/SKILL.md), [`WORKFLOW-RECIPES`](https://github.com/PrunaAI/pruna-skills/tree/main/docs/WORKFLOW-RECIPES.md) |\n| `p-video-replace` | [`p-video-replace-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/video/p-video-replace-quality-checklist.md) | [`p-video-replace`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/p-video-replace/skills/p-video-replace/SKILL.md), [`generate_video_comparison.py`](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/_shared/scripts/generate_video_comparison.py) |\n| `music-2.5` + music video assembly | [`music-video-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/music-video-quality-checklist.md) | [`music-video`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/music-video/skills/music-video/SKILL.md), [`music-2.5`](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/music-2.5/skills/music-2.5/SKILL.md) |\n\n## Core checklist (all models)\n\n- **[Generation diversity](./generation-diversity.md)** — ritual seed + rotate scenario axes on **every** model (image, video, try-on, avatar, …).\n- **[Random seed ritual](./random-seed-ritual.md) (SSoT)** — generate and state a ritual string **before** every generation; derive prompt axes via sum-mod; never copy example strings from docs.\n- Goal and acceptance criteria are explicit (what \"good\" looks like is written down).\n- Input assets are valid and licensed (URL/file reachable, rights cleared).\n- Prompt and settings match the intended output format (`aspect_ratio`, duration, resolution, style lock). **Video default:** `720p`, `24` fps unless the brief asks for final `1080p` / `48`.\n- Output contains no accidental watermarks, UI overlays, or stray text unless requested.\n- Brand, legal, and safety constraints are satisfied before handoff.\n- Manifest/log captures model, input fields, prediction id, output URL, and **`ritual_seed`** for traceability.\n\n## Model-specific checklists\n\n- [`p-image-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/image/p-image-quality-checklist.md)\n- [`p-image-edit-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/image/p-image-edit-quality-checklist.md)\n- [`p-image-upscale-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/image/p-image-upscale-quality-checklist.md)\n- [`p-image-try-on-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/image/p-image-try-on-quality-checklist.md)\n- [`p-video-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/video/p-video-quality-checklist.md)\n- [`p-video-avatar-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/video/p-video-avatar-quality-checklist.md)\n- [`p-video-animate-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/video/p-video-animate-quality-checklist.md)\n- [`p-video-replace-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/video/p-video-replace-quality-checklist.md)\n- [`music-video-quality-checklist.md`](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/music-video-quality-checklist.md)\n\n## Visual variety (launch reels)\n\nBefore **any** generation, run [generation-diversity.md](./generation-diversity.md). Launch reels: also [visual-variety-bible.md](https://github.com/PrunaAI/pruna-skills/tree/main/shared/visual-variety-bible.md) **Variety checklist**. Persona/playground bar: [realistic-persona-showcase.md](https://github.com/PrunaAI/pruna-skills/tree/main/shared/realistic-persona-showcase.md).\n\nFor phased human review before expensive video jobs, see [staged-generation-gate.md](https://github.com/PrunaAI/pruna-skills/tree/main/references/policies/staged-generation-gate.md) and the per-skill index [workflow-feedback-gates.md](https://github.com/PrunaAI/pruna-skills/tree/main/references/policies/workflow-feedback-gates.md).\n\n## Workflow note\n\nFor multi-scene projects, run these checks per scene and add a final continuity pass\n(style, character identity, voice, and pacing consistency across scenes).\n\n**Narrated cinematic B-roll:** validate [scene anchor triple](https://github.com/PrunaAI/pruna-skills/tree/main/video/scene-anchor-triple.md) inputs before `p-video` — start still, end still, uploaded narration URL per row.\n\nFile v1.0.6:references/random-seed-ritual.md\n\n# Random seed ritual (SSoT — mandatory before every generation)\n\nThe random seed ritual is a lean [String Seed of Thought](https://pub.sakana.ai/ssot/) (DAG) protocol. **Every** Pruna generation — every prompt, every `POST /v1/predictions`, every scene row — starts here.\n\nThis prevents copy-pasting example strings (`k7Qm2xP9`, `482901`, …) and reduces accidental duplicate outputs across sessions.\n\n## The ritual (do this first)\n\nBefore writing prompts, curl, or runner JSON:\n\n1. **Generate a random string** in-agent (8–16 chars, mixed case + digits).\n2. **Log it** as `ritual_seed` in the manifest / internal plan. Do **not** require a user-visible *\"Ritual seed: …\"* line unless the user asks for transparency.\n3. **Derive prompt choices** from the string — sum char codes, mod N — pick axes from [generation-diversity.md](./generation-diversity.md) (`aspect_ratio`, `camera_tag`, `render_category_tag`, …).\n4. **Write the prompt** using [explicit prompt structure](./generation-diversity.md#explicit-prompt-structure-required) and derived axes.\n5. **Record** axes chosen and prediction id in the manifest alongside `ritual_seed`.\n\n**Do not pass the ritual string to API `seed`.** API runs without `seed` unless the user explicitly requests reproducibility (`api_seed`).\n\n**Never** proceed to `POST /v1/predictions` without completing steps 1–2 (unless the user supplied an explicit `api_seed` — see below).\n\n## Reuse rules\n\n| Situation | Action |\n|-----------|--------|\n| **New hero / independent still / mood-board panel** | Fresh ritual string |\n| **Same-brief slop retry** | Reuse same `ritual_seed`; note `retry_ritual_seed` in manifest |\n| **Same character arc** | Lock **hero plate URL** + cast descriptor; reuse `ritual_seed` only on same-brief regen |\n| **User says \"lock seed\" / provides integer** | Pass **their** number as `api_seed` → `input.seed`; skip new ritual for that chain |\n\nCharacter continuity = approved plate URL + cast descriptor — **not** the ritual string on the API.\n\n## Anti-patterns\n\n| Wrong | Right |\n|-------|--------|\n| Copy example strings from SKILL.md | Fresh ritual string each independent generation |\n| Pass ritual string as API `seed` | Ritual is planning-only; `api_seed` only when user asks |\n| One ritual string for entire mood board | New ritual per independent **`p-image`** |\n| Skip ritual because API `seed` is optional | Ritual always; API omits `seed` by default |\n\n## Example (internal plan / optional user-visible)\n\nManifest: `\"ritual_seed\": \"k7Qm2xP9\"`. Derived: aspect_ratio 16:9, camera_tag fish-eye, render_category_tag cartoon_anime_fantasy.  \nPrompt: Disco ball reflections on an otter DJ scratching vinyl at a packed 1970s roller rink, fish-eye lens, glitter confetti mid-air, funky energy.  \n…then curl / runner **without** `\"seed\"` in `input`.\n\n## Manifest snippet\n\n```json\n{\n  \"ritual_seed_policy\": \"ssot_dag_before_every_generation\",\n  \"ritual_seed\": \"k7Qm2xP9\",\n  \"seed_log\": [\n    { \"phase\": \"hero_p_image\", \"ritual_seed\": \"k7Qm2xP9\", \"creature_tag\": \"otter_dj\", \"setting_tag\": \"1970s_roller_rink\", \"prompt_hash\": \"…\" },\n    { \"phase\": \"scene_2_avatar\", \"ritual_seed\": \"k7Qm2xP9\", \"scene_id\": 2 }\n  ]\n}\n```\n\n## Where this applies\n\nAll Pruna generation skills and workflow runners — **every invocation**:\n\n- **`p-image`**, **`p-image-edit`**, **`p-image-try-on`**, **`p-image-upscale`**\n- **`p-video`**, **`p-video-avatar`**, **`p-video-animate`**, **`p-video-replace`**\n\n## Related\n\n- [generation-diversity.md](./generation-diversity.md) — ritual + axis rotation + sum-mod derivation\n- [realistic-persona-showcase.md](https://github.com/PrunaAI/pruna-skills/tree/main/shared/realistic-persona-showcase.md) — persona planning\n- [staged-generation-gate.md](https://github.com/PrunaAI/pruna-skills/tree/main/references/policies/staged-generation-gate.md) — approval phases\n- [approval-red-flags.md](https://github.com/PrunaAI/pruna-skills/tree/main/references/policies/approval-red-flags.md) — red flags\n\nFile v1.0.6:references/replicate-api.md\n\n# Replicate API (minimal)\n\nUsed by external tool skills (e.g. [stable-audio-2.5](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/stable-audio-2.5/skills/stable-audio-2.5/SKILL.md), [gemini-3.1-flash-tts](../SKILL.md)).\n\n**Missing token:** agents must stop and point the user to [api-credentials.md](./api-credentials.md) — sign up at [replicate.com/account/api-tokens](https://replicate.com/account/api-tokens) ([sign in](https://replicate.com/signin) if needed), then `export REPLICATE_API_TOKEN=r8_...`.\n\n## Auth\n\n```bash\nexport REPLICATE_API_TOKEN=r8_...\n```\n\nHeader: `Authorization: Bearer ${REPLICATE_API_TOKEN}`\n\n## Create + poll\n\n```bash\n# POST https://api.replicate.com/v1/models/{owner}/{name}/predictions\n# Body: {\"input\": { ... }}\n\n# Poll GET on response.urls.get until status == succeeded\n# Download output URL (string or list depending on model)\n```\n\nShared client: [`workflows/_shared/scripts/replicate_api.py`](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/_shared/scripts/replicate_api.py)\n\n## Stable Audio 2.5\n\nModel: `stability-ai/stable-audio-2.5`  \nRequired input: `prompt`  \nOptional: `duration` (1–190), `steps` (4–8), `cfg_scale`, `seed`\n\n## Music 2.5 (MiniMax)\n\nModel: `minimax/music-2.5`  \nRequired input: `lyrics` (1–3,500 chars, structure tags supported)  \nOptional: `prompt` (style), `sample_rate`, `bitrate`, `audio_format` (`mp3` default)\n\nWorkflow: [music-video](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/music-video/skills/music-video/SKILL.md) · tool skill: [music-2.5](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/music-2.5/skills/music-2.5/SKILL.md)\n\n## Gemini 3.1 Flash TTS\n\nModel: `google/gemini-3.1-flash-tts`  \nRequired input: `text`  \nOptional: `voice` (default `Kore`), `prompt` (style/scene), `language_code` (default `en-US`)\n\nOutput: audio file URL. Use for narration — upload to Pruna as part of [scene anchor triple](./scene-anchor-triple.md) (`input.audio` + `input.image` + `input.last_frame_image` on `p-video`). Layering with beds: [audio-post-production.md](./audio-post-production.md)\n\nFile v1.0.6:references/scene-anchor-triple.md\n\n# Scene anchor triple (single narrated beat → multi-scene extension)\n\nCanonical payload pattern for **one narrated `p-video` prediction**: three uploaded anchors (`image`, `last_frame_image`, `audio`) plus a motion **`prompt`**. Use this for a **single story beat** first ([image-to-video](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/image-to-video/skills/image-to-video/SKILL.md), [p-video](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/p-video/skills/p-video/SKILL.md)).\n\n**Multi-scene extension** (`frame_chain`, concat, parallel batches, plan JSON with many rows) belongs only in [narrated-multi-scene](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/narrated-multi-scene/skills/narrated-multi-scene/SKILL.md) — do not treat this doc as permission for single-clip skills to orchestrate full films.\n\nRelated: [scene-anchor-pair.md](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/scene-anchor-pair/SKILL.md) (visual-only) · [audio-post-production.md](./audio-post-production.md) · [p-video](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/p-video/skills/p-video/SKILL.md)\n\n## The triple (one prediction)\n\nEach beat supplies **three Pruna file URLs** (from `POST /v1/files`) plus a motion **`prompt`**:\n\n| Anchor | `input` field | Role |\n|--------|---------------|------|\n| **First frame** | `image` | Opening composition |\n| **Last frame** | `last_frame_image` | Closing composition |\n| **Narration / VO / music slice** | `audio` | Sets **clip duration** (min(audio length, **20s** P-API max)); model syncs motion to speech or beats |\n\n**Omit `duration`** when `audio` is set. Optional **`save_audio`: true** keeps narration on the output clip.\n\nWhen audio is provided, **always** upload and pass it to `p-video` at render time. Do not generate silent clips and mux narration in ffmpeg afterward.\n\n**20-second ceiling:** audio-led clips cannot run longer than P-API `duration` max (**20s**). Write TTS to **≤ ~19s** (probe with `ffprobe` after Gemini). Truncated VO with “audio passed” usually means the line was too long, not that `input.audio` was missing. Helper: [`validate_narration_duration`](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/_shared/scripts/p_video_payload.py).\n\n**Over-long narration — fix in order:** (1) **shorten** copy; (2) **pace** — brisk delivery in TTS `style_prompt`; (3) **split** into a new beat (that becomes a multi-scene project — hand off to narrated-multi-scene). Never rely on post-mux to extend a silent clip.\n\n```json\n{\n  \"prompt\": \"Dog tosses plush upward, tail wagging, motion matches narrator, warm light\",\n  \"image\": \"https://api.pruna.ai/v1/files/START_ID\",\n  \"last_frame_image\": \"https://api.pruna.ai/v1/files/END_ID\",\n  \"audio\": \"https://api.pruna.ai/v1/files/NARRATION_ID\",\n  \"resolution\": \"720p\",\n  \"fps\": 24,\n  \"save_audio\": true\n}\n```\n\n## Stills phase (`p-image-edit`)\n\n| Still | Typical source | Plan field |\n|-------|----------------|------------|\n| Start | Hero + `edit_prompt` | `edit_prompt` |\n| End | Start still + `last_frame_edit_prompt` | `last_frame_edit_prompt` |\n\nFor a **single beat**, generate start then end (or both once start exists). Parallel fan-out across many scenes is a **multi-scene** concern — see below.\n\n## Audio phase (Replicate → Pruna)\n\n1. [Gemini 3.1 Flash TTS](../SKILL.md) (or [Music 2.5](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/music-2.5/skills/music-2.5/SKILL.md) slice for music videos)\n2. Download MP3/WAV\n3. Upload to `/v1/files` → use `urls.get` as `input.audio`\n\n**Do not** post-mux narration over silent `p-video` clips unless re-render is impossible — truncated VO is a common failure mode.\n\n## Video phase (one beat)\n\nWhen start URL, end URL, and audio URL exist:\n\n- **`POST /v1/predictions`** with `Model: p-video` — one async job\n- Poll `get_url` until done\n\n## Multi-scene extension (narrated-multi-scene only)\n\nThe sections below apply when the user explicitly requested a **multi-scene film**. Single-clip skills must stop and hand off to [narrated-multi-scene](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/narrated-multi-scene/skills/narrated-multi-scene/SKILL.md) instead of executing them.\n\n**Explainer interaction (preferred):** alternate **narrator** triple beats with **character** `p-video-avatar` dialogue — see [interactive-explainer-scenes.md](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/interactive-explainer-scenes.md) and [interactive-explainer](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/interactive-explainer/skills/interactive-explainer/SKILL.md).\n\n**Explainer motion & format:** dynamic `OPEN:` / `MID:` / `CLOSE:` `video_prompt` per scene; default **`720p`** + **`24` fps** — see [interactive-explainer-motion.md](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/interactive-explainer-motion.md).\n\n**Visual style for explainers:** keep a single `style_bible` on every `p-image` / `p-image-edit` / `p-video` prompt.\n\n### Parallel stills / video across scenes\n\nRun start stills **in parallel** from hero; then end stills **in parallel** from each start still. After **all** URLs exist for every scene row, `POST /v1/predictions` in a **parallel** batch. Patterns: [parallel-execution.md](https://github.com/PrunaAI/pruna-skills/tree/main/policies/parallel-execution.md).\n\n### Frame chain\n\n**Chain only when motion continues** — same location, same moment, no time jump. Use a **composed start still** (hard cut) for new story beats, emotional pauses, or location changes.\n\n| Situation | `chain_from_previous` | Join style |\n|-----------|----------------------|------------|\n| Continuous action (toss → arc in air) | `true` | Short crossfade (~0.15s) after extract |\n| New beat / pause (loss → realization) | `false` | Hard cut — composed OPENING still |\n| First scene | `false` | — |\n\nPer-scene flag in plan (overrides legacy global `frame_chain`):\n\n```json\n{ \"id\": \"03_realization\", \"chain_from_previous\": false, \"edit_prompt\": \"OPENING SHOT: …\" }\n```\n\n| `frame_chain_mode` | Next scene `image` when chained | Render order |\n|--------------------|---------------------------------|--------------|\n| **`extract_last_frame`** | ffmpeg last frame from prior clip | **Sequential** when any scene chains |\n| **`planned_stills`** (legacy) | prior scene end still | Parallel |\n\n**Why extract?** Planned end stills often differ from the model's actual last frame → visible jump at cuts.\n\n```text\nScene 1: composed start,  last=end_1,  audio=vo_1   chain→2\nScene 2: extract(clip_1),  last=end_2,  audio=vo_2   hard cut→3\nScene 3: composed start,  last=end_3,  audio=vo_3   chain→4\n```\n\n### Scene + narration flow\n\nEach scene row should read as one complete beat:\n\n1. **OPEN** — `edit_prompt` / first frame matches the **opening words** of narration\n2. **MID** — `video_prompt` motion develops the line\n3. **CLOSE** — `last_frame_edit_prompt` holds a **clear ending pose** before the cut\n\nWrite narration to describe what is on screen at open → close. Avoid lines that reference action that hasn't happened yet or already finished.\n\nUse [`concat_clips.py`](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/_shared/scripts/concat_clips.py) with per-join **`crossfades`** — chain joins get ~0.15s fade; hard cuts get 0.\n\n### Assembly\n\n1. **Concat** clips in scene order with optional crossfade (narration already embedded per clip)\n2. **Optional bed** — [stable-audio-2.5](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/stable-audio-2.5/skills/stable-audio-2.5/SKILL.md) mixed **under** narration via [`launch_background_music.py`](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/_shared/scripts/launch_background_music.py) (~0.08–0.15 volume)\n\n### Plan JSON shape\n\n```json\n{\n  \"frame_chain_mode\": \"extract_last_frame\",\n  \"assembly\": {\n    \"chain_crossfade_seconds\": 0.15,\n    \"hard_cut_crossfade_seconds\": 0\n  },\n  \"narration\": {\n    \"enabled\": true,\n    \"voice\": \"Sulafat\",\n    \"mode\": \"p_video_audio\",\n    \"scene_lines\": { \"01_beat\": \"[warmly] …\" }\n  },\n  \"scenes\": [\n    {\n      \"id\": \"01_beat\",\n      \"chain_from_previous\": false,\n      \"edit_prompt\": \"OPENING SHOT: start still from hero…\",\n      \"last_frame_edit_prompt\": \"CLOSING SHOT: end still from start…\",\n      \"video_prompt\": \"OPEN: hold. MID: motion. CLOSE: settle on end pose.\"\n    },\n    {\n      \"id\": \"02_beat\",\n      \"chain_from_previous\": true,\n      \"edit_prompt\": \"…\",\n      \"last_frame_edit_prompt\": \"…\",\n      \"video_prompt\": \"…\"\n    }\n  ]\n}\n```\n\nUpgrade a **pair** to a **triple** by adding TTS → upload → `input.audio` and omitting `duration`. Visual-only transitions: [scene-anchor-pair.md](https://github.com/PrunaAI/pruna-skills/tree/main/workflows/scene-anchor-pair/SKILL.md).\n\n## Variants on other models\n\n| Model | Triple analogue |\n|-------|-----------------|\n| **`p-video-avatar`** | `image` (portrait) + optional `last_frame_image` + **`audio`** (uploaded TTS) *or* native `voice_script` |\n| **`p-video` (music video B-roll)** | `image` + **`audio`** (song slice) — `last_frame_image` optional per beat |\n| **`p-video-animate`** | `image` + **`video`** (motion template) — different axis; not narration triple — use the [p-video-animate](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/p-video-animate/skills/p-video-animate/SKILL.md) skill |\n\n## Workflows that implement this\n\n- [image-to-video](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/image-to-video/skills/image-to-video/SKILL.md) — **one beat** (this skill’s default)\n- [narrated-multi-scene](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/narrated-multi-scene/skills/narrated-multi-scene/SKILL.md) — primary narrated multi-scene workflow\n- [visual-transition-reel](https://github.com/PrunaAI/pruna-skills/tree/main/plugins/visual-transition-reel/skills/visual-transition-reel/SKILL.md) — visual pair (no VO)\n- [pruna-generative-pipeline](https://github.com/PrunaAI/pruna-skills/tree/main/docs/WORKFLOW-RECIPES.m\n\nArchive v1.0.2: 11 files, 28913 bytes\n\nFiles: README-INSTALL.md (333b), references/api-credentials.md (3138b), references/audio-post-production.md (9236b), references/generation-diversity.md (26090b), references/random-seed-ritual.md (4088b), references/replicate-api.md (2101b), references/scene-anchor-triple.md (10618b), skill-card.md (2966b), skill.manifest.json (250b), SKILL.md (8570b), _meta.json (139b)","readmeExcerpt":"Skill: gemini-3.1-flash-tts Owner: pruna-ai Summary: Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. Tags: ai:1.0.14, generative:1.0.14, latest:1.0.14, pruna:1.0.14 Version history: v1.0.14 | 2026-09-29T15:35:50.539Z | auto - Bumped skill version to 1.0.14. - Updated version metadata in SKILL.md. - Removed the redundant skill-card.md f","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"export REPLICATE_API_TOKEN=r8_..."},{"language":"bash","snippet":"curl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{"},{"language":"bash","snippet":"curl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"text\": \"[warmly] The plush went flying. [short pause] And then it was gone.\",\n      \"voice\": \"Sulafat\",\n      \"prompt\": \"Warm storybook narrator, gentle pace, empathetic, no announcer voice.\",\n      \"language_code\": \"en-US\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/google/gemini-3.1-flash-tts/predictions\""},{"language":"bash","snippet":"export REPLICATE_API_TOKEN=r8_..."},{"language":"bash","snippet":"curl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{"},{"language":"bash","snippet":"curl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"text\": \"[warmly] The plush went flying. [short pause] And then it was gone.\",\n      \"voice\": \"Sulafat\",\n      \"prompt\": \"Warm storybook narrator, gentle pace, empathetic, no announcer voice.\",\n      \"language_code\": \"en-US\"\n    }\n  }' \\\n  \"https://api.replicate.com/v1/models/google/gemini-3.1-flash-tts/predictions\""}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: gemini-3.1-flash-tts\ndescription: Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video.\nlicense: MIT\nmetadata:\n  version: \"1.0.14\"\n  package: pruna-skills\n  provider: replicate\n  replicate_model: google/gemini-3.1-flash-tts\n---\n\n## Prerequisites\n\nInstall and load these skills before generating (skip if already in context via `@pruna`):\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFollow each skill's **Before generating** / craft sections — do not restate guide content here.\n\n## Agent habit\n\nIn the **first reply**, name `` `gemini-3.1-flash-tts` `` in backticks, confirm `REPLICATE_API_TOKEN` (or stop with signup links from `pruna-api`), then ask for required inputs. Open intake → **`generation-diversity`** clarification intake (locale, voice, script) before the first `POST`. Redirect when **When NOT to use** fits better.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `p-video-avatar` | Use when someone wants a person on camera speaking a script — lip-synced host, spokesperson, or narrated avatar from a portrait photo. | `npx skills add PrunaAI/pruna-skills@p-video-avatar -y` |\n| `music-2.5` | Use when someone wants an original AI song with vocals — sung lyrics, a style prompt track, or source audio for a music video. | `npx skills add PrunaAI/pruna-skills@music-2.5 -y` |\n| `stable-audio-2.5` | Use when someone wants light instrumental background music — an ambient bed under dialogue or underscore for reels and explainers. | `npx skills add PrunaAI/pruna-skills@stable-audio-2.5 -y` |\n\n## Environment\n\n```bash\nexport REPLICATE_API_TOKEN=r8_...\n```\n\nRequires **`ffmpeg`** / **`ffprobe`** when trimming, concatenating scene VO, or mixing with a bed.\n\n## HTTP (curl)\n\n```bash\ncurl -s -X POST \\\n  -H \"Authorization: Bearer ${REPLICATE_API_TOKEN}\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\n    \"input\": {\n      \"text\": \"[warmly] The plush went flying. [short pause] And then"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"gemini-3-1-flash-tts\",\n  \"version\": \"1.0.14\",\n  \"publishedAt\": 1790696150539\n}"},{"path":"skill-card.md","content":"## Description:\n\nGuides agents in generating spoken narration and voiceovers from scripts using Gemini 3.1 Flash TTS through Replicate.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[pruna-ai](https://clawhub.ai/user/pruna-ai)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nCreators and developers use this skill to prepare scripts and generate spoken narration or voiceovers for explainers, documentaries, and videos.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Narration text and style prompts are sent to Replicate with an API token.\n\nMitigation: Only submit content you are comfortable sharing with Replicate and keep the token private.\n\nRisk: Suggested skill installation commands may fetch additional third-party code.\n\nMitigation: Review or pin the suggested npx skill installs before running them, especially the full-suite install.\n\n## Reference(s):\n\n- [ClawHub skill release](https://clawhub.ai/pruna-ai/skills/gemini-3-1-flash-tts)\n- [Gemini 3.1 Flash TTS model documentation on Replicate](https://replicate.com/google/gemini-3.1-flash-tts/readme)\n\n## Skill Output:\n\n**Output Type(s):** [Guidance, Shell commands, Audio]\n\n**Output Format:** [Text guidance and commands; generated audio URL or downloaded audio file]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Requires a Replicate API token; narration text and style prompt are sent to Replicate.]\n\n## Skill Version(s):\n\n1.0.14 (source: server release and skill frontmatter)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."},{"path":"skill.manifest.json","content":"{\n  \"references\": []\n}"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. Skill: gemini-3.1-flash-tts Owner: pruna-ai Summary: Use when someone needs spoken narration or voiceover — explainer tracks, documentary lines, or voice to pair with generated video. Tags: ai:1.0.14, generative:1.0.14, latest:1.0.14, pruna:1.0.14 Version history: v1.0.14 | 2026-09-29T15:35:50.539Z | auto - Bumped skill version to 1.0.14. - Updated version metadata in SKILL.md. - Removed the redundant skill-card.md f","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1277,"uniquenessScore":47,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-10T12:32:12.994Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-10T12:32:12.994Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T14:50:28.999Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}