{"id":"12aa762c-5e99-4f79-9c5d-d7c2cc3c19ec","entityType":"agent","slug":"clawhub-pruna-ai-video-prompting","name":"video-prompting","canonicalUrl":"https://www.xpersona.co/agent/clawhub-pruna-ai-video-prompting","canonicalPath":"/agent/clawhub-pruna-ai-video-prompting","generatedAt":"2026-10-11T20:58:44.205Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-11T17:42:52.136Z","emptyReason":null},"description":"Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. Skill: video-prompting Owner: pruna-ai Summary: Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. Tags: ai:1.0.14, generative:1.0.14, latest:1.0.14, pruna:1.0.14 Version history: v1.0.14 | 2026-09-29T15:31:08.407Z | auto - Updated to version 1.0.14 - Internal documentation and quality checklist references updated - Sample/re","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s173djpkkm1x4yfzg522h5vfp989mrn5:video-prompting","sourceUrl":"https://clawhub.ai/pruna-ai/video-prompting","homepage":"https://clawhub.ai/pruna-ai/skills/video-prompting","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/pruna-ai/video-prompting","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/pruna-ai/skills/video-prompting","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":60,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. Skill: video-pro"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T17:42:52.136Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T17:42:52.136Z","emptyReason":null},"stars":null,"forks":null,"downloads":1016,"packageName":null,"latestVersion":"1.0.14","tractionLabel":"1K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T17:42:52.120Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T17:42:52.136Z","lastCrawledAt":"2026-10-11T17:42:52.120Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T17:42:52.120Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.14","createdAt":"2026-09-29T15:31:08.407Z","changelog":"- Updated to version 1.0.14 - Internal documentation and quality checklist references updated - Sample/reference files revised for clarity and alignment - Removed outdated skill card file","fileCount":24,"zipByteSize":42156},{"version":"1.0.13","createdAt":"2026-09-17T13:52:27.830Z","changelog":"- Added support and documentation for the new `p-video-2-pro` cinematic video generation workflow. - Introduced separate prompting and quality checklists for `p-video-2-pro`. - Updated usage guidance and tool selection tables to include distinctions between `p-video-2-pro`, `p-video-2`, and `p-video`. - Improved descriptions and decision criteria for when to use each video generation model. - Removed obsolete references (e.g., skill-card.md) and clarified the guide flow for model selection and dramaturgy.","fileCount":24,"zipByteSize":42331},{"version":"1.0.12","createdAt":"2026-09-10T13:50:39.926Z","changelog":"- Adds support and documentation for the new p-video-2 tool, with guidance for high-quality video prompting. - Updates core workflow and references: distinguishes between p-video-2 (best quality) and p-video (simpler/draft clips). - Adds dedicated reference docs: p-video-2-prompting.md and p-video-2-quality-checklist.md. - Updates \"When to use\", \"Works with\", and tool selection tables for p-video-2. - Removes legacy skill-card.md file. - Clarifies tool-specific best practices and quality path in the main guide.","fileCount":22,"zipByteSize":37920},{"version":"1.0.11","createdAt":"2026-09-03T14:06:20.774Z","changelog":"- Added references for p-video-edit prompting and quality checklist. - Updated documentation to include support and guidance for the new p-video-edit tool. - Expanded \"Works with\" and installation tables to reference p-video-edit. - Removed obsolete skill-card.md file. - Version updated to 1.0.11.","fileCount":20,"zipByteSize":34666},{"version":"1.0.10","createdAt":"2026-08-28T07:52:37.130Z","changelog":"version 1.0.10 - Bumped version to 1.0.10 in metadata. - Removed the redundant skill-card.md file.","fileCount":18,"zipByteSize":31458},{"version":"1.0.9","createdAt":"2026-08-04T06:16:10.900Z","changelog":"video-prompting 1.0.9 - Updated version metadata to 1.0.9 in SKILL.md. - Removed redundant file: skill-card.md.","fileCount":18,"zipByteSize":31277},{"version":"1.0.8","createdAt":"2026-07-28T17:18:20.605Z","changelog":"- Version bumped to 1.0.8. - Clarified guide habit: intake for aspect, resolution, duration, or embed-vs-post audio now directs to **`generation-diversity`** clarification. - Minor copy updates in SKILL.md for first-reply and intake process. - Removed the redundant skill-card.md file.","fileCount":18,"zipByteSize":31210},{"version":"1.0.7","createdAt":"2026-07-23T12:32:52.798Z","changelog":"video-prompting 1.0.7 - Expanded documentation clarifying use cases, compatible tools, and when to use related skills. - Added detailed guidance on dramaturgy, camera craft, and physics-safe motion for generative video prompts. - Included specific instructions for frame anchor use, audio integration, multi-clip chaining, and model selection. - Provided direct links to additional references and quality checklists for improved prompt design. - Improved installation commands and comparison tables for easier integration with Pruna and other video models.","fileCount":18,"zipByteSize":31298}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s173djpkkm1x4yfzg522h5vfp989mrn5:video-prompting","setupComplexity":"low","setupSteps":["Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-video-prompting/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-video-prompting/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-video-prompting/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-video-prompting/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-video-prompting/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-video-prompting/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T20:58:44.202Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-video-prompting/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-video-prompting/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-video-prompting/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-pruna-ai-video-prompting/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-11T17:42:52.136Z","emptyReason":null},"readme":"Skill: video-prompting\n\nOwner: pruna-ai\n\nSummary: Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining.\n\nTags: ai:1.0.14, generative:1.0.14, latest:1.0.14, pruna:1.0.14\n\nVersion history:\n\nv1.0.14 | 2026-09-29T15:31:08.407Z | auto\n\n- Updated to version 1.0.14\n- Internal documentation and quality checklist references updated\n- Sample/reference files revised for clarity and alignment\n- Removed outdated skill card file\n\nv1.0.13 | 2026-09-17T13:52:27.830Z | auto\n\n- Added support and documentation for the new `p-video-2-pro` cinematic video generation workflow.\n- Introduced separate prompting and quality checklists for `p-video-2-pro`.\n- Updated usage guidance and tool selection tables to include distinctions between `p-video-2-pro`, `p-video-2`, and `p-video`.\n- Improved descriptions and decision criteria for when to use each video generation model.\n- Removed obsolete references (e.g., skill-card.md) and clarified the guide flow for model selection and dramaturgy.\n\nv1.0.12 | 2026-09-10T13:50:39.926Z | auto\n\n- Adds support and documentation for the new p-video-2 tool, with guidance for high-quality video prompting.\n- Updates core workflow and references: distinguishes between p-video-2 (best quality) and p-video (simpler/draft clips).\n- Adds dedicated reference docs: p-video-2-prompting.md and p-video-2-quality-checklist.md.\n- Updates \"When to use\", \"Works with\", and tool selection tables for p-video-2.\n- Removes legacy skill-card.md file.\n- Clarifies tool-specific best practices and quality path in the main guide.\n\nv1.0.11 | 2026-09-03T14:06:20.774Z | auto\n\n- Added references for p-video-edit prompting and quality checklist.\n- Updated documentation to include support and guidance for the new p-video-edit tool.\n- Expanded \"Works with\" and installation tables to reference p-video-edit.\n- Removed obsolete skill-card.md file.\n- Version updated to 1.0.11.\n\nv1.0.10 | 2026-08-28T07:52:37.130Z | auto\n\nversion 1.0.10\n\n- Bumped version to 1.0.10 in metadata.\n- Removed the redundant skill-card.md file.\n\nv1.0.9 | 2026-08-04T06:16:10.900Z | auto\n\nvideo-prompting 1.0.9\n\n- Updated version metadata to 1.0.9 in SKILL.md.\n- Removed redundant file: skill-card.md.\n\nv1.0.8 | 2026-07-28T17:18:20.605Z | auto\n\n- Version bumped to 1.0.8.\n- Clarified guide habit: intake for aspect, resolution, duration, or embed-vs-post audio now directs to **`generation-diversity`** clarification.\n- Minor copy updates in SKILL.md for first-reply and intake process.\n- Removed the redundant skill-card.md file.\n\nv1.0.7 | 2026-07-23T12:32:52.798Z | auto\n\nvideo-prompting 1.0.7\n\n- Expanded documentation clarifying use cases, compatible tools, and when to use related skills.\n- Added detailed guidance on dramaturgy, camera craft, and physics-safe motion for generative video prompts.\n- Included specific instructions for frame anchor use, audio integration, multi-clip chaining, and model selection.\n- Provided direct links to additional references and quality checklists for improved prompt design.\n- Improved installation commands and comparison tables for easier integration with Pruna and other video models.\n\nArchive index:\n\nArchive v1.0.14: 24 files, 42156 bytes\n\nFiles: references/audio-in-video-prompting.md (3342b), references/camera-lighting-vocabulary.md (2812b), references/clip-chaining.md (2668b), references/p-video-2-pro-prompting.md (5188b), references/p-video-2-pro-quality-checklist.md (2539b), references/p-video-2-prompting.md (3352b), references/p-video-2-quality-checklist.md (2231b), references/p-video-animate-prompting.md (2230b), references/p-video-animate-quality-checklist.md (1940b), references/p-video-avatar-prompting.md (2733b), references/p-video-avatar-quality-checklist.md (2586b), references/p-video-edit-prompting.md (3745b), references/p-video-edit-quality-checklist.md (2068b), references/p-video-quality-checklist.md (2507b), references/p-video-replace-prompting.md (3103b), references/p-video-replace-quality-checklist.md (3852b), references/physics-safe-motion.md (2859b), references/prompt-dramaturgy.md (3959b), references/scene-anchor-pair.md (11179b), references/scene-anchor-triple.md (7523b), skill-card.md (2147b), skill.manifest.json (729b), SKILL.md (8368b), _meta.json (135b)\n\nFile v1.0.14:SKILL.md\n\n---\nname: video-prompting\ndescription: Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining.\nlicense: MIT\nmetadata:\n  version: \"1.0.14\"\n  package: pruna-skills\n---\n\n# Video prompting\n\nVendor-neutral craft for **short video / motion** generation. Works with Pruna `p-video-2-pro` / `p-video-2` / `p-video` family, Runway, Kling, Luma, Veo, and similar APIs.\n\n## Install\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n\n## When to use\n\n- Text-to-video or image-to-video prompts\n- Start/end frame (anchor pair) or narrated beat (anchor triple) specs\n- Camera and lighting vocabulary in motion lines\n- Physics-safe subject motion\n- Multi-clip continuity / clip chaining\n- Talking-head, motion-transfer, slot-replace, or instruction-based video-edit prompts\n\n## Works with\n\nPruna `p-video-2-pro` / `p-video-2` / `p-video` / `p-video-avatar` / `p-video-animate` / `p-video-replace` / `p-video-edit`, Runway Gen-3, Kling, Luma Dream Machine, Veo, and other video models. Cinematic generation (generated audio, first/last frame): `p-video-2-pro`. 1080p / imported audio / draft: `p-video-2`. Cheaper, faster drafts: `p-video`.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `image-prompting` | Use when crafting still-image prompts for any generative model — composition, identity sheets, edits, try-on, and photoreal personas. | `npx skills add PrunaAI/pruna-skills@image-prompting -y` |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `music-video` | Use when someone wants a full music video — original song or vocals, performance clips, B-roll, and lyric-synced edits. | `npx skills add PrunaAI/pruna-skills@music-video -y` |\n| `narrated-multi-scene` | Use when someone wants a multi-part story with voiceover — episodic B-roll, chaptered promo, or several linked video scenes without on-camera dialogue. | `npx skills add PrunaAI/pruna-skills@narrated-multi-scene -y` |\n| `p-video-2-pro` | Use when someone wants a cinematic clip from text or start/end frames — product ads, documentary shots, or dialogue with generated audio. Not for 1080p, imported audio tracks, or talking-head-only hosts. | `npx skills add PrunaAI/pruna-skills@p-video-2-pro -y` |\n| `p-video-2` | Use when someone wants a polished short clip from text, images, or imported audio — 1080p B-roll, start/end frame animation, or a motion shot with a mixed track. Not for cinematic generated-audio clips or talking-head-only hosts. | `npx skills add PrunaAI/pruna-skills@p-video-2 -y` |\n| `p-video` | Use when someone wants a simple short clip from text or images — quick B-roll, drafts, or start/end frame animation. Not when the brief needs cinematic generation, highest quality, tight lip-sync, or imported audio at 1080p. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\n## Guide habit\n\nIn the **first reply**, name `` `video-prompting` `` in backticks. When aspect, resolution, duration, or embed-vs-post audio are open, open intake → **`generation-diversity`** clarification intake. For `p-video-2-pro` / `p-video-2` / `p-video` motion lines, cite OPEN/MID/CLOSE dramaturgy and **Worked example — product B-roll** in [prompt-dramaturgy.md](./references/prompt-dramaturgy.md). Cinematic generation: `p-video-2-pro` (`p-video-2-pro-prompting`). 1080p / imported audio / draft: `p-video-2` (`p-video-2-prompting`). Cheaper drafts: `p-video`. Audio-led clips: **≤ ~19s** TTS before embed on `p-video-2` — see [audio-in-video-prompting.md](./references/audio-in-video-prompting.md).\n\n## Before generating\n\n1. Follow `generation-diversity` first.\n2. Read in order:\n   - [prompt-dramaturgy.md](./references/prompt-dramaturgy.md) — Details Law, OPEN/MID/CLOSE\n   - [camera-lighting-vocabulary.md](./references/camera-lighting-vocabulary.md)\n   - [physics-safe-motion.md](./references/physics-safe-motion.md)\n   - [audio-in-video-prompting.md](./references/audio-in-video-prompting.md) when sound matters\n   - [clip-chaining.md](./references/clip-chaining.md) for multi-clip continuity\n   - [scene-anchor-pair.md](./references/scene-anchor-pair.md) / [scene-anchor-triple.md](./references/scene-anchor-triple.md) for frame (+ audio) payloads\n3. Tool-specific craft when needed:\n   - [p-video-2-pro-prompting.md](./references/p-video-2-pro-prompting.md)\n   - [p-video-2-prompting.md](./references/p-video-2-prompting.md)\n   - [p-video-avatar-prompting.md](./references/p-video-avatar-prompting.md)\n   - [p-video-animate-prompting.md](./references/p-video-animate-prompting.md)\n   - [p-video-replace-prompting.md](./references/p-video-replace-prompting.md)\n   - [p-video-edit-prompting.md](./references/p-video-edit-prompting.md)\n4. Validate with the matching `*-quality-checklist.md` in `./references/`.\n\nProduct B-roll and OPEN/MID/CLOSE samples: **Worked example — product B-roll** in [prompt-dramaturgy.md](./references/prompt-dramaturgy.md).\n\n## Pruna tools\n\nMatching install for every model named above. Pick what you need:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `p-video-2-pro` | Use when someone wants a cinematic clip from text or start/end frames — product ads, documentary shots, or dialogue with generated audio. Not for 1080p, imported audio tracks, or talking-head-only hosts. | `npx skills add PrunaAI/pruna-skills@p-video-2-pro -y` |\n| `p-video-2` | Use when someone wants a polished short clip from text, images, or imported audio — 1080p B-roll, start/end frame animation, or a motion shot with a mixed track. Not for cinematic generated-audio clips or talking-head-only hosts. | `npx skills add PrunaAI/pruna-skills@p-video-2 -y` |\n| `p-video` | Use when someone wants a simple short clip from text or images — quick B-roll, drafts, or start/end frame animation. Not when the brief needs cinematic generation, highest quality, tight lip-sync, or imported audio at 1080p. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n| `p-video-avatar` | Use when someone wants a person on camera speaking a script — lip-synced host, spokesperson, or narrated avatar from a portrait photo. | `npx skills add PrunaAI/pruna-skills@p-video-avatar -y` |\n| `p-video-animate` | Use when someone wants a photo to move like another video — motion transfer, dance remixes, or performance variations from a template clip. | `npx skills add PrunaAI/pruna-skills@p-video-animate -y` |\n| `p-video-replace` | Use when someone wants to swap a person, outfit, or product inside existing footage while keeping the camera move and audio. | `npx skills add PrunaAI/pruna-skills@p-video-replace -y` |\n| `p-video-edit` | Use when someone wants to edit an existing video with a text instruction — recolor, restyle, remove or add objects, change environment or lighting, update on-screen text, or apply optional reference-guided product and accessory edits. Not for a new clip from scratch or ffmpeg assembly. | `npx skills add PrunaAI/pruna-skills@p-video-edit -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFile v1.0.14:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"video-prompting\",\n  \"version\": \"1.0.14\",\n  \"publishedAt\": 1790695868407\n}\n\nFile v1.0.14:references/audio-in-video-prompting.md\n\n# Audio-in-video prompting (`p-video`)\n\nHow to **write prompts** when sound matters on `p-video`. Layering / tool picker: `audio-prompting`. Talking heads: [p-video-avatar-prompting.md](./p-video-avatar-prompting.md).\n\n## Three modes\n\n| Mode | API | Prompt job |\n|------|-----|------------|\n| **A — Native SFX / dialogue** | `prompt` + optional `save_audio`; use `duration` | Name **diegetic** sounds the picture should emit |\n| **B — Uploaded audio (preferred for VO/music)** | `audio` URL; **omit `duration`**; `save_audio: true` | Motion matches **mood/beats** of the track — do not paste VO text into `prompt` |\n| **C — Post bed** | Stable Audio mixed under VO in ffmpeg | Bed prompt is separate (`stable-audio-2.5` + `audio-prompting`); video prompt ignores the bed |\n\nNever generate silent `p-video` and post-mux narration unless re-render is impossible (truncation risk). Probe TTS ≤ ~19s before Mode B.\n\n## Mode A — native SFX / dialogue\n\n**When:** `save_audio: true`, no uploaded `audio`, user wants diegetic SFX and/or spoken lines in the clip.\n\n### Diegetic SFX\n\nBe concrete; skip `cinematic soundscape`:\n\n| Bad | Good |\n|-----|------|\n| `epic soundtrack vibe` | `rain ticks on the awning, distant train horn once` |\n| `dramatic music` | `crowd murmur swells, single glass clink` |\n\nKeep cues short — the model invents audio from visual+prompt context when `save_audio` is on.\n\n### Native dialogue\n\nPut **exact spoken words in double quotes** inside the motion `prompt` (Mode A only). Placeholders in square brackets below are docs notation, not literal syntax — do not confuse with Gemini TTS `[tags]`.\n\n**Template** (inside OPEN/MID/CLOSE):\n\n```text\nMID: [same subject] says \"[LINE]\" — mouth open, [one gesture toward target]; [optional diegetic SFX cue]\n```\n\nRules:\n\n- Exact spoken words in **double quotes** — never paraphrase the line\n- Name **who** speaks — repeat the subject label for continuity (`same [role]`)\n- Pair every line with **mouth state** + **one gesture** (point, turn to camera, hand on prop)\n- Keep lines **short** (1–2 per beat)\n- Diegetic SFX can sit beside dialogue — stay concrete (table above)\n- **Mode B:** when `input.audio` is set, do **not** put the VO transcript in `prompt` — motion matches mood only\n\n## Mode B — motion matches uploaded audio\n\n```text\nOPEN: hold wide on dog in tall grass, warm afternoon light.\nMID: gentle push-in as he searches; tail motion matches narrator energy; grass sways.\nCLOSE: settle on end pose — curious head tilt.\n```\n\nRules:\n\n- Describe **picture motion**, not the spoken words.  \n- Match energy: tense VO → tighter push-in; calm story → slow drift.  \n- Optional: `motion matches narrator mood` once — not a transcript.  \n- Triple anchors: [scene-anchor-triple.md](./scene-anchor-triple.md).\n\n## Avatar path (not Mode B fields)\n\n`p-video-avatar` uses `voice_script` + `voice_prompt` + `video_prompt` — not `p-video` `input.audio` for the spoken line. See avatar prompting ref. Do not paste script lines into `voice_prompt`.\n\n## Pre-send\n\n- [ ] Mode A / B / C chosen  \n- [ ] Mode B: `duration` omitted; TTS length probed  \n- [ ] No VO transcript inside motion `prompt` (Mode B)  \n- [ ] Diegetic cues concrete (Mode A) or mood-aligned (Mode B)\n- [ ] Mode A dialogue (if any): quoted line + named speaker + mouth + one gesture\n\nFile v1.0.14:references/camera-lighting-vocabulary.md\n\n# Camera and lighting vocabulary\n\nShared lexicon for `camera_tag` / `lighting_tag` (stills) and motion lines in `p-video` / `p-video-avatar` prompts. Diversity axes: generation-diversity.md#visual-variety (`generation-diversity`). Dramaturgy: [prompt-dramaturgy.md](./prompt-dramaturgy.md).\n\n**Sources:** patterns adapted from [smixs/visual-skills](https://github.com/smixs/visual-skills) and [inference-sh/skills](https://github.com/inference-sh/skills) (MIT); rewritten for Pruna.\n\n## Framing ladder\n\n| Term | Use when |\n|------|----------|\n| ECU (extreme close-up) | Eyes, hands, product detail |\n| CU (close-up) | Face, emotion |\n| MCU (medium close-up) | Talking head default |\n| MS (medium shot) | Waist-up action |\n| MLS / FS | Full body travel |\n| WS / EWS | Environment as character |\n\nLog as `camera_tag`, e.g. `medium close-up, slight low angle`.\n\n## Lens roles (optional but sharp)\n\n| Lens | Feel |\n|------|------|\n| 24mm | Wide, immersive, exaggerated space |\n| 35mm | Documentary natural |\n| 50mm | Intimate human perspective |\n| 85mm | Portrait, compressed background |\n| Macro | Texture, product detail |\n\nExample: `shot on 50mm, eye-level`.\n\n## Camera moves (pick one for MID)\n\n| Move | Prompt cue |\n|------|------------|\n| Dolly / push-in | `slow dolly in`, `gentle push-in` |\n| Dolly out | `slow pull back revealing the room` |\n| Pan | `gentle pan left across the alley` |\n| Tilt | `tilt up from hands to face` |\n| Track / truck | `shoulder-height tracking shot beside the subject` |\n| Crane | `slow crane down past neon signs` |\n| Static + atmosphere | `locked camera, steam rises, light shifts` |\n| Handheld | `subtle handheld drift` (use sparingly) |\n\nAvoid whip pans and stacked contradictory moves in one short clip.\n\n## Motivated lighting\n\nPrefer **named sources** over “beautiful lighting”:\n\n| Source | Example cue |\n|--------|-------------|\n| Window / dawn | `dawn light spreads across the desk` |\n| Practical | `warm lamp spill, cool window fill` |\n| Neon / gel | `magenta-cyan neon rim, wet reflections` |\n| Overhead institutional | `cold fluorescent flicker` |\n| Fire / candle | `candle flicker on faces` (avatar: often too transition-y — prefer steady) |\n| Overcast soft | `soft overcast skylight, low contrast` |\n\nLog as `lighting_tag`. Hex in stills when brand colors matter (`#0d3d2d rim`).\n\n## Palette cues (one look)\n\nPick one coherent palette phrase: `teal-magenta night`, `warm tungsten interior`, `bleached noon desert`, `desaturated documentary`.\n\nDo not stack competing genre looks in one prompt.\n\n## Avatar-friendly defaults\n\nTalking heads: **MCU**, one slow push-in or static, **steady light**, mouth visible. Variety across scenes = change angle/background still, not five camera moves mid-line. See [p-video-avatar-prompting.md](./p-video-avatar-prompting.md).\n\nFile v1.0.14:references/clip-chaining.md\n\n# Clip chaining (multi-scene video)\n\nWhen and how to continue motion across `p-video` clips. Plan JSON examples stay in [scene-anchor-pair.md](./scene-anchor-pair.md) and [scene-anchor-triple.md](./scene-anchor-triple.md); this page is the decision tree + prompt rules.\n\nWorkflows: `visual-transition-reel` · `narrated-multi-scene`.\n\n## Decision tree\n\n```text\nDoes motion continue in the same place/moment (no time jump)?\n  NO  → chain_from_previous: false — hard cut; compose a new OPENING still\n  YES → chain_from_previous: true\n        Prefer frame_chain_mode: extract_last_frame (sequential renders)\n        Only use planned_stills if you accept possible cut jumps\n```\n\n| Situation | `chain_from_previous` | Join |\n|-----------|----------------------|------|\n| Continuous action (run → leap) | `true` | Short crossfade ~0.12–0.15s after extract |\n| New beat / location / pause | `false` | Hard cut (0 crossfade) |\n| First scene | `false` | — |\n| Montage vignettes (no shared motion) | `false` + `parallel_vignettes` | Hard cuts; parallel renders OK |\n\n| `frame_chain_mode` | Next scene `image` | Render order |\n|--------------------|--------------------|--------------|\n| **`extract_last_frame`** | ffmpeg last frame of prior clip | **Sequential** when any scene chains |\n| **`parallel_vignettes`** | each scene’s own start still | **Parallel** |\n| **`planned_stills`** | prior scene end still URL | Parallel once stills exist — higher jump risk |\n\n**Why extract?** Planned end stills often differ from the model’s actual last frame → visible jump.\n\n## Prompt rules for chained beats\n\n1. **Same subject language** — repeat “same [character]” in OPEN/MID/CLOSE.  \n2. **No teleport** — ban `cut to`, `suddenly in`, `walls disappear`; use `gradually`, `walks through`, `ease into`.  \n3. **Match lighting era** — chained clips share `style_bible` and time-of-day.  \n4. **Exit / enter continuity** — if scene 1 CLOSE faces right, scene 2 OPEN should not hard-flip screen direction without a motivated turn.  \n5. **Hard-cut scenes** — treat as fresh OPENING; do not assume prior pose.\n\n## Assembly notes\n\n1. Concat in scene order (ffmpeg concat — see the workflow skill).  \n2. Per-join `crossfades`: chain ~0.12–0.15s; hard cuts 0.  \n3. Normalize audio (48 kHz stereo) when mixing formats.  \n4. Optional bed under native SFX — `audio-prompting`.\n\n## Intake checklist\n\n- [ ] Each scene: chain flag only if motion truly continues  \n- [ ] `frame_chain_mode` chosen  \n- [ ] Chained prompts pass Details Law ([prompt-dramaturgy.md](./prompt-dramaturgy.md))  \n- [ ] Physics tier OK ([physics-safe-motion.md](./physics-safe-motion.md))\n\nFile v1.0.14:references/p-video-2-pro-prompting.md\n\n# p-video-2-pro prompting\n\nPrompt craft unique to `p-video-2-pro` (cinematic generation lane next to `p-video-2`). Shared dramaturgy, camera, and physics: [prompt-dramaturgy.md](./prompt-dramaturgy.md), [camera-lighting-vocabulary.md](./camera-lighting-vocabulary.md), [physics-safe-motion.md](./physics-safe-motion.md). Visual frame payloads: [scene-anchor-pair.md](./scene-anchor-pair.md). QA: [p-video-2-pro-quality-checklist.md](./p-video-2-pro-quality-checklist.md). Uploaded audio triples stay on `p-video-2` — [scene-anchor-triple.md](./scene-anchor-triple.md).\n\n**Cinematic generation.** Use `p-video-2-pro` for text-to-video or first/last-frame clips with **generated audio**. Use `p-video-2` for **1080p**, **imported audio**, or **draft** previews. Use `p-video` for cheaper, faster drafts.\n\n## Strengths to write toward\n\n- **Multi-beat cinematic blocking** in one clip (product unfold, dialogue, chase, weather) — 8s is a typical locked-in length\n- **Heavier physics:** water, snow spray, fire, fabric, hair\n- **Full-body anatomy** that stays coherent through fast motion\n- **Two-shot dialogue** with lip-sync — write the spoken line into the prompt\n- **First + last frame** as a first-class control\n- **Three-level prompt upsampler** (`off` / `turbo` / `max`), independent of `mode`\n- `mode: cost` (same quality as `speed`, but cheaper and slower) vs `mode: speed` (default, faster) vs `mode: quality` on the same prompt\n- **Generated audio** already in the clip — write score, ambience, SFX, or dialogue in the prompt\n\n## Limits (do not fight them)\n\n- **No audio import** — muxed VO / music belongs on `p-video-2`\n- Max **768p** and **15s** — not 1080p / 20s / 48 fps\n- **No draft mode** — `mode: cost` is the same quality as `speed`, but cheaper and slower; `speed` is the faster default; `quality` is slower\n- Not an editor (`p-video-edit`) or talking-head avatar (`p-video-avatar`)\n- Native **4K** is not supported\n- **More than two speakers** — speaker separation degrades\n\n## Fast pass vs locked-in\n\n**Fast pass** — subject + action + scene. Enough for first looks. Iterate in `mode: cost` (price) or `mode: speed` (wall-clock).\n\n```text\nA matte stainless pour-over kettle sits on a seamless light-gray studio sweep. Thin steam rises from the spout.\n```\n\n**Locked-in** — add camera, lighting, style, and audio. Optional first-frame or last-frame still for repeatable runs.\n\n```text\nA narrative film scene, 1970s Roman trattoria at night, warm tungsten, cigarette haze, Super-8 grain. A man and a woman in period clothes sit at a small table with wine. She leans in and quietly says, \"Then we leave before sunrise.\" The camera slowly dollies around the table. Audio: Italian radio pop from a small speaker, plates, low room tone, her line clear and close.\n```\n\n## Mode recipes (rewrite for the brief — do not paste)\n\n**Text-to-video** — subject + action + camera + lighting + audio intent in one coherent line.\n\n```text\nA dramatic cinematic shot of an isolated lighthouse standing on a rocky cliff during an enormous Atlantic storm at dusk. The camera begins close to the lighthouse keeper standing outside near the railing as violent wind pulls at his coat. He turns toward the ocean just as a massive wave crashes against the cliff and sends water high above the lighthouse. The camera rapidly pulls back to reveal the scale of the storm. Photorealistic water physics, heavy rain, turbulent clouds, realistic human movement, and epic scale.\n```\n\n**Image-to-video (first frame)** — keep motion aligned with the reference frame.\n\n```text\nThe camera slowly pushes in. The person turns their head and smiles naturally. Soft studio lighting, shallow depth of field.\n```\n\n**First + last frame** — motion consistent with both stills; end exactly on the last-frame composition.\n\n```text\nStart on the still product hero. The camera holds, then a slow push-in as light moves across the metal. End exactly on the last-frame packshot, label readable, no extra props.\n```\n\n**Native speech (T2V)** — write the spoken line into the prompt. There is no `audio` field.\n\n```text\nA woman faces the camera and says \"I'll be there in five.\" She pauses, delivers the line, then rests. Natural mouth motion, soft window light, camera holds steady.\n```\n\n## Duration, mode, upsampler\n\n- Set `duration` (5–15s; default **5**). Partner demos often use **8s** for locked-in cinematic beats.\n- `mode: cost` when price is the priority (same quality as `speed`, but cheaper and slower); `mode: speed` (default) when generation time matters; `mode: quality` for the slower, higher-fidelity recipe.\n- `prompt_upsampler`: `off` when copy is already locked; `turbo` (default); `max` when the source prompt is short and the scene needs more described detail. Compare `off` vs `turbo` vs `max` on the **same `seed`** before scaling.\n- Defaults: `resolution: 768p`, `aspect_ratio: 16:9`, 24 fps, generated audio.\n\n## Iterate\n\nThere is no `draft` flag. Iterate in `mode: cost` (same quality as `speed`, but cheaper and slower) or `mode: speed` (faster); then final with `mode: quality` when fidelity matters. Do not send `audio`, `fps`, `save_audio`, or `prompt_upsampling`.\n\nFile v1.0.14:references/p-video-2-pro-quality-checklist.md\n\n# p-video-2-pro quality checklist\n\nAfter each `p-video-2-pro` output is saved, **open the clip and review it visually** (and listen — generated audio is expected) against this checklist (agent vision review — see `generation-diversity`).\n\nShared motion / frame-anchor items: also run [p-video-quality-checklist.md](./p-video-quality-checklist.md).\n\n## Applies to\n\nSee the canonical mapping in `generation-diversity`. Cinematic generation path for visual-only `image-to-video` and `visual-transition-reel`. Audio-led jobs stay on `p-video-2`.\n\n## Quality vs `p-video-2` / `p-video`\n\n- Multi-beat blocking holds (product unfold, dialogue, weather) without collapsing into a single pose.\n- Heavier physics stay readable (water, fire, fabric, hair) when the prompt asked for them.\n- Full-body anatomy stays coherent through fast motion.\n- Close-up / foreground objects stay readable (hands, product, face, packshot label).\n- Input-image identity holds when `image` was set; last-frame composition matches when `last_frame_image` was set.\n\n## Lip-sync and speakers\n\nWhen the prompt implies speech or singing (there is **no imported `audio`**):\n\n- Mouth follows the **written line** — pause / line / rest.\n- At most **two** speaking faces stay separable; reject crowded dialogue with 3+ talkers.\n- Talking-head-only jobs with no native scene audio should have used `p-video-avatar` instead.\n\n## Generated audio\n\n- The file is not unintentionally silent — output includes generated audio.\n- Score, ambience, SFX, or dialogue match what the prompt named.\n- Do not fail the job for missing an **uploaded** track — that brief belongs on `p-video-2`.\n- Do not treat weak SFX as a pass if the brief was sound-effect-led — warn and offer a bed in post (`audio-prompting`) or a `p-video-2` mux.\n\n## Camera and length\n\n- Runtime is in the **5–15s** window and matches the requested `duration` (default 5).\n- Output is **480p or 768p** as requested — not 720p / 1080p / 4K.\n- Output is **24 fps**.\n- When `mode: cost` was set, treat visual quality as equivalent to `mode: speed` of the same seed — do not fail for looking like speed.\n- When `mode: quality` was set, the clip should look tighter than a `mode: speed` or `mode: cost` preview of the same seed — if it does not, say so.\n\n## Scene anchors\n\nWhen using a first/last-frame pair, apply the pair sections in [p-video-quality-checklist.md](./p-video-quality-checklist.md), substituting `p-video-2-pro` for the prediction model. Do not expect a scene-anchor triple on this model.\n\nFile v1.0.14:references/p-video-2-prompting.md\n\n# p-video-2 prompting\n\nPrompt craft unique to `p-video-2` (quality-focused successor to `p-video`). Shared dramaturgy, camera, and physics: [prompt-dramaturgy.md](./prompt-dramaturgy.md), [camera-lighting-vocabulary.md](./camera-lighting-vocabulary.md), [physics-safe-motion.md](./physics-safe-motion.md). Frame payloads: [scene-anchor-pair.md](./scene-anchor-pair.md) / [scene-anchor-triple.md](./scene-anchor-triple.md). QA: [p-video-2-quality-checklist.md](./p-video-2-quality-checklist.md).\n\n**1080p / imported-audio path.** Use `p-video-2` when the brief needs 1080p, a mixed track, or draft previews. Use `p-video-2-pro` for cinematic generation with generated audio. Use `p-video` for simpler, quicker clips.\n\n## Strengths to write toward\n\n- Sharper subjects, backgrounds, and motion than `p-video` — especially **close-ups and foreground objects**\n- Stronger **lip-sync on native speech** — T2V with `save_audio: true` and **no imported `audio`**: the model pauses, delivers the line, then rests, and the mouth follows. **Lead dialogue / lip-sync jobs here**\n- **Native audio** in the output (`save_audio` defaults true). Imported `audio` still works and keeps a **cleaner face / identity**; viseme lock vs `p-video` is mixed — do not lead lip-sync demos with a muxed wav\n- Stronger **identity / input-image consistency**\n- One endpoint: T2V + I2V + audio-conditioned\n\n## Limits (do not fight them)\n\n- Not designed for **extreme cinematic camera** (crash zooms, whip pans, chaotic handheld)\n- Complex **multi-scene storytelling** in one clip is weaker — split beats\n- Native **4K** is not supported\n- **More than two speakers** — speaker separation degrades\n- **SFX-led** clips (the ask is sound effects, not picture + optional bed) are currently limited\n\n## Mode recipes (rewrite for the brief — do not paste)\n\n**Text-to-video** — subject + motion + camera + lighting + audio intent in one coherent line.\n\n```text\nA sports car drifting through a neon-lit city at night, cinematic aerial shot, wet asphalt reflections, engine roar and tire screech.\n```\n\n**Image-to-video** — keep motion **subtle** and aligned with the reference frame. Prefer a **stable** camera.\n\n```text\nThe camera slowly pushes in. The person turns their head and smiles naturally. Soft studio lighting, shallow depth of field.\n```\n\n**Native speech (T2V, no `audio`)** — write the spoken line into the prompt. This is the lip-sync path.\n\n```text\nA woman faces the camera and says \"I'll be there in five.\" She pauses, delivers the line, then rests. Natural mouth motion, soft window light, camera holds steady.\n```\n\n**Audio-conditioned** — name the performer and hold. One or two faces max. Use for singing / VO length, not as the primary lip-sync demo.\n\n```text\nClose-up of a singer performing the uploaded track. Natural lip-sync, expressive face, stage lighting, camera holds steady on the performer.\n```\n\n## Duration\n\n- Set `duration` (1–20s) when the user locked a length.\n- **Leave `duration` empty** to let the model choose length from the prompt.\n- When `audio` is set, omit `duration` — length follows the audio (cap **20s**; TTS ≤ ~19s).\n\n## Iterate\n\nUse `draft: true` for cheap previews, then `draft: false` for the paid final. Defaults: `prompt_upsampling: true`, `save_audio: true`, `resolution: 720p`, `fps: 24`, `aspect_ratio: 16:9`.\n\nFile v1.0.14:references/p-video-2-quality-checklist.md\n\n# p-video-2 quality checklist\n\nAfter each `p-video-2` output is saved, **open the clip and review it visually** (and listen when audio is expected) against this checklist (agent vision review — see `generation-diversity`).\n\nShared motion / frame-anchor items: also run [p-video-quality-checklist.md](./p-video-quality-checklist.md).\n\n## Applies to\n\nSee the canonical mapping in `generation-diversity`. Audio-led / 1080p / draft path for `image-to-video`, `narrated-multi-scene`, `interactive-explainer`, and B-roll rows with imported audio. Visual-only cinematic pairs → `p-video-2-pro`. Use `p-video` for simpler / quicker clips.\n\n## Quality vs `p-video`\n\n- Subjects, backgrounds, and motion look sharper than a typical `p-video` draft of the same brief.\n- Close-up / foreground objects stay readable (hands, product, face).\n- Input-image identity holds when `image` was set.\n\n## Lip-sync and speakers\n\nWhen the prompt or `audio` implies speech or singing:\n\n- **Native speech (no imported `audio`):** mouth follows pause / line / rest — this is the quality bar vs `p-video`.\n- **Imported `audio`:** face/identity stays clean; do not fail the job solely because visemes are not tighter than `p-video`.\n- At most **two** speaking faces stay separable; reject crowded dialogue with 3+ talkers.\n- Talking-head-only jobs with no native scene audio should have used `p-video-avatar` instead.\n\n## Native audio\n\n- When `save_audio` is true (default), the file is not unintentionally silent.\n- Uploaded `audio` is embedded and not truncated (TTS was ≤ ~19s; API cap 20s).\n- Do not treat weak SFX as a pass if the brief was sound-effect-led — warn and offer a bed in post (`audio-prompting`).\n\n## Camera and length\n\n- Camera move is stable (slow push / hold / one dolly) — not an extreme cinematic stunt.\n- If `duration` was omitted (no audio), runtime matches the prompt’s implied beat.\n- If `duration` was set, runtime is in the 1–20s window and matches the request.\n- Output is 720p or 1080p as requested — not 4K.\n\n## Scene anchors\n\nWhen using pair or triple payloads, apply the pair/triple sections in [p-video-quality-checklist.md](./p-video-quality-checklist.md), substituting `p-video-2` for the prediction model.\n\nFile v1.0.14:references/p-video-animate-prompting.md\n\n# p-video-animate prompting\n\nMotion-transfer craft for `p-video-animate`. QA: [p-video-animate-quality-checklist.md](./p-video-animate-quality-checklist.md). Mixed reels: `avatar-multi-scene`.\n\n**Appearance from `image`, motion from `video`.** Wrong tool for identity swap on real footage → [p-video-replace-prompting.md](./p-video-replace-prompting.md).\n\n## Pairing gates (before every job)\n\nAsk:\n\n1. **Framing** — same body region (head-and-shoulders / medium / full body)?  \n2. **Pose** — facing and limb position roughly aligned with the template’s first frame?  \n3. **Visibility** — same body parts visible; no crop the video lacks?\n\n| Factor | Guidance |\n|--------|----------|\n| Shot size | Match close-up / medium / full |\n| Facing | Front still + profile motion → artifacts |\n| Limbs | If template waves arms, still must show arms |\n| Proportions | Human full-body dance on chibi often breaks gait |\n| Speaking templates | Mouth clear and large when source has dialogue |\n\n**Pairing failure:** head-and-shoulders still + full-body dance → model does **not** invent limbs; choreography is lost. Repose with `p-image-edit` or pick a closer template.\n\n## `instruction_prompt`\n\nOptional. Overrides **behavior**, not identity. **Leave blank** when source motion is already right.\n\n**Useful** — one specific end beat:\n\n```text\nAt the very end of the clip, just after her last gesture, she gives a clear thumbs-up toward the camera. Keep the source motion otherwise.\n```\n\n**Less useful** — redescribes the still:\n\n```text\nA confident woman in a charcoal blazer speaks to the camera in a modern office.\n```\n\n## Style variety\n\nPhotoreal, cartoon, 3D, and mascot stills can share one template when framing aligns — the still’s render style carries through.\n\n## Speaking motion sources\n\nWhen this template feeds lip-sync showcases, the source clip (often from `p-video-avatar`) must show clear speaking / lip movement. See [p-video-avatar-prompting.md](./p-video-avatar-prompting.md) and animate-beats.\n\n## Pre-send\n\n- [ ] Framing / pose / limbs match  \n- [ ] `instruction_prompt` blank or one concrete beat  \n- [ ] Not using animate for in-place replace  \n- [ ] Long templates split (~5s compute per 1s video)\n\nFile v1.0.14:references/p-video-animate-quality-checklist.md\n\n# p-video-animate quality checklist\n\nAfter each animate job, **open the source video, reference image, and output clip** and review them against this list (agent vision review — see `generation-diversity`).\n\n## Applies to\n\nSee the canonical mapping in `generation-diversity`.\n\n## Input gate (pre-render)\n\n- Source **`video`** is the intended motion template (camera path, acting, timing, scene structure); motion is **clear and readable** (not blurry, fast-cut, or low-contrast).\n- Reference **`image`** clearly shows the subject to animate (face/body unobstructed, rights cleared).\n- **First-frame alignment:** framing, pose, and visible limbs match the **first frame** of the source video (or repose with **`p-image-edit`** first).\n- **Mismatch risk:** head-and-shoulders still + full-body template → expect lost choreography, not full-body motion.\n- **Proportion fit:** human full-body motion on meme/mascot/chibi subjects often breaks legs, arms, and contact points—flag before generate.\n- **`instruction_prompt`** (if used) describes **behavior overrides only** — not a repeat of the image description.\n- **`resolution`** and **`target_fps`** match delivery spec.\n- Source longer than budget: plan to **split** the template and animate segments (~5 s compute per 1 s video).\n\n## Motion transfer fidelity\n\n- Output preserves source motion, timing, and camera movement (not a generic re-enactment).\n- Acting beats and scene structure track the reference video.\n- Subject identity and style come from the reference image, not the source video's actor.\n\n## Technical quality\n\n- No severe flicker, warping, or unstable anatomy during motion.\n- Audio (when `save_audio` is true) stays aligned with visual motion.\n- Output duration matches the source video length.\n\n## Clean delivery\n\n- No accidental overlays, stray text, or watermark-like artifacts unless requested.\n- Clip is ready for downstream edit, concat, or platform upload.\n\nFile v1.0.14:references/p-video-avatar-prompting.md\n\n# p-video-avatar prompting\n\nTalking-head prompt craft for `p-video-avatar`. Templates: `avatar-multi-scene`. Camera: [camera-lighting-vocabulary.md](./camera-lighting-vocabulary.md). Physics: [physics-safe-motion.md](./physics-safe-motion.md).\n\n**Do not** use OPEN/MID/CLOSE — the model treats beats as cuts and the clip feels cutty.\n\n## Three-layer stack\n\n| Layer | Field | Job |\n|-------|-------|-----|\n| 1. Plate | `image` | Locked approved still — mouth visible; quality caps the avatar |\n| 2. Voice | `voice_script` + `voice_prompt` | What they say + how they sound |\n| 3. Motion | `video_prompt` | Unique camera/gesture per clip |\n\nNever ship multi-scene reels where every row reuses `medium close-up, gentle dolly push-in`.\n\n## Field hygiene\n\n| Field | Write | Never |\n|-------|-------|-------|\n| **`voice_script`** | Natural spoken lines | Brochure / slogan paste as the only line without human rhythm |\n| **`voice_prompt`** | Short delivery: pacing, warmth, archetype | Product names, script lines, long scene descriptions |\n| **`video_prompt`** | MCU, one slow push-in or static, speaks to camera | OPEN/MID/CLOSE; walk across room; hold up documents; wild gestures |\n\nStylized hosts: match energy to medium (anime slightly more expressive; documentary restrained). Separate hero stills per `visual_style_tag`.\n\n## Micro-actions (talking-head Details Law)\n\nPrefer face/eye micro-moves over locomotion:\n\n- eyes lift to lens, slight nod, natural blink rhythm, subtle lean  \n- one small hand-to-chest max  \n\nAvoid physics traps while speaking ([physics-safe-motion.md](./physics-safe-motion.md) avatar subsection).\n\n## Negative prompt (experimental)\n\nAPI `negative_prompt` = **noun suppression list** (subtitles, captions, watermark…), not creative wording. Strength ~0.3–0.4. Primary fix: positive-only stills (`plain unmarked walls`). See SKILL for defaults.\n\n## Good / bad triples\n\n**Good**\n\n```text\nvoice_script: \"So we tried something weird last quarter — and it actually worked.\"\nvoice_prompt: Natural conversational tone, relaxed pacing, real pauses, honest not salesy.\nvideo_prompt: Medium close-up speaking directly to lens, one very slow push-in, steady light, natural head motion, no cuts.\n```\n\n**Bad**\n\n```text\nvoice_prompt: Mention Pruna and our 10x faster inference in an exciting cinematic way.\nvideo_prompt: OPEN: hold. MID: she walks across the room waving a laptop. CLOSE: product hero shot.\n```\n\n## Pre-send\n\n- [ ] Plate approved; mouth visible  \n- [ ] `voice` locked per character across scenes  \n- [ ] Unique `video_prompt` per clip  \n- [ ] No OPEN/MID/CLOSE  \n- [ ] `voice_prompt` has no script/product paste  \n\nQA: [p-video-avatar-quality-checklist.md](./p-video-avatar-quality-checklist.md).\n\nArchive v1.0.13: 24 files, 42331 bytes\n\nFiles: references/audio-in-video-prompting.md (3342b), references/camera-lighting-vocabulary.md (2812b), references/clip-chaining.md (2668b), references/p-video-2-pro-prompting.md (4820b), references/p-video-2-pro-quality-checklist.md (2385b), references/p-video-2-prompting.md (3352b), references/p-video-2-quality-checklist.md (2231b), references/p-video-animate-prompting.md (2230b), references/p-video-animate-quality-checklist.md (1940b), references/p-video-avatar-prompting.md (2733b), references/p-video-avatar-quality-checklist.md (2586b), references/p-video-edit-prompting.md (3745b), references/p-video-edit-quality-checklist.md (2068b), references/p-video-quality-checklist.md (2507b), references/p-video-replace-prompting.md (3103b), references/p-video-replace-quality-checklist.md (3852b), references/physics-safe-motion.md (2859b), references/prompt-dramaturgy.md (3959b), references/scene-anchor-pair.md (11078b), references/scene-anchor-triple.md (7523b), skill-card.md (3639b), skill.manifest.json (729b), SKILL.md (8368b), _meta.json (135b)\n\nFile v1.0.13:SKILL.md\n\n---\nname: video-prompting\ndescription: Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining.\nlicense: MIT\nmetadata:\n  version: \"1.0.13\"\n  package: pruna-skills\n---\n\n# Video prompting\n\nVendor-neutral craft for **short video / motion** generation. Works with Pruna `p-video-2-pro` / `p-video-2` / `p-video` family, Runway, Kling, Luma, Veo, and similar APIs.\n\n## Install\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n\n## When to use\n\n- Text-to-video or image-to-video prompts\n- Start/end frame (anchor pair) or narrated beat (anchor triple) specs\n- Camera and lighting vocabulary in motion lines\n- Physics-safe subject motion\n- Multi-clip continuity / clip chaining\n- Talking-head, motion-transfer, slot-replace, or instruction-based video-edit prompts\n\n## Works with\n\nPruna `p-video-2-pro` / `p-video-2` / `p-video` / `p-video-avatar` / `p-video-animate` / `p-video-replace` / `p-video-edit`, Runway Gen-3, Kling, Luma Dream Machine, Veo, and other video models. Cinematic generation (generated audio, first/last frame): `p-video-2-pro`. 1080p / imported audio / draft: `p-video-2`. Cheaper, faster drafts: `p-video`.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `image-prompting` | Use when crafting still-image prompts for any generative model — composition, identity sheets, edits, try-on, and photoreal personas. | `npx skills add PrunaAI/pruna-skills@image-prompting -y` |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `music-video` | Use when someone wants a full music video — original song or vocals, performance clips, B-roll, and lyric-synced edits. | `npx skills add PrunaAI/pruna-skills@music-video -y` |\n| `narrated-multi-scene` | Use when someone wants a multi-part story with voiceover — episodic B-roll, chaptered promo, or several linked video scenes without on-camera dialogue. | `npx skills add PrunaAI/pruna-skills@narrated-multi-scene -y` |\n| `p-video-2-pro` | Use when someone wants a cinematic clip from text or start/end frames — product ads, documentary shots, or dialogue with generated audio. Not for 1080p, imported audio tracks, or talking-head-only hosts. | `npx skills add PrunaAI/pruna-skills@p-video-2-pro -y` |\n| `p-video-2` | Use when someone wants a polished short clip from text, images, or imported audio — 1080p B-roll, start/end frame animation, or a motion shot with a mixed track. Not for cinematic generated-audio clips or talking-head-only hosts. | `npx skills add PrunaAI/pruna-skills@p-video-2 -y` |\n| `p-video` | Use when someone wants a simple short clip from text or images — quick B-roll, drafts, or start/end frame animation. Not when the brief needs cinematic generation, highest quality, tight lip-sync, or imported audio at 1080p. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\n## Guide habit\n\nIn the **first reply**, name `` `video-prompting` `` in backticks. When aspect, resolution, duration, or embed-vs-post audio are open, open intake → **`generation-diversity`** clarification intake. For `p-video-2-pro` / `p-video-2` / `p-video` motion lines, cite OPEN/MID/CLOSE dramaturgy and **Worked example — product B-roll** in [prompt-dramaturgy.md](./references/prompt-dramaturgy.md). Cinematic generation: `p-video-2-pro` (`p-video-2-pro-prompting`). 1080p / imported audio / draft: `p-video-2` (`p-video-2-prompting`). Cheaper drafts: `p-video`. Audio-led clips: **≤ ~19s** TTS before embed on `p-video-2` — see [audio-in-video-prompting.md](./references/audio-in-video-prompting.md).\n\n## Before generating\n\n1. Follow `generation-diversity` first.\n2. Read in order:\n   - [prompt-dramaturgy.md](./references/prompt-dramaturgy.md) — Details Law, OPEN/MID/CLOSE\n   - [camera-lighting-vocabulary.md](./references/camera-lighting-vocabulary.md)\n   - [physics-safe-motion.md](./references/physics-safe-motion.md)\n   - [audio-in-video-prompting.md](./references/audio-in-video-prompting.md) when sound matters\n   - [clip-chaining.md](./references/clip-chaining.md) for multi-clip continuity\n   - [scene-anchor-pair.md](./references/scene-anchor-pair.md) / [scene-anchor-triple.md](./references/scene-anchor-triple.md) for frame (+ audio) payloads\n3. Tool-specific craft when needed:\n   - [p-video-2-pro-prompting.md](./references/p-video-2-pro-prompting.md)\n   - [p-video-2-prompting.md](./references/p-video-2-prompting.md)\n   - [p-video-avatar-prompting.md](./references/p-video-avatar-prompting.md)\n   - [p-video-animate-prompting.md](./references/p-video-animate-prompting.md)\n   - [p-video-replace-prompting.md](./references/p-video-replace-prompting.md)\n   - [p-video-edit-prompting.md](./references/p-video-edit-prompting.md)\n4. Validate with the matching `*-quality-checklist.md` in `./references/`.\n\nProduct B-roll and OPEN/MID/CLOSE samples: **Worked example — product B-roll** in [prompt-dramaturgy.md](./references/prompt-dramaturgy.md).\n\n## Pruna tools\n\nMatching install for every model named above. Pick what you need:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `p-video-2-pro` | Use when someone wants a cinematic clip from text or start/end frames — product ads, documentary shots, or dialogue with generated audio. Not for 1080p, imported audio tracks, or talking-head-only hosts. | `npx skills add PrunaAI/pruna-skills@p-video-2-pro -y` |\n| `p-video-2` | Use when someone wants a polished short clip from text, images, or imported audio — 1080p B-roll, start/end frame animation, or a motion shot with a mixed track. Not for cinematic generated-audio clips or talking-head-only hosts. | `npx skills add PrunaAI/pruna-skills@p-video-2 -y` |\n| `p-video` | Use when someone wants a simple short clip from text or images — quick B-roll, drafts, or start/end frame animation. Not when the brief needs cinematic generation, highest quality, tight lip-sync, or imported audio at 1080p. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n| `p-video-avatar` | Use when someone wants a person on camera speaking a script — lip-synced host, spokesperson, or narrated avatar from a portrait photo. | `npx skills add PrunaAI/pruna-skills@p-video-avatar -y` |\n| `p-video-animate` | Use when someone wants a photo to move like another video — motion transfer, dance remixes, or performance variations from a template clip. | `npx skills add PrunaAI/pruna-skills@p-video-animate -y` |\n| `p-video-replace` | Use when someone wants to swap a person, outfit, or product inside existing footage while keeping the camera move and audio. | `npx skills add PrunaAI/pruna-skills@p-video-replace -y` |\n| `p-video-edit` | Use when someone wants to edit an existing video with a text instruction — recolor, restyle, remove or add objects, change environment or lighting, update on-screen text, or apply optional reference-guided product and accessory edits. Not for a new clip from scratch or ffmpeg assembly. | `npx skills add PrunaAI/pruna-skills@p-video-edit -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFile v1.0.13:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"video-prompting\",\n  \"version\": \"1.0.13\",\n  \"publishedAt\": 1789653147830\n}\n\nFile v1.0.13:references/audio-in-video-prompting.md\n\n# Audio-in-video prompting (`p-video`)\n\nHow to **write prompts** when sound matters on `p-video`. Layering / tool picker: `audio-prompting`. Talking heads: [p-video-avatar-prompting.md](./p-video-avatar-prompting.md).\n\n## Three modes\n\n| Mode | API | Prompt job |\n|------|-----|------------|\n| **A — Native SFX / dialogue** | `prompt` + optional `save_audio`; use `duration` | Name **diegetic** sounds the picture should emit |\n| **B — Uploaded audio (preferred for VO/music)** | `audio` URL; **omit `duration`**; `save_audio: true` | Motion matches **mood/beats** of the track — do not paste VO text into `prompt` |\n| **C — Post bed** | Stable Audio mixed under VO in ffmpeg | Bed prompt is separate (`stable-audio-2.5` + `audio-prompting`); video prompt ignores the bed |\n\nNever generate silent `p-video` and post-mux narration unless re-render is impossible (truncation risk). Probe TTS ≤ ~19s before Mode B.\n\n## Mode A — native SFX / dialogue\n\n**When:** `save_audio: true`, no uploaded `audio`, user wants diegetic SFX and/or spoken lines in the clip.\n\n### Diegetic SFX\n\nBe concrete; skip `cinematic soundscape`:\n\n| Bad | Good |\n|-----|------|\n| `epic soundtrack vibe` | `rain ticks on the awning, distant train horn once` |\n| `dramatic music` | `crowd murmur swells, single glass clink` |\n\nKeep cues short — the model invents audio from visual+prompt context when `save_audio` is on.\n\n### Native dialogue\n\nPut **exact spoken words in double quotes** inside the motion `prompt` (Mode A only). Placeholders in square brackets below are docs notation, not literal syntax — do not confuse with Gemini TTS `[tags]`.\n\n**Template** (inside OPEN/MID/CLOSE):\n\n```text\nMID: [same subject] says \"[LINE]\" — mouth open, [one gesture toward target]; [optional diegetic SFX cue]\n```\n\nRules:\n\n- Exact spoken words in **double quotes** — never paraphrase the line\n- Name **who** speaks — repeat the subject label for continuity (`same [role]`)\n- Pair every line with **mouth state** + **one gesture** (point, turn to camera, hand on prop)\n- Keep lines **short** (1–2 per beat)\n- Diegetic SFX can sit beside dialogue — stay concrete (table above)\n- **Mode B:** when `input.audio` is set, do **not** put the VO transcript in `prompt` — motion matches mood only\n\n## Mode B — motion matches uploaded audio\n\n```text\nOPEN: hold wide on dog in tall grass, warm afternoon light.\nMID: gentle push-in as he searches; tail motion matches narrator energy; grass sways.\nCLOSE: settle on end pose — curious head tilt.\n```\n\nRules:\n\n- Describe **picture motion**, not the spoken words.  \n- Match energy: tense VO → tighter push-in; calm story → slow drift.  \n- Optional: `motion matches narrator mood` once — not a transcript.  \n- Triple anchors: [scene-anchor-triple.md](./scene-anchor-triple.md).\n\n## Avatar path (not Mode B fields)\n\n`p-video-avatar` uses `voice_script` + `voice_prompt` + `video_prompt` — not `p-video` `input.audio` for the spoken line. See avatar prompting ref. Do not paste script lines into `voice_prompt`.\n\n## Pre-send\n\n- [ ] Mode A / B / C chosen  \n- [ ] Mode B: `duration` omitted; TTS length probed  \n- [ ] No VO transcript inside motion `prompt` (Mode B)  \n- [ ] Diegetic cues concrete (Mode A) or mood-aligned (Mode B)\n- [ ] Mode A dialogue (if any): quoted line + named speaker + mouth + one gesture\n\nFile v1.0.13:references/camera-lighting-vocabulary.md\n\n# Camera and lighting vocabulary\n\nShared lexicon for `camera_tag` / `lighting_tag` (stills) and motion lines in `p-video` / `p-video-avatar` prompts. Diversity axes: generation-diversity.md#visual-variety (`generation-diversity`). Dramaturgy: [prompt-dramaturgy.md](./prompt-dramaturgy.md).\n\n**Sources:** patterns adapted from [smixs/visual-skills](https://github.com/smixs/visual-skills) and [inference-sh/skills](https://github.com/inference-sh/skills) (MIT); rewritten for Pruna.\n\n## Framing ladder\n\n| Term | Use when |\n|------|----------|\n| ECU (extreme close-up) | Eyes, hands, product detail |\n| CU (close-up) | Face, emotion |\n| MCU (medium close-up) | Talking head default |\n| MS (medium shot) | Waist-up action |\n| MLS / FS | Full body travel |\n| WS / EWS | Environment as character |\n\nLog as `camera_tag`, e.g. `medium close-up, slight low angle`.\n\n## Lens roles (optional but sharp)\n\n| Lens | Feel |\n|------|------|\n| 24mm | Wide, immersive, exaggerated space |\n| 35mm | Documentary natural |\n| 50mm | Intimate human perspective |\n| 85mm | Portrait, compressed background |\n| Macro | Texture, product detail |\n\nExample: `shot on 50mm, eye-level`.\n\n## Camera moves (pick one for MID)\n\n| Move | Prompt cue |\n|------|------------|\n| Dolly / push-in | `slow dolly in`, `gentle push-in` |\n| Dolly out | `slow pull back revealing the room` |\n| Pan | `gentle pan left across the alley` |\n| Tilt | `tilt up from hands to face` |\n| Track / truck | `shoulder-height tracking shot beside the subject` |\n| Crane | `slow crane down past neon signs` |\n| Static + atmosphere | `locked camera, steam rises, light shifts` |\n| Handheld | `subtle handheld drift` (use sparingly) |\n\nAvoid whip pans and stacked contradictory moves in one short clip.\n\n## Motivated lighting\n\nPrefer **named sources** over “beautiful lighting”:\n\n| Source | Example cue |\n|--------|-------------|\n| Window / dawn | `dawn light spreads across the desk` |\n| Practical | `warm lamp spill, cool window fill` |\n| Neon / gel | `magenta-cyan neon rim, wet reflections` |\n| Overhead institutional | `cold fluorescent flicker` |\n| Fire / candle | `candle flicker on faces` (avatar: often too transition-y — prefer steady) |\n| Overcast soft | `soft overcast skylight, low contrast` |\n\nLog as `lighting_tag`. Hex in stills when brand colors matter (`#0d3d2d rim`).\n\n## Palette cues (one look)\n\nPick one coherent palette phrase: `teal-magenta night`, `warm tungsten interior`, `bleached noon desert`, `desaturated documentary`.\n\nDo not stack competing genre looks in one prompt.\n\n## Avatar-friendly defaults\n\nTalking heads: **MCU**, one slow push-in or static, **steady light**, mouth visible. Variety across scenes = change angle/background still, not five camera moves mid-line. See [p-video-avatar-prompting.md](./p-video-avatar-prompting.md).\n\nFile v1.0.13:references/clip-chaining.md\n\n# Clip chaining (multi-scene video)\n\nWhen and how to continue motion across `p-video` clips. Plan JSON examples stay in [scene-anchor-pair.md](./scene-anchor-pair.md) and [scene-anchor-triple.md](./scene-anchor-triple.md); this page is the decision tree + prompt rules.\n\nWorkflows: `visual-transition-reel` · `narrated-multi-scene`.\n\n## Decision tree\n\n```text\nDoes motion continue in the same place/moment (no time jump)?\n  NO  → chain_from_previous: false — hard cut; compose a new OPENING still\n  YES → chain_from_previous: true\n        Prefer frame_chain_mode: extract_last_frame (sequential renders)\n        Only use planned_stills if you accept possible cut jumps\n```\n\n| Situation | `chain_from_previous` | Join |\n|-----------|----------------------|------|\n| Continuous action (run → leap) | `true` | Short crossfade ~0.12–0.15s after extract |\n| New beat / location / pause | `false` | Hard cut (0 crossfade) |\n| First scene | `false` | — |\n| Montage vignettes (no shared motion) | `false` + `parallel_vignettes` | Hard cuts; parallel renders OK |\n\n| `frame_chain_mode` | Next scene `image` | Render order |\n|--------------------|--------------------|--------------|\n| **`extract_last_frame`** | ffmpeg last frame of prior clip | **Sequential** when any scene chains |\n| **`parallel_vignettes`** | each scene’s own start still | **Parallel** |\n| **`planned_stills`** | prior scene end still URL | Parallel once stills exist — higher jump risk |\n\n**Why extract?** Planned end stills often differ from the model’s actual last frame → visible jump.\n\n## Prompt rules for chained beats\n\n1. **Same subject language** — repeat “same [character]” in OPEN/MID/CLOSE.  \n2. **No teleport** — ban `cut to`, `suddenly in`, `walls disappear`; use `gradually`, `walks through`, `ease into`.  \n3. **Match lighting era** — chained clips share `style_bible` and time-of-day.  \n4. **Exit / enter continuity** — if scene 1 CLOSE faces right, scene 2 OPEN should not hard-flip screen direction without a motivated turn.  \n5. **Hard-cut scenes** — treat as fresh OPENING; do not assume prior pose.\n\n## Assembly notes\n\n1. Concat in scene order (ffmpeg concat — see the workflow skill).  \n2. Per-join `crossfades`: chain ~0.12–0.15s; hard cuts 0.  \n3. Normalize audio (48 kHz stereo) when mixing formats.  \n4. Optional bed under native SFX — `audio-prompting`.\n\n## Intake checklist\n\n- [ ] Each scene: chain flag only if motion truly continues  \n- [ ] `frame_chain_mode` chosen  \n- [ ] Chained prompts pass Details Law ([prompt-dramaturgy.md](./prompt-dramaturgy.md))  \n- [ ] Physics tier OK ([physics-safe-motion.md](./physics-safe-motion.md))\n\nFile v1.0.13:references/p-video-2-pro-prompting.md\n\n# p-video-2-pro prompting\n\nPrompt craft unique to `p-video-2-pro` (cinematic generation lane next to `p-video-2`). Shared dramaturgy, camera, and physics: [prompt-dramaturgy.md](./prompt-dramaturgy.md), [camera-lighting-vocabulary.md](./camera-lighting-vocabulary.md), [physics-safe-motion.md](./physics-safe-motion.md). Visual frame payloads: [scene-anchor-pair.md](./scene-anchor-pair.md). QA: [p-video-2-pro-quality-checklist.md](./p-video-2-pro-quality-checklist.md). Uploaded audio triples stay on `p-video-2` — [scene-anchor-triple.md](./scene-anchor-triple.md).\n\n**Cinematic generation.** Use `p-video-2-pro` for text-to-video or first/last-frame clips with **generated audio**. Use `p-video-2` for **1080p**, **imported audio**, or **draft** previews. Use `p-video` for cheaper, faster drafts.\n\n## Strengths to write toward\n\n- **Multi-beat cinematic blocking** in one clip (product unfold, dialogue, chase, weather) — 8s is a typical locked-in length\n- **Heavier physics:** water, snow spray, fire, fabric, hair\n- **Full-body anatomy** that stays coherent through fast motion\n- **Two-shot dialogue** with lip-sync — write the spoken line into the prompt\n- **First + last frame** as a first-class control\n- **Three-level prompt upsampler** (`off` / `turbo` / `max`), independent of `mode`\n- `mode: speed` vs `mode: quality` on the same prompt\n- **Generated audio** already in the clip — write score, ambience, SFX, or dialogue in the prompt\n\n## Limits (do not fight them)\n\n- **No audio import** — muxed VO / music belongs on `p-video-2`\n- Max **768p** and **15s** — not 1080p / 20s / 48 fps\n- **No draft mode** — `mode: speed` is the fast recipe; `quality` is slower\n- Not an editor (`p-video-edit`) or talking-head avatar (`p-video-avatar`)\n- Native **4K** is not supported\n- **More than two speakers** — speaker separation degrades\n\n## Fast pass vs locked-in\n\n**Fast pass** — subject + action + scene. Enough for first looks. Iterate in `mode: speed`.\n\n```text\nA matte stainless pour-over kettle sits on a seamless light-gray studio sweep. Thin steam rises from the spout.\n```\n\n**Locked-in** — add camera, lighting, style, and audio. Optional first-frame or last-frame still for repeatable runs.\n\n```text\nA narrative film scene, 1970s Roman trattoria at night, warm tungsten, cigarette haze, Super-8 grain. A man and a woman in period clothes sit at a small table with wine. She leans in and quietly says, \"Then we leave before sunrise.\" The camera slowly dollies around the table. Audio: Italian radio pop from a small speaker, plates, low room tone, her line clear and close.\n```\n\n## Mode recipes (rewrite for the brief — do not paste)\n\n**Text-to-video** — subject + action + camera + lighting + audio intent in one coherent line.\n\n```text\nA dramatic cinematic shot of an isolated lighthouse standing on a rocky cliff during an enormous Atlantic storm at dusk. The camera begins close to the lighthouse keeper standing outside near the railing as violent wind pulls at his coat. He turns toward the ocean just as a massive wave crashes against the cliff and sends water high above the lighthouse. The camera rapidly pulls back to reveal the scale of the storm. Photorealistic water physics, heavy rain, turbulent clouds, realistic human movement, and epic scale.\n```\n\n**Image-to-video (first frame)** — keep motion aligned with the reference frame.\n\n```text\nThe camera slowly pushes in. The person turns their head and smiles naturally. Soft studio lighting, shallow depth of field.\n```\n\n**First + last frame** — motion consistent with both stills; end exactly on the last-frame composition.\n\n```text\nStart on the still product hero. The camera holds, then a slow push-in as light moves across the metal. End exactly on the last-frame packshot, label readable, no extra props.\n```\n\n**Native speech (T2V)** — write the spoken line into the prompt. There is no `audio` field.\n\n```text\nA woman faces the camera and says \"I'll be there in five.\" She pauses, delivers the line, then rests. Natural mouth motion, soft window light, camera holds steady.\n```\n\n## Duration, mode, upsampler\n\n- Set `duration` (5–15s; default **5**). Partner demos often use **8s** for locked-in cinematic beats.\n- `mode: speed` (default) for iteration; `mode: quality` for the slower, higher-fidelity recipe.\n- `prompt_upsampler`: `off` when copy is already locked; `turbo` (default); `max` when the source prompt is short and the scene needs more described detail. Compare `off` vs `turbo` vs `max` on the **same `seed`** before scaling.\n- Defaults: `resolution: 768p`, `aspect_ratio: 16:9`, 24 fps, generated audio.\n\n## Iterate\n\nThere is no `draft` flag. Iterate in `mode: speed`, then final with `mode: quality` when fidelity matters. Do not send `audio`, `fps`, `save_audio`, or `prompt_upsampling`.\n\nFile v1.0.13:references/p-video-2-pro-quality-checklist.md\n\n# p-video-2-pro quality checklist\n\nAfter each `p-video-2-pro` output is saved, **open the clip and review it visually** (and listen — generated audio is expected) against this checklist (agent vision review — see `generation-diversity`).\n\nShared motion / frame-anchor items: also run [p-video-quality-checklist.md](./p-video-quality-checklist.md).\n\n## Applies to\n\nSee the canonical mapping in `generation-diversity`. Cinematic generation path for visual-only `image-to-video` and `visual-transition-reel`. Audio-led jobs stay on `p-video-2`.\n\n## Quality vs `p-video-2` / `p-video`\n\n- Multi-beat blocking holds (product unfold, dialogue, weather) without collapsing into a single pose.\n- Heavier physics stay readable (water, fire, fabric, hair) when the prompt asked for them.\n- Full-body anatomy stays coherent through fast motion.\n- Close-up / foreground objects stay readable (hands, product, face, packshot label).\n- Input-image identity holds when `image` was set; last-frame composition matches when `last_frame_image` was set.\n\n## Lip-sync and speakers\n\nWhen the prompt implies speech or singing (there is **no imported `audio`**):\n\n- Mouth follows the **written line** — pause / line / rest.\n- At most **two** speaking faces stay separable; reject crowded dialogue with 3+ talkers.\n- Talking-head-only jobs with no native scene audio should have used `p-video-avatar` instead.\n\n## Generated audio\n\n- The file is not unintentionally silent — output includes generated audio.\n- Score, ambience, SFX, or dialogue match what the prompt named.\n- Do not fail the job for missing an **uploaded** track — that brief belongs on `p-video-2`.\n- Do not treat weak SFX as a pass if the brief was sound-effect-led — warn and offer a bed in post (`audio-prompting`) or a `p-video-2` mux.\n\n## Camera and length\n\n- Runtime is in the **5–15s** window and matches the requested `duration` (default 5).\n- Output is **480p or 768p** as requested — not 720p / 1080p / 4K.\n- Output is **24 fps**.\n- When `mode: quality` was set, the clip should look tighter than a `mode: speed` preview of the same seed — if it does not, say so.\n\n## Scene anchors\n\nWhen using a first/last-frame pair, apply the pair sections in [p-video-quality-checklist.md](./p-video-quality-checklist.md), substituting `p-video-2-pro` for the prediction model. Do not expect a scene-anchor triple on this model.\n\nFile v1.0.13:references/p-video-2-prompting.md\n\n# p-video-2 prompting\n\nPrompt craft unique to `p-video-2` (quality-focused successor to `p-video`). Shared dramaturgy, camera, and physics: [prompt-dramaturgy.md](./prompt-dramaturgy.md), [camera-lighting-vocabulary.md](./camera-lighting-vocabulary.md), [physics-safe-motion.md](./physics-safe-motion.md). Frame payloads: [scene-anchor-pair.md](./scene-anchor-pair.md) / [scene-anchor-triple.md](./scene-anchor-triple.md). QA: [p-video-2-quality-checklist.md](./p-video-2-quality-checklist.md).\n\n**1080p / imported-audio path.** Use `p-video-2` when the brief needs 1080p, a mixed track, or draft previews. Use `p-video-2-pro` for cinematic generation with generated audio. Use `p-video` for simpler, quicker clips.\n\n## Strengths to write toward\n\n- Sharper subjects, backgrounds, and motion than `p-video` — especially **close-ups and foreground objects**\n- Stronger **lip-sync on native speech** — T2V with `save_audio: true` and **no imported `audio`**: the model pauses, delivers the line, then rests, and the mouth follows. **Lead dialogue / lip-sync jobs here**\n- **Native audio** in the output (`save_audio` defaults true). Imported `audio` still works and keeps a **cleaner face / identity**; viseme lock vs `p-video` is mixed — do not lead lip-sync demos with a muxed wav\n- Stronger **identity / input-image consistency**\n- One endpoint: T2V + I2V + audio-conditioned\n\n## Limits (do not fight them)\n\n- Not designed for **extreme cinematic camera** (crash zooms, whip pans, chaotic handheld)\n- Complex **multi-scene storytelling** in one clip is weaker — split beats\n- Native **4K** is not supported\n- **More than two speakers** — speaker separation degrades\n- **SFX-led** clips (the ask is sound effects, not picture + optional bed) are currently limited\n\n## Mode recipes (rewrite for the brief — do not paste)\n\n**Text-to-video** — subject + motion + camera + lighting + audio intent in one coherent line.\n\n```text\nA sports car drifting through a neon-lit city at night, cinematic aerial shot, wet asphalt reflections, engine roar and tire screech.\n```\n\n**Image-to-video** — keep motion **subtle** and aligned with the reference frame. Prefer a **stable** camera.\n\n```text\nThe camera slowly pushes in. The person turns their head and smiles naturally. Soft studio lighting, shallow depth of field.\n```\n\n**Native speech (T2V, no `audio`)** — write the spoken line into the prompt. This is the lip-sync path.\n\n```text\nA woman faces the camera and says \"I'll be there in five.\" She pauses, delivers the line, then rests. Natural mouth motion, soft window light, camera holds steady.\n```\n\n**Audio-conditioned** — name the performer and hold. One or two faces max. Use for singing / VO length, not as the primary lip-sync demo.\n\n```text\nClose-up of a singer performing the uploaded track. Natural lip-sync, expressive face, stage lighting, camera holds steady on the performer.\n```\n\n## Duration\n\n- Set `duration` (1–20s) when the user locked a length.\n- **Leave `duration` empty** to let the model choose length from the prompt.\n- When `audio` is set, omit `duration` — length follows the audio (cap **20s**; TTS ≤ ~19s).\n\n## Iterate\n\nUse `draft: true` for cheap previews, then `draft: false` for the paid final. Defaults: `prompt_upsampling: true`, `save_audio: true`, `resolution: 720p`, `fps: 24`, `aspect_ratio: 16:9`.\n\nFile v1.0.13:references/p-video-2-quality-checklist.md\n\n# p-video-2 quality checklist\n\nAfter each `p-video-2` output is saved, **open the clip and review it visually** (and listen when audio is expected) against this checklist (agent vision review — see `generation-diversity`).\n\nShared motion / frame-anchor items: also run [p-video-quality-checklist.md](./p-video-quality-checklist.md).\n\n## Applies to\n\nSee the canonical mapping in `generation-diversity`. Audio-led / 1080p / draft path for `image-to-video`, `narrated-multi-scene`, `interactive-explainer`, and B-roll rows with imported audio. Visual-only cinematic pairs → `p-video-2-pro`. Use `p-video` for simpler / quicker clips.\n\n## Quality vs `p-video`\n\n- Subjects, backgrounds, and motion look sharper than a typical `p-video` draft of the same brief.\n- Close-up / foreground objects stay readable (hands, product, face).\n- Input-image identity holds when `image` was set.\n\n## Lip-sync and speakers\n\nWhen the prompt or `audio` implies speech or singing:\n\n- **Native speech (no imported `audio`):** mouth follows pause / line / rest — this is the quality bar vs `p-video`.\n- **Imported `audio`:** face/identity stays clean; do not fail the job solely because visemes are not tighter than `p-video`.\n- At most **two** speaking faces stay separable; reject crowded dialogue with 3+ talkers.\n- Talking-head-only jobs with no native scene audio should have used `p-video-avatar` instead.\n\n## Native audio\n\n- When `save_audio` is true (default), the file is not unintentionally silent.\n- Uploaded `audio` is embedded and not truncated (TTS was ≤ ~19s; API cap 20s).\n- Do not treat weak SFX as a pass if the brief was sound-effect-led — warn and offer a bed in post (`audio-prompting`).\n\n## Camera and length\n\n- Camera move is stable (slow push / hold / one dolly) — not an extreme cinematic stunt.\n- If `duration` was omitted (no audio), runtime matches the prompt’s implied beat.\n- If `duration` was set, runtime is in the 1–20s window and matches the request.\n- Output is 720p or 1080p as requested — not 4K.\n\n## Scene anchors\n\nWhen using pair or triple payloads, apply the pair/triple sections in [p-video-quality-checklist.md](./p-video-quality-checklist.md), substituting `p-video-2` for the prediction model.\n\nFile v1.0.13:references/p-video-animate-prompting.md\n\n# p-video-animate prompting\n\nMotion-transfer craft for `p-video-animate`. QA: [p-video-animate-quality-checklist.md](./p-video-animate-quality-checklist.md). Mixed reels: `avatar-multi-scene`.\n\n**Appearance from `image`, motion from `video`.** Wrong tool for identity swap on real footage → [p-video-replace-prompting.md](./p-video-replace-prompting.md).\n\n## Pairing gates (before every job)\n\nAsk:\n\n1. **Framing** — same body region (head-and-shoulders / medium / full body)?  \n2. **Pose** — facing and limb position roughly aligned with the template’s first frame?  \n3. **Visibility** — same body parts visible; no crop the video lacks?\n\n| Factor | Guidance |\n|--------|----------|\n| Shot size | Match close-up / medium / full |\n| Facing | Front still + profile motion → artifacts |\n| Limbs | If template waves arms, still must show arms |\n| Proportions | Human full-body dance on chibi often breaks gait |\n| Speaking templates | Mouth clear and large when source has dialogue |\n\n**Pairing failure:** head-and-shoulders still + full-body dance → model does **not** invent limbs; choreography is lost. Repose with `p-image-edit` or pick a closer template.\n\n## `instruction_prompt`\n\nOptional. Overrides **behavior**, not identity. **Leave blank** when source motion is already right.\n\n**Useful** — one specific end beat:\n\n```text\nAt the very end of the clip, just after her last gesture, she gives a clear thumbs-up toward the camera. Keep the source motion otherwise.\n```\n\n**Less useful** — redescribes the still:\n\n```text\nA confident woman in a charcoal blazer speaks to the camera in a modern office.\n```\n\n## Style variety\n\nPhotoreal, cartoon, 3D, and mascot stills can share one template when framing aligns — the still’s render style carries through.\n\n## Speaking motion sources\n\nWhen this template feeds lip-sync showcases, the source clip (often from `p-video-avatar`) must show clear speaking / lip movement. See [p-video-avatar-prompting.md](./p-video-avatar-prompting.md) and animate-beats.\n\n## Pre-send\n\n- [ ] Framing / pose / limbs match  \n- [ ] `instruction_prompt` blank or one concrete beat  \n- [ ] Not using animate for in-place replace  \n- [ ] Long templates split (~5s compute per 1s video)\n\nFile v1.0.13:references/p-video-animate-quality-checklist.md\n\n# p-video-animate quality checklist\n\nAfter each animate job, **open the source video, reference image, and output clip** and review them against this list (agent vision review — see `generation-diversity`).\n\n## Applies to\n\nSee the canonical mapping in `generation-diversity`.\n\n## Input gate (pre-render)\n\n- Source **`video`** is the intended motion template (camera path, acting, timing, scene structure); motion is **clear and readable** (not blurry, fast-cut, or low-contrast).\n- Reference **`image`** clearly shows the subject to animate (face/body unobstructed, rights cleared).\n- **First-frame alignment:** framing, pose, and visible limbs match the **first frame** of the source video (or repose with **`p-image-edit`** first).\n- **Mismatch risk:** head-and-shoulders still + full-body template → expect lost choreography, not full-body motion.\n- **Proportion fit:** human full-body motion on meme/mascot/chibi subjects often breaks legs, arms, and contact points—flag before generate.\n- **`instruction_prompt`** (if used) describes **behavior overrides only** — not a repeat of the image description.\n- **`resolution`** and **`target_fps`** match delivery spec.\n- Source longer than budget: plan to **split** the template and animate segments (~5 s compute per 1 s video).\n\n## Motion transfer fidelity\n\n- Output preserves source motion, timing, and camera movement (not a generic re-enactment).\n- Acting beats and scene structure track the reference video.\n- Subject identity and style come from the reference image, not the source video's actor.\n\n## Technical quality\n\n- No severe flicker, warping, or unstable anatomy during motion.\n- Audio (when `save_audio` is true) stays aligned with visual motion.\n- Output duration matches the source video length.\n\n## Clean delivery\n\n- No accidental overlays, stray text, or watermark-like artifacts unless requested.\n- Clip is ready for downstream edit, concat, or platform upload.\n\nFile v1.0.13:references/p-video-avatar-prompting.md\n\n# p-video-avatar prompting\n\nTalking-head prompt craft for `p-video-avatar`. Templates: `avatar-multi-scene`. Camera: [camera-lighting-vocabulary.md](./camera-lighting-vocabulary.md). Physics: [physics-safe-motion.md](./physics-safe-motion.md).\n\n**Do not** use OPEN/MID/CLOSE — the model treats beats as cuts and the clip feels cutty.\n\n## Three-layer stack\n\n| Layer | Field | Job |\n|-------|-------|-----|\n| 1. Plate | `image` | Locked approved still — mouth visible; quality caps the avatar |\n| 2. Voice | `voice_script` + `voice_prompt` | What they say + how they sound |\n| 3. Motion | `video_prompt` | Unique camera/gesture per clip |\n\nNever ship multi-scene reels where every row reuses `medium close-up, gentle dolly push-in`.\n\n## Field hygiene\n\n| Field | Write | Never |\n|-------|-------|-------|\n| **`voice_script`** | Natural spoken lines | Brochure / slogan paste as the only line without human rhythm |\n| **`voice_prompt`** | Short delivery: pacing, warmth, archetype | Product names, script lines, long scene descriptions |\n| **`video_prompt`** | MCU, one slow push-in or static, speaks to camera | OPEN/MID/CLOSE; walk across room; hold up documents; wild gestures |\n\nStylized hosts: match energy to medium (anime slightly more expressive; documentary restrained). Separate hero stills per `visual_style_tag`.\n\n## Micro-actions (talking-head Details Law)\n\nPrefer face/eye micro-moves over locomotion:\n\n- eyes lift to lens, slight nod, natural blink rhythm, subtle lean  \n- one small hand-to-chest max  \n\nAvoid physics traps while speaking ([physics-safe-motion.md](./physics-safe-motion.md) avatar subsection).\n\n## Negative prompt (experimental)\n\nAPI `negative_prompt` = **noun suppression list** (subtitles, captions, watermark…), not creative wording. Strength ~0.3–0.4. Primary fix: positive-only stills (`plain unmarked walls`). See SKILL for defaults.\n\n## Good / bad triples\n\n**Good**\n\n```text\nvoice_script: \"So we tried something weird last quarter — and it actually worked.\"\nvoice_prompt: Natural conversational tone, relaxed pacing, real pauses, honest not salesy.\nvideo_prompt: Medium close-up speaking directly to lens, one very slow push-in, steady light, natural head motion, no cuts.\n```\n\n**Bad**\n\n```text\nvoice_prompt: Mention Pruna and our 10x faster inference in an exciting cinematic way.\nvideo_prompt: OPEN: hold. MID: she walks across the room waving a laptop. CLOSE: product hero shot.\n```\n\n## Pre-send\n\n- [ ] Plate approved; mouth visible  \n- [ ] `voice` locked per character across scenes  \n- [ ] Unique `video_prompt` per clip  \n- [ ] No OPEN/MID/CLOSE  \n- [ ] `voice_prompt` has no script/product paste  \n\nQA: [p-video-avatar-quality-checklist.md](./p-video-avatar-quality-checklist.md).\n\nArchive v1.0.12: 22 files, 37920 bytes\n\nFiles: references/audio-in-video-prompting.md (3342b), references/camera-lighting-vocabulary.md (2812b), references/clip-chaining.md (2668b), references/p-video-2-prompting.md (3246b), references/p-video-2-quality-checklist.md (2145b), references/p-video-animate-prompting.md (2230b), references/p-video-animate-quality-checklist.md (1940b), references/p-video-avatar-prompting.md (2733b), references/p-video-avatar-quality-checklist.md (2586b), references/p-video-edit-prompting.md (3745b), references/p-video-edit-quality-checklist.md (2068b), references/p-video-quality-checklist.md (2507b), references/p-video-replace-prompting.md (3103b), references/p-video-replace-quality-checklist.md (3852b), references/physics-safe-motion.md (2859b), references/prompt-dramaturgy.md (3959b), references/scene-anchor-pair.md (10723b), references/scene-anchor-triple.md (7408b), skill-card.md (2971b), skill.manifest.json (653b), SKILL.md (7368b), _meta.json (135b)\n\nFile v1.0.12:SKILL.md\n\n---\nname: video-prompting\ndescription: Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining.\nlicense: MIT\nmetadata:\n  version: \"1.0.12\"\n  package: pruna-skills\n---\n\n# Video prompting\n\nVendor-neutral craft for **short video / motion** generation. Works with Pruna `p-video-2` / `p-video` family, Runway, Kling, Luma, Veo, and similar APIs.\n\n## Install\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n\n## When to use\n\n- Text-to-video or image-to-video prompts\n- Start/end frame (anchor pair) or narrated beat (anchor triple) specs\n- Camera and lighting vocabulary in motion lines\n- Physics-safe subject motion\n- Multi-clip continuity / clip chaining\n- Talking-head, motion-transfer, slot-replace, or instruction-based video-edit prompts\n\n## Works with\n\nPruna `p-video-2` / `p-video` / `p-video-avatar` / `p-video-animate` / `p-video-replace` / `p-video-edit`, Runway Gen-3, Kling, Luma Dream Machine, Veo, and other video models. Best quality: `p-video-2`. Simpler clips: `p-video`.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `image-prompting` | Use when crafting still-image prompts for any generative model — composition, identity sheets, edits, try-on, and photoreal personas. | `npx skills add PrunaAI/pruna-skills@image-prompting -y` |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `music-video` | Use when someone wants a full music video — original song or vocals, performance clips, B-roll, and lyric-synced edits. | `npx skills add PrunaAI/pruna-skills@music-video -y` |\n| `narrated-multi-scene` | Use when someone wants a multi-part story with voiceover — episodic B-roll, chaptered promo, or several linked video scenes without on-camera dialogue. | `npx skills add PrunaAI/pruna-skills@narrated-multi-scene -y` |\n| `p-video-2` | Use when someone wants the best-quality short clip from text, images, or audio — polished B-roll, start/end frame animation, or a motion shot with stronger lip-sync. Not for full multi-scene films or talking-head-only hosts. | `npx skills add PrunaAI/pruna-skills@p-video-2 -y` |\n| `p-video` | Use when someone wants a simple short clip from text or images — quick B-roll, drafts, or start/end frame animation. Not when the brief needs the highest quality or tight lip-sync. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\n## Guide habit\n\nIn the **first reply**, name `` `video-prompting` `` in backticks. When aspect, resolution, duration, or embed-vs-post audio are open, open intake → **`generation-diversity`** clarification intake. For `p-video-2` / `p-video` motion lines, cite OPEN/MID/CLOSE dramaturgy and **Worked example — product B-roll** in [prompt-dramaturgy.md](./references/prompt-dramaturgy.md). Quality path: `p-video-2` (`p-video-2-prompting`). Simpler clips: `p-video`. Audio-led clips: **≤ ~19s** TTS before embed — see [audio-in-video-prompting.md](./references/audio-in-video-prompting.md).\n\n## Before generating\n\n1. Follow `generation-diversity` first.\n2. Read in order:\n   - [prompt-dramaturgy.md](./references/prompt-dramaturgy.md) — Details Law, OPEN/MID/CLOSE\n   - [camera-lighting-vocabulary.md](./references/camera-lighting-vocabulary.md)\n   - [physics-safe-motion.md](./references/physics-safe-motion.md)\n   - [audio-in-video-prompting.md](./references/audio-in-video-prompting.md) when sound matters\n   - [clip-chaining.md](./references/clip-chaining.md) for multi-clip continuity\n   - [scene-anchor-pair.md](./references/scene-anchor-pair.md) / [scene-anchor-triple.md](./references/scene-anchor-triple.md) for frame (+ audio) payloads\n3. Tool-specific craft when needed:\n   - [p-video-2-prompting.md](./references/p-video-2-prompting.md)\n   - [p-video-avatar-prompting.md](./references/p-video-avatar-prompting.md)\n   - [p-video-animate-prompting.md](./references/p-video-animate-prompting.md)\n   - [p-video-replace-prompting.md](./references/p-video-replace-prompting.md)\n   - [p-video-edit-prompting.md](./references/p-video-edit-prompting.md)\n4. Validate with the matching `*-quality-checklist.md` in `./references/`.\n\nProduct B-roll and OPEN/MID/CLOSE samples: **Worked example — product B-roll** in [prompt-dramaturgy.md](./references/prompt-dramaturgy.md).\n\n## Pruna tools\n\nMatching install for every model named above. Pick what you need:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `p-video-2` | Use when someone wants the best-quality short clip from text, images, or audio — polished B-roll, start/end frame animation, or a motion shot with stronger lip-sync. Not for full multi-scene films or talking-head-only hosts. | `npx skills add PrunaAI/pruna-skills@p-video-2 -y` |\n| `p-video` | Use when someone wants a simple short clip from text or images — quick B-roll, drafts, or start/end frame animation. Not when the brief needs the highest quality or tight lip-sync. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n| `p-video-avatar` | Use when someone wants a person on camera speaking a script — lip-synced host, spokesperson, or narrated avatar from a portrait photo. | `npx skills add PrunaAI/pruna-skills@p-video-avatar -y` |\n| `p-video-animate` | Use when someone wants a photo to move like another video — motion transfer, dance remixes, or performance variations from a template clip. | `npx skills add PrunaAI/pruna-skills@p-video-animate -y` |\n| `p-video-replace` | Use when someone wants to swap a person, outfit, or product inside existing footage while keeping the camera move and audio. | `npx skills add PrunaAI/pruna-skills@p-video-replace -y` |\n| `p-video-edit` | Use when someone wants to edit an existing video with a text instruction — recolor, restyle, remove or add objects, change environment or lighting, update on-screen text, or apply optional reference-guided product and accessory edits. Not for a new clip from scratch or ffmpeg assembly. | `npx skills add PrunaAI/pruna-skills@p-video-edit -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFile v1.0.12:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"video-prompting\",\n  \"version\": \"1.0.12\",\n  \"publishedAt\": 1789048239926\n}\n\nFile v1.0.12:references/audio-in-video-prompting.md\n\n# Audio-in-video prompting (`p-video`)\n\nHow to **write prompts** when sound matters on `p-video`. Layering / tool picker: `audio-prompting`. Talking heads: [p-video-avatar-prompting.md](./p-video-avatar-prompting.md).\n\n## Three modes\n\n| Mode | API | Prompt job |\n|------|-----|------------|\n| **A — Native SFX / dialogue** | `prompt` + optional `save_audio`; use `duration` | Name **diegetic** sounds the picture should emit |\n| **B — Uploaded audio (preferred for VO/music)** | `audio` URL; **omit `duration`**; `save_audio: true` | Motion matches **mood/beats** of the track — do not paste VO text into `prompt` |\n| **C — Post bed** | Stable Audio mixed under VO in ffmpeg | Bed prompt is separate (`stable-audio-2.5` + `audio-prompting`); video prompt ignores the bed |\n\nNever generate silent `p-video` and post-mux narration unless re-render is impossible (truncation risk). Probe TTS ≤ ~19s before Mode B.\n\n## Mode A — native SFX / dialogue\n\n**When:** `save_audio: true`, no uploaded `audio`, user wants diegetic SFX and/or spoken lines in the clip.\n\n### Diegetic SFX\n\nBe concrete; skip `cinematic soundscape`:\n\n| Bad | Good |\n|-----|------|\n| `epic soundtrack vibe` | `rain ticks on the awning, distant train horn once` |\n| `dramatic music` | `crowd murmur swells, single glass clink` |\n\nKeep cues short — the model invents audio from visual+prompt context when `save_audio` is on.\n\n### Native dialogue\n\nPut **exact spoken words in double quotes** inside the motion `prompt` (Mode A only). Placeholders in square brackets below are docs notation, not literal syntax — do not confuse with Gemini TTS `[tags]`.\n\n**Template** (inside OPEN/MID/CLOSE):\n\n```text\nMID: [same subject] says \"[LINE]\" — mouth open, [one gesture toward target]; [optional diegetic SFX cue]\n```\n\nRules:\n\n- Exact spoken words in **double quotes** — never paraphrase the line\n- Name **who** speaks — repeat the subject label for continuity (`same [role]`)\n- Pair every line with **mouth state** + **one gesture** (point, turn to camera, hand on prop)\n- Keep lines **short** (1–2 per beat)\n- Diegetic SFX can sit beside dialogue — stay concrete (table above)\n- **Mode B:** when `input.audio` is set, do **not** put the VO transcript in `prompt` — motion matches mood only\n\n## Mode B — motion matches uploaded audio\n\n```text\nOPEN: hold wide on dog in tall grass, warm afternoon light.\nMID: gentle push-in as he searches; tail motion matches narrator energy; grass sways.\nCLOSE: settle on end pose — curious head tilt.\n```\n\nRules:\n\n- Describe **picture motion**, not the spoken words.  \n- Match energy: tense VO → tighter push-in; calm story → slow drift.  \n- Optional: `motion matches narrator mood` once — not a transcript.  \n- Triple anchors: [scene-anchor-triple.md](./scene-anchor-triple.md).\n\n## Avatar path (not Mode B fields)\n\n`p-video-avatar` uses `voice_script` + `voice_prompt` + `video_prompt` — not `p-video` `input.audio` for the spoken line. See avatar prompting ref. Do not paste script lines into `voice_prompt`.\n\n## Pre-send\n\n- [ ] Mode A / B / C chosen  \n- [ ] Mode B: `duration` omitted; TTS length probed  \n- [ ] No VO transcript inside motion `prompt` (Mode B)  \n- [ ] Diegetic cues concrete (Mode A) or mood-aligned (Mode B)\n- [ ] Mode A dialogue (if any): quoted line + named speaker + mouth + one gesture\n\nFile v1.0.12:references/camera-lighting-vocabulary.md\n\n# Camera and lighting vocabulary\n\nShared lexicon for `camera_tag` / `lighting_tag` (stills) and motion lines in `p-video` / `p-video-avatar` prompts. Diversity axes: generation-diversity.md#visual-variety (`generation-diversity`). Dramaturgy: [prompt-dramaturgy.md](./prompt-dramaturgy.md).\n\n**Sources:** patterns adapted from [smixs/visual-skills](https://github.com/smixs/visual-skills) and [inference-sh/skills](https://github.com/inference-sh/skills) (MIT); rewritten for Pruna.\n\n## Framing ladder\n\n| Term | Use when |\n|------|----------|\n| ECU (extreme close-up) | Eyes, hands, product detail |\n| CU (close-up) | Face, emotion |\n| MCU (medium close-up) | Talking head default |\n| MS (medium shot) | Waist-up action |\n| MLS / FS | Full body travel |\n| WS / EWS | Environment as character |\n\nLog as `camera_tag`, e.g. `medium close-up, slight low angle`.\n\n## Lens roles (optional but sharp)\n\n| Lens | Feel |\n|------|------|\n| 24mm | Wide, immersive, exaggerated space |\n| 35mm | Documentary natural |\n| 50mm | Intimate human perspective |\n| 85mm | Portrait, compressed background |\n| Macro | Texture, product detail |\n\nExample: `shot on 50mm, eye-level`.\n\n## Camera moves (pick one for MID)\n\n| Move | Prompt cue |\n|------|------------|\n| Dolly / push-in | `slow dolly in`, `gentle push-in` |\n| Dolly out | `slow pull back revealing the room` |\n| Pan | `gentle pan left across the alley` |\n| Tilt | `tilt up from hands to face` |\n| Track / truck | `shoulder-height tracking shot beside the subject` |\n| Crane | `slow crane down past neon signs` |\n| Static + atmosphere | `locked camera, steam rises, light shifts` |\n| Handheld | `subtle handheld drift` (use sparingly) |\n\nAvoid whip pans and stacked contradictory moves in one short clip.\n\n## Motivated lighting\n\nPrefer **named sources** over “beautiful lighting”:\n\n| Source | Example cue |\n|--------|-------------|\n| Window / dawn | `dawn light spreads across the desk` |\n| Practical | `warm lamp spill, cool window fill` |\n| Neon / gel | `magenta-cyan neon rim, wet reflections` |\n| Overhead institutional | `cold fluorescent flicker` |\n| Fire / candle | `candle flicker on faces` (avatar: often too transition-y — prefer steady) |\n| Overcast soft | `soft overcast skylight, low contrast` |\n\nLog as `lighting_tag`. Hex in stills when brand colors matter (`#0d3d2d rim`).\n\n## Palette cues (one look)\n\nPick one coherent palette phrase: `teal-magenta night`, `warm tungsten interior`, `bleached noon desert`, `desaturated documentary`.\n\nDo not stack competing genre looks in one prompt.\n\n## Avatar-friendly defaults\n\nTalking heads: **MCU**, one slow push-in or static, **steady light**, mouth visible. Variety across scenes = change angle/background still, not five camera moves mid-line. See [p-video-avatar-prompting.md](./p-video-avatar-prompting.md).\n\nFile v1.0.12:references/clip-chaining.md\n\n# Clip chaining (multi-scene video)\n\nWhen and how to continue motion across `p-video` clips. Plan JSON examples stay in [scene-anchor-pair.md](./scene-anchor-pair.md) and [scene-anchor-triple.md](./scene-anchor-triple.md); this page is the decision tree + prompt rules.\n\nWorkflows: `visual-transition-reel` · `narrated-multi-scene`.\n\n## Decision tree\n\n```text\nDoes motion continue in the same place/moment (no time jump)?\n  NO  → chain_from_previous: false — hard cut; compose a new OPENING still\n  YES → chain_from_previous: true\n        Prefer frame_chain_mode: extract_last_frame (sequential renders)\n        Only use planned_stills if you accept possible cut jumps\n```\n\n| Situation | `chain_from_previous` | Join |\n|-----------|----------------------|------|\n| Continuous action (run → leap) | `true` | Short crossfade ~0.12–0.15s after extract |\n| New beat / location / pause | `false` | Hard cut (0 crossfade) |\n| First scene | `false` | — |\n| Montage vignettes (no shared motion) | `false` + `parallel_vignettes` | Hard cuts; parallel renders OK |\n\n| `frame_chain_mode` | Next scene `image` | Render order |\n|--------------------|--------------------|--------------|\n| **`extract_last_frame`** | ffmpeg last frame of prior clip | **Sequential** when any scene chains |\n| **`parallel_vignettes`** | each scene’s own start still | **Parallel** |\n| **`planned_stills`** | prior scene end still URL | Parallel once stills exist — higher jump risk |\n\n**Why extract?** Planned end stills often differ from the model’s actual last frame → visible jump.\n\n## Prompt rules for chained beats\n\n1. **Same subject language** — repeat “same [character]” in OPEN/MID/CLOSE.  \n2. **No teleport** — ban `cut to`, `suddenly in`, `walls disappear`; use `gradually`, `walks through`, `ease into`.  \n3. **Match lighting era** — chained clips share `style_bible` and time-of-day.  \n4. **Exit / enter continuity** — if scene 1 CLOSE faces right, scene 2 OPEN should not hard-flip screen direction without a motivated turn.  \n5. **Hard-cut scenes** — treat as fresh OPENING; do not assume prior pose.\n\n## Assembly notes\n\n1. Concat in scene order (ffmpeg concat — see the workflow skill).  \n2. Per-join `crossfades`: chain ~0.12–0.15s; hard cuts 0.  \n3. Normalize audio (48 kHz stereo) when mixing formats.  \n4. Optional bed under native SFX — `audio-prompting`.\n\n## Intake checklist\n\n- [ ] Each scene: chain flag only if motion truly continues  \n- [ ] `frame_chain_mode` chosen  \n- [ ] Chained prompts pass Details Law ([prompt-dramaturgy.md](./prompt-dramaturgy.md))  \n- [ ] Physics tier OK ([physics-safe-motion.md](./physics-safe-motion.md))\n\nFile v1.0.12:references/p-video-2-prompting.md\n\n# p-video-2 prompting\n\nPrompt craft unique to `p-video-2` (quality-focused successor to `p-video`). Shared dramaturgy, camera, and physics: [prompt-dramaturgy.md](./prompt-dramaturgy.md), [camera-lighting-vocabulary.md](./camera-lighting-vocabulary.md), [physics-safe-motion.md](./physics-safe-motion.md). Frame payloads: [scene-anchor-pair.md](./scene-anchor-pair.md) / [scene-anchor-triple.md](./scene-anchor-triple.md). QA: [p-video-2-quality-checklist.md](./p-video-2-quality-checklist.md).\n\n**Quality path.** Use `p-video-2` when the brief needs the best output. Use `p-video` for simpler, quicker clips.\n\n## Strengths to write toward\n\n- Sharper subjects, backgrounds, and motion than `p-video` — especially **close-ups and foreground objects**\n- Stronger **lip-sync on native speech** — T2V with `save_audio: true` and **no imported `audio`**: the model pauses, delivers the line, then rests, and the mouth follows. **Lead dialogue / lip-sync jobs here**\n- **Native audio** in the output (`save_audio` defaults true). Imported `audio` still works and keeps a **cleaner face / identity**; viseme lock vs `p-video` is mixed — do not lead lip-sync demos with a muxed wav\n- Stronger **identity / input-image consistency**\n- One endpoint: T2V + I2V + audio-conditioned\n\n## Limits (do not fight them)\n\n- Not designed for **extreme cinematic camera** (crash zooms, whip pans, chaotic handheld)\n- Complex **multi-scene storytelling** in one clip is weaker — split beats\n- Native **4K** is not supported\n- **More than two speakers** — speaker separation degrades\n- **SFX-led** clips (the ask is sound effects, not picture + optional bed) are currently limited\n\n## Mode recipes (rewrite for the brief — do not paste)\n\n**Text-to-video** — subject + motion + camera + lighting + audio intent in one coherent line.\n\n```text\nA sports car drifting through a neon-lit city at night, cinematic aerial shot, wet asphalt reflections, engine roar and tire screech.\n```\n\n**Image-to-video** — keep motion **subtle** and aligned with the reference frame. Prefer a **stable** camera.\n\n```text\nThe camera slowly pushes in. The person turns their head and smiles naturally. Soft studio lighting, shallow depth of field.\n```\n\n**Native speech (T2V, no `audio`)** — write the spoken line into the prompt. This is the lip-sync path.\n\n```text\nA woman faces the camera and says \"I'll be there in five.\" She pauses, delivers the line, then rests. Natural mouth motion, soft window light, camera holds steady.\n```\n\n**Audio-conditioned** — name the performer and hold. One or two faces max. Use for singing / VO length, not as the primary lip-sync demo.\n\n```text\nClose-up of a singer performing the uploaded track. Natural lip-sync, expressive face, stage lighting, camera holds steady on the performer.\n```\n\n## Duration\n\n- Set `duration` (1–20s) when the user locked a length.\n- **Leave `duration` empty** to let the model choose length from the prompt.\n- When `audio` is set, omit `duration` — length follows the audio (cap **20s**; TTS ≤ ~19s).\n\n## Iterate\n\nUse `draft: true` for cheap previews, then `draft: false` for the paid final. Defaults: `prompt_upsampling: true`, `save_audio: true`, `resolution: 720p`, `fps: 24`, `aspect_ratio: 16:9`.\n\nFile v1.0.12:references/p-video-2-quality-checklist.md\n\n# p-video-2 quality checklist\n\nAfter each `p-video-2` output is saved, **open the clip and review it visually** (and listen when audio is expected) against this checklist (agent vision review — see `generation-diversity`).\n\nShared motion / frame-anchor items: also run [p-video-quality-checklist.md](./p-video-quality-checklist.md).\n\n## Applies to\n\nSee the canonical mapping in `generation-diversity`. Quality path for `image-to-video`, `narrated-multi-scene`, `visual-transition-reel`, and B-roll rows. Use `p-video` for simpler / quicker clips.\n\n## Quality vs `p-video`\n\n- Subjects, backgrounds, and motion look sharper than a typical `p-video` draft of the same brief.\n- Close-up / foreground objects stay readable (hands, product, face).\n- Input-image identity holds when `image` was set.\n\n## Lip-sync and speakers\n\nWhen the prompt or `audio` implies speech or singing:\n\n- **Native speech (no imported `audio`):** mouth follows pause / line / rest — this is the quality bar vs `p-video`.\n- **Imported `audio`:** face/identity stays clean; do not fail the job solely because visemes are not tighter than `p-video`.\n- At most **two** speaking faces stay separable; reject crowded dialogue with 3+ talkers.\n- Talking-head-only jobs with no native scene audio should have used `p-video-avatar` instead.\n\n## Native audio\n\n- When `save_audio` is true (default), the file is not unintentionally silent.\n- Uploaded `audio` is embedded and not truncated (TTS was ≤ ~19s; API cap 20s).\n- Do not treat weak SFX as a pass if the brief was sound-effect-led — warn and offer a bed in post (`audio-prompting`).\n\n## Camera and length\n\n- Camera move is stable (slow push / hold / one dolly) — not an extreme cinematic stunt.\n- If `duration` was omitted (no audio), runtime matches the prompt’s implied beat.\n- If `duration` was set, runtime is in the 1–20s window and matches the request.\n- Output is 720p or 1080p as requested — not 4K.\n\n## Scene anchors\n\nWhen using pair or triple payloads, apply the pair/triple sections in [p-video-quality-checklist.md](./p-video-quality-checklist.md), substituting `p-video-2` for the prediction model.\n\nFile v1.0.12:references/p-video-animate-prompting.md\n\n# p-video-animate prompting\n\nMotion-transfer craft for `p-video-animate`. QA: [p-video-animate-quality-checklist.md](./p-video-animate-quality-checklist.md). Mixed reels: `avatar-multi-scene`.\n\n**Appearance from `image`, motion from `video`.** Wrong tool for identity swap on real footage → [p-video-replace-prompting.md](./p-video-replace-prompting.md).\n\n## Pairing gates (before every job)\n\nAsk:\n\n1. **Framing** — same body region (head-and-shoulders / medium / full body)?  \n2. **Pose** — facing and limb position roughly aligned with the template’s first frame?  \n3. **Visibility** — same body parts visible; no crop the video lacks?\n\n| Factor | Guidance |\n|--------|----------|\n| Shot size | Match close-up / medium / full |\n| Facing | Front still + profile motion → artifacts |\n| Limbs | If template waves arms, still must show arms |\n| Proportions | Human full-body dance on chibi often breaks gait |\n| Speaking templates | Mouth clear and large when source has dialogue |\n\n**Pairing failure:** head-and-shoulders still + full-body dance → model does **not** invent limbs; choreography is lost. Repose with `p-image-edit` or pick a closer template.\n\n## `instruction_prompt`\n\nOptional. Overrides **behavior**, not identity. **Leave blank** when source motion is already right.\n\n**Useful** — one specific end beat:\n\n```text\nAt the very end of the clip, just after her last gesture, she gives a clear thumbs-up toward the camera. Keep the source motion otherwise.\n```\n\n**Less useful** — redescribes the still:\n\n```text\nA confident woman in a charcoal blazer speaks to the camera in a modern office.\n```\n\n## Style variety\n\nPhotoreal, cartoon, 3D, and mascot stills can share one template when framing aligns — the still’s render style carries through.\n\n## Speaking motion sources\n\nWhen this template feeds lip-sync showcases, the source clip (often from `p-video-avatar`) must show clear speaking / lip movement. See [p-video-avatar-prompting.md](./p-video-avatar-prompting.md) and animate-beats.\n\n## Pre-send\n\n- [ ] Framing / pose / limbs match  \n- [ ] `instruction_prompt` blank or one concrete beat  \n- [ ] Not using animate for in-place replace  \n- [ ] Long templates split (~5s compute per 1s video)\n\nFile v1.0.12:references/p-video-animate-quality-checklist.md\n\n# p-video-animate quality checklist\n\nAfter each animate job, **open the source video, reference image, and output clip** and review them against this list (agent vision review — see `generation-diversity`).\n\n## Applies to\n\nSee the canonical mapping in `generation-diversity`.\n\n## Input gate (pre-render)\n\n- Source **`video`** is the intended motion template (camera path, acting, timing, scene structure); motion is **clear and readable** (not blurry, fast-cut, or low-contrast).\n- Reference **`image`** clearly shows the subject to animate (face/body unobstructed, rights cleared).\n- **First-frame alignment:** framing, pose, and visible limbs match the **first frame** of the source video (or repose with **`p-image-edit`** first).\n- **Mismatch risk:** head-and-shoulders still + full-body template → expect lost choreography, not full-body motion.\n- **Proportion fit:** human full-body motion on meme/mascot/chibi subjects often breaks legs, arms, and contact points—flag before generate.\n- **`instruction_prompt`** (if used) describes **behavior overrides only** — not a repeat of the image description.\n- **`resolution`** and **`target_fps`** match delivery spec.\n- Source longer than budget: plan to **split** the template and animate segments (~5 s compute per 1 s video).\n\n## Motion transfer fidelity\n\n- Output preserves source motion, timing, and camera movement (not a generic re-enactment).\n- Acting beats and scene structure track the reference video.\n- Subject identity and style come from the reference image, not the source video's actor.\n\n## Technical quality\n\n- No severe flicker, warping, or unstable anatomy during motion.\n- Audio (when `save_audio` is true) stays aligned with visual motion.\n- Output duration matches the source video length.\n\n## Clean delivery\n\n- No accidental overlays, stray text, or watermark-like artifacts unless requested.\n- Clip is ready for downstream edit, concat, or platform upload.\n\nFile v1.0.12:references/p-video-avatar-prompting.md\n\n# p-video-avatar prompting\n\nTalking-head prompt craft for `p-video-avatar`. Templates: `avatar-multi-scene`. Camera: [camera-lighting-vocabulary.md](./camera-lighting-vocabulary.md). Physics: [physics-safe-motion.md](./physics-safe-motion.md).\n\n**Do not** use OPEN/MID/CLOSE — the model treats beats as cuts and the clip feels cutty.\n\n## Three-layer stack\n\n| Layer | Field | Job |\n|-------|-------|-----|\n| 1. Plate | `image` | Locked approved still — mouth visible; quality caps the avatar |\n| 2. Voice | `voice_script` + `voice_prompt` | What they say + how they sound |\n| 3. Motion | `video_prompt` | Unique camera/gesture per clip |\n\nNever ship multi-scene reels where every row reuses `medium close-up, gentle dolly push-in`.\n\n## Field hygiene\n\n| Field | Write | Never |\n|-------|-------|-------|\n| **`voice_script`** | Natural spoken lines | Brochure / slogan paste as the only line without human rhythm |\n| **`voice_prompt`** | Short delivery: pacing, warmth, archetype | Product names, script lines, long scene descriptions |\n| **`video_prompt`** | MCU, one slow push-in or static, speaks to camera | OPEN/MID/CLOSE; walk across room; hold up documents; wild gestures |\n\nStylized hosts: match energy to medium (anime slightly more expressive; documentary restrained). Separate hero stills per `visual_style_tag`.\n\n## Micro-actions (talking-head Details Law)\n\nPrefer face/eye micro-moves over locomotion:\n\n- eyes lift to lens, slight nod, natural blink rhythm, subtle lean  \n- one small hand-to-chest max  \n\nAvoid physics traps while speaking ([physics-safe-motion.md](./physics-safe-motion.md) avatar subsection).\n\n## Negative prompt (experimental)\n\nAPI `negative_prompt` = **noun suppression list** (subtitles, captions, watermark…), not creative wording. Strength ~0.3–0.4. Primary fix: positive-only stills (`plain unmarked walls`). See SKILL for defaults.\n\n## Good / bad triples\n\n**Good**\n\n```text\nvoice_script: \"So we tried something weird last quarter — and it actually worked.\"\nvoice_prompt: Natural conversational tone, relaxed pacing, real pauses, honest not salesy.\nvideo_prompt: Medium close-up speaking directly to lens, one very slow push-in, steady light, natural head motion, no cuts.\n```\n\n**Bad**\n\n```text\nvoice_prompt: Mention Pruna and our 10x faster inference in an exciting cinematic way.\nvideo_prompt: OPEN: hold. MID: she walks across the room waving a laptop. CLOSE: product hero shot.\n```\n\n## Pre-send\n\n- [ ] Plate approved; mouth visible  \n- [ ] `voice` locked per character across scenes  \n- [ ] Unique `video_prompt` per clip  \n- [ ] No OPEN/MID/CLOSE  \n- [ ] `voice_prompt` has no script/product paste  \n\nQA: [p-video-avatar-quality-checklist.md](./p-video-avatar-quality-checklist.md).\n\nFile v1.0.12:references/p-video-avatar-quality-checklist.md\n\n# p-video-avatar quality checklist\n\nBefore calling the model and after each avatar clip, **open the still or video and review it visually** against this checklist (agent vision review — see `generation-diversity`).\n\n## Applies to\n\nSee the canonical mapping in `generation-diversity`.\n\n## Input still gate (pre-render)\n\n- Face and mouth/beak are large and clear enough for lip-sync.\n- Mouth/beak and eyes are unobstructed (no hair/props/foreground clutter crossing them).\n- Head pose is speaking-friendly (avoid extreme angles, tiny head crop, or chin cutoff).\n- Identity/style match cast bible and scene continuity.\n- **Photoreal path:** skin reads natural (not mushy/waxy); plate matches `image-prompting` intent.\n- **Try-on → avatar path:** try-on preservation passed; outfit details visible if script references them.\n\n## Speech and performance\n\n- Spoken output matches intended script/audio content.\n- Voice choice is consistent for recurring characters.\n- Delivery tone matches brief; `voice_prompt` is short and does not leak unintended text.\n- **`voice_script`** reads as speakable human dialogue — not brochure/marketing copy.\n\n## Motion and scene dynamism\n\n- **`video_prompt`** is **unique to this clip** — not duplicated from other scenes in the same project.\n- Motion grammar matches the still (props, setting, angle) — e.g. glance targets exist in plate.\n- Multi-scene reels vary camera angle and movement — not every clip `medium close-up, gentle dolly push-in`.\n- **Stylized clips:** motion energy matches `visual_style_tag` (anime vs documentary vs clay).\n- **Motion templates:** when the clip is a source for `p-video-animate`, verify audible speech and visible lip sync — reject smile/wave-only outputs with no dialogue motion.\n\n## Lip-sync and visual stability\n\n- Mouth movement is plausible and synchronized.\n- No facial warping, jitter, or unstable eye/teeth regions.\n- Hands/props near face do not cause ambiguous anatomy artifacts.\n\n## Clean delivery\n\n- `video_prompt` results in clean framing/motion without prompt side effects.\n- No accidental overlays, stray text, or watermark-like artifacts unless requested.\n- For explainers / any text-prone still: `negative_prompt` + `negative_prompt_strength` > 0 on `p-video-avatar` (runner default or plan `defaults.avatar_negative_*`). Tune strength up only if artifacts persist — high values can harm identity/motion.\n- Still lines stayed free of signage/label triggers; `style_bible` holds negations, not `edit_prompt`.\n- Clip is ready for assembly with consistent style/voice across adjacent scenes.\n\nFile v1.0.12:references/p-video-edit-prompting.md\n\n# p-video-edit prompting\n\nSurgical edit prompts for `p-video-edit`. QA: [p-video-edit-quality-checklist.md](./p-video-edit-quality-checklist.md). Identity/slot swap on footage → [p-video-replace-prompting.md](./p-video-replace-prompting.md).\n\n**`prompt`** is the edit instruction. Optional **`images`** (0–4) guide product, accessory, or style when the brief names a reference.\n\n## Core formula\n\nOne principal change per run. Describe the desired final state, then name what must stay unchanged.\n\n```text\nChange only [specific thing]. Preserve [geometry / motion / camera / lighting / unmentioned subjects].\n```\n\n| Do | Don't |\n|----|-------|\n| `Change only the jacket from blue to red. Preserve its cut, folds and motion.` | `Make it cinematic` |\n| Name the source element (`SUV body paint`, `text overlay \"Miam\"`, `plants in the corner`) | Vague `edit the video` |\n| Point at refs: `Add the cargo box from the first reference to the roof.` | `Use the reference` with no slot |\n| Keep-list for camera, motion, and subjects you are not changing | Hope the model leaves them alone |\n\n## Decide edit intent first\n\n| Intent | Prompt must |\n|--------|-------------|\n| **Attribute** | Name the property (color, material, shade) and keep cut/geometry/motion |\n| **Remove** | Name only the object to delete; reconstruct the occluded surface |\n| **Add** | Name what to add and where it attaches; keep the rest |\n| **Environment** | Change setting or walls; preserve subject, furniture, camera path |\n| **Relight** | Change atmosphere/palette only; keep faces, products, blocking |\n| **Text** | Quote the exact string to add, replace, or remove |\n| **Reference-guided** | Map each image to a source slot (`first reference` → roof box / sunglasses) |\n\n## Weak jobs (warn before pay)\n\n- A brand-new scene, plot, or video rather than an edit of this clip\n- Adding an object with its own independent motion — especially a new in-hand shape\n- Changing camera angle, camera motion, or zoom\n\nSplit those into a new `p-video-2` generation, or keep the camera locked and edit only appearance.\n\n## Patterns (rewrite for the brief — do not paste as the user's prompt)\n\n**Attribute**\n\n```text\nChange only the SUV body paint to deep metallic red. Preserve the vehicle geometry, glass, trim, headlights, tires, reflections and shadows. Keep the environment, lighting and camera movement unchanged.\n```\n\n**Remove**\n\n```text\nRemove only the potted plant from the countertop. Reconstruct the countertop and wall naturally. Keep the person, bottle, camera and lighting unchanged.\n```\n\n**Add (optional reference)**\n\n```text\nAdd the roof cargo box shown in the first reference to the SUV. Match its shape, matte-black material and proportions. Keep it rigidly attached and correctly aligned to the vehicle roof throughout the camera movement. Preserve the vehicle body, windows, trim, wheels, environment and lighting.\n```\n\n**Environment**\n\n```text\nChange only the room wall from beige plaster to deep sage-green plaster. Preserve the room geometry and shadows. Keep the furniture and camera motion unchanged.\n```\n\n**Text**\n\n```text\nRemove only the text \"Miam\". Reconstruct the underlying image naturally where the text was located.\n```\n\n## Reference stills\n\nUse when the user supplied a product, accessory, or SKU to match. Bare packshots — no extra hands or scene props. Index refs in `prompt` (`first reference`, `image 2`). Max 4 images (`jpg`, `jpeg`, `png`, `webp`).\n\n## Pre-send\n\n- [ ] Edit intent chosen (one principal change)\n- [ ] Source element named + keep-list present\n- [ ] Source `video` ≤ 15 seconds\n- [ ] Refs (if any) indexed and ≤ 4\n- [ ] Not a new scene, independent in-hand motion, or camera rewrite\n- [ ] `draft` / `save_audio` decided\n\nArchive v1.0.11: 20 files, 34666 bytes\n\nFiles: references/audio-in-video-prompting.md (3342b), references/camera-lighting-vocabulary.md (2812b), references/clip-chaining.md (2668b), references/p-video-animate-prompting.md (2230b), references/p-video-animate-quality-checklist.md (1940b), references/p-video-avatar-prompting.md (2733b), references/p-video-avatar-quality-checklist.md (2586b), references/p-video-edit-prompting.md (3743b), references/p-video-edit-quality-checklist.md (2068b), references/p-video-quality-checklist.md (2507b), references/p-video-replace-prompting.md (3103b), references/p-video-replace-quality-checklist.md (3852b), references/physics-safe-motion.md (2859b), references/prompt-dramaturgy.md (3959b), references/scene-anchor-pair.md (10563b), references/scene-anchor-triple.md (7326b), skill-card.md (3245b), skill.manifest.json (585b), SKILL.md (6525b), _meta.json (135b)\n\nFile v1.0.11:SKILL.md\n\n---\nname: video-prompting\ndescription: Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining.\nlicense: MIT\nmetadata:\n  version: \"1.0.11\"\n  package: pruna-skills\n---\n\n# Video prompting\n\nVendor-neutral craft for **short video / motion** generation. Works with Pruna `p-video` family, Runway, Kling, Luma, Veo, and similar APIs.\n\n## Install\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n\n## When to use\n\n- Text-to-video or image-to-video prompts\n- Start/end frame (anchor pair) or narrated beat (anchor triple) specs\n- Camera and lighting vocabulary in motion lines\n- Physics-safe subject motion\n- Multi-clip continuity / clip chaining\n- Talking-head, motion-transfer, slot-replace, or instruction-based video-edit prompts\n\n## Works with\n\nPruna `p-video` / `p-video-avatar` / `p-video-animate` / `p-video-replace` / `p-video-edit`, Runway Gen-3, Kling, Luma Dream Machine, Veo, and other video models.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `image-prompting` | Use when crafting still-image prompts for any generative model — composition, identity sheets, edits, try-on, and photoreal personas. | `npx skills add PrunaAI/pruna-skills@image-prompting -y` |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `music-video` | Use when someone wants a full music video — original song or vocals, performance clips, B-roll, and lyric-synced edits. | `npx skills add PrunaAI/pruna-skills@music-video -y` |\n| `narrated-multi-scene` | Use when someone wants a multi-part story with voiceover — episodic B-roll, chaptered promo, or several linked video scenes without on-camera dialogue. | `npx skills add PrunaAI/pruna-skills@narrated-multi-scene -y` |\n| `p-video` | Use when someone wants one short video clip from text or images — B-roll, start/end frame animation, or a quick motion shot. Not for full multi-scene films or lip-synced hosts. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n| `video-editing` | Use when assembling or polishing already-rendered clips with ffmpeg — concat, crossfades, burned captions and subtitles, text/logo overlays, before/after sliders, background music beds, platform export — or when composing a multi-layer HTML combination video with Hyperframes. Not for AI video generation, prompt craft, or model-based video edits. | `npx skills add PrunaAI/pruna-skills@video-editing -y` |\n\n## Guide habit\n\nIn the **first reply**, name `` `video-prompting` `` in backticks. When aspect, resolution, duration, or embed-vs-post audio are open, open intake → **`generation-diversity`** clarification intake. For `p-video` motion lines, cite OPEN/MID/CLOSE dramaturgy and **Worked example — product B-roll** in [prompt-dramaturgy.md](./references/prompt-dramaturgy.md). Audio-led clips: **≤ ~19s** TTS before embed — see [audio-in-video-prompting.md](./references/audio-in-video-prompting.md).\n\n## Before generating\n\n1. Follow `generation-diversity` first.\n2. Read in order:\n   - [prompt-dramaturgy.md](./references/prompt-dramaturgy.md) — Details Law, OPEN/MID/CLOSE\n   - [camera-lighting-vocabulary.md](./references/camera-lighting-vocabulary.md)\n   - [physics-safe-motion.md](./references/physics-safe-motion.md)\n   - [audio-in-video-prompting.md](./references/audio-in-video-prompting.md) when sound matters\n   - [clip-chaining.md](./references/clip-chaining.md) for multi-clip continuity\n   - [scene-anchor-pair.md](./references/scene-anchor-pair.md) / [scene-anchor-triple.md](./references/scene-anchor-triple.md) for frame (+ audio) payloads\n3. Tool-specific craft when needed:\n   - [p-video-avatar-prompting.md](./references/p-video-avatar-prompting.md)\n   - [p-video-animate-prompting.md](./references/p-video-animate-prompting.md)\n   - [p-video-replace-prompting.md](./references/p-video-replace-prompting.md)\n   - [p-video-edit-prompting.md](./references/p-video-edit-prompting.md)\n4. Validate with the matching `*-quality-checklist.md` in `./references/`.\n\nProduct B-roll and OPEN/MID/CLOSE samples: **Worked example — product B-roll** in [prompt-dramaturgy.md](./references/prompt-dramaturgy.md).\n\n## Pruna tools\n\nMatching install for every model named above. Pick what you need:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `p-video` | Use when someone wants one short video clip from text or images — B-roll, start/end frame animation, or a quick motion shot. Not for full multi-scene films or lip-synced hosts. | `npx skills add PrunaAI/pruna-skills@p-video -y` |\n| `p-video-avatar` | Use when someone wants a person on camera speaking a script — lip-synced host, spokesperson, or narrated avatar from a portrait photo. | `npx skills add PrunaAI/pruna-skills@p-video-avatar -y` |\n| `p-video-animate` | Use when someone wants a photo to move like another video — motion transfer, dance remixes, or performance variations from a template clip. | `npx skills add PrunaAI/pruna-skills@p-video-animate -y` |\n| `p-video-replace` | Use when someone wants to swap a person, outfit, or product inside existing footage while keeping the camera move and audio. | `npx skills add PrunaAI/pruna-skills@p-video-replace -y` |\n| `p-video-edit` | Use when someone wants to edit an existing video with a text instruction — recolor, restyle, remove or add objects, change environment or lighting, update on-screen text, or apply optional reference-guided product and accessory edits. Not for a new clip from scratch or ffmpeg assembly. | `npx skills add PrunaAI/pruna-skills@p-video-edit -y` |\n| `pruna-api` | Use before any Pruna or Replicate HTTP call — credentials, upload/poll/download, parallel batches, and agent safety. | `npx skills add PrunaAI/pruna-skills@pruna-api -y` |\n\nOr install the full suite once: `npx skills add PrunaAI/pruna-skills@pruna -y`\n\nFile v1.0.11:_meta.json\n\n{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"video-prompting\",\n  \"version\": \"1.0.11\",\n  \"publishedAt\": 1788444380774\n}\n\nFile v1.0.11:references/audio-in-video-prompting.md\n\n# Audio-in-video prompting (`p-video`)\n\nHow to **write prompts** when sound matters on `p-video`. Layering / tool picker: `audio-prompting`. Talking heads: [p-video-avatar-prompting.md](./p-video-avatar-prompting.md).\n\n## Three modes\n\n| Mode | API | Prompt job |\n|------|-----|------------|\n| **A — Native SFX / dialogue** | `prompt` + optional `save_audio`; use `duration` | Name **diegetic** sounds the picture should emit |\n| **B — Uploaded audio (preferred for VO/music)** | `audio` URL; **omit `duration`**; `save_audio: true` | Motion matches **mood/beats** of the track — do not paste VO text into `prompt` |\n| **C — Post bed** | Stable Audio mixed under VO in ffmpeg | Bed prompt is separate (`stable-audio-2.5` + `audio-prompting`); video prompt ignores the bed |\n\nNever generate silent `p-video` and post-mux narration unless re-render is impossible (truncation risk). Probe TTS ≤ ~19s before Mode B.\n\n## Mode A — native SFX / dialogue\n\n**When:** `save_audio: true`, no uploaded `audio`, user wants diegetic SFX and/or spoken lines in the clip.\n\n### Diegetic SFX\n\nBe concrete; skip `cinematic soundscape`:\n\n| Bad | Good |\n|-----|------|\n| `epic soundtrack vibe` | `rain ticks on the awning, distant train horn once` |\n| `dramatic music` | `crowd murmur swells, single glass clink` |\n\nKeep cues short — the model invents audio from visual+prompt context when `save_audio` is on.\n\n### Native dialogue\n\nPut **exact spoken words in double quotes** inside the motion `prompt` (Mode A only). Placeholders in square brackets below are docs notation, not literal syntax — do not confuse with Gemini TTS `[tags]`.\n\n**Template** (inside OPEN/MID/CLOSE):\n\n```text\nMID: [same subject] says \"[LINE]\" — mouth open, [one gesture toward target]; [optional diegetic SFX cue]\n```\n\nRules:\n\n- Exact spoken words in **double quotes** — never paraphrase the line\n- Name **who** speaks — repeat the subject label for continuity (`same [role]`)\n- Pair every line with **mouth state** + **one gesture** (point, turn to camera, hand on prop)\n- Keep lines **short** (1–2 per beat)\n- Diegetic SFX can sit beside dialogue — stay concrete (table above)\n- **Mode B:** when `input.audio` is set, do **not** put the VO transcript in `prompt` — motion matches mood only\n\n## Mode B — motion matches uploaded audio\n\n```text\nOPEN: hold wide on dog in tall grass, warm afternoon light.\nMID: gentle push-in as he searches; tail motion matches narrator energy; grass sways.\nCLOSE: settle on end pose — curious head tilt.\n```\n\nRules:\n\n- Describe **picture motion**, not the spoken words.  \n- Match energy: tense VO → tighter push-in; calm story → slow drift.  \n- Optional: `motion matches narrator mood` once — not a transcript.  \n- Triple anchors: [scene-anchor-triple.md](./scene-anchor-triple.md).\n\n## Avatar path (not Mode B fields)\n\n`p-video-avatar` uses `voice_script` + `voice_prompt` + `video_prompt` — not `p-video` `input.audio` for the spoken line. See avatar prompting ref. Do not paste script lines into `voice_prompt`.\n\n## Pre-send\n\n- [ ] Mode A / B / C chosen  \n- [ ] Mode B: `duration` omitted; TTS length probed  \n- [ ] No VO transcript inside motion `prompt` (Mode B)  \n- [ ] Diegetic cues concrete (Mode A) or mood-aligned (Mode B)\n- [ ] Mode A dialogue (if any): quoted line + named speaker + mouth + one gesture\n\nFile v1.0.11:references/camera-lighting-vocabulary.md\n\n# Camera and lighting vocabulary\n\nShared lexicon for `camera_tag` / `lighting_tag` (stills) and motion lines in `p-video` / `p-video-avatar` prompts. Diversity axes: generation-diversity.md#visual-variety (`generation-diversity`). Dramaturgy: [prompt-dramaturgy.md](./prompt-dramaturgy.md).\n\n**Sources:** patterns adapted from [smixs/visual-skills](https://github.com/smixs/visual-skills) and [inference-sh/skills](https://github.com/inference-sh/skills) (MIT); rewritten for Pruna.\n\n## Framing ladder\n\n| Term | Use when |\n|------|----------|\n| ECU (extreme close-up) | Eyes, hands, product detail |\n| CU (close-up) | Face, emotion |\n| MCU (medium close-up) | Talking head default |\n| MS (medium shot) | Waist-up action |\n| MLS / FS | Full body travel |\n| WS / EWS | Environment as character |\n\nLog as `camera_tag`, e.g. `medium close-up, slight low angle`.\n\n## Lens roles (optional but sharp)\n\n| Lens | Feel |\n|------|------|\n| 24mm | Wide, immersive, exaggerated space |\n| 35mm | Documentary natural |\n| 50mm | Intimate human perspective |\n| 85mm | Portrait, compressed background |\n| Macro | Texture, product detail |\n\nExample: `shot on 50mm, eye-level`.\n\n## Camera moves (pick one for MID)\n\n| Move | Prompt cue |\n|------|------------|\n| Dolly / push-in | `slow dolly in`, `gentle push-in` |\n| Dolly out | `slow pull back revealing the room` |\n| Pan | `gentle pan left across the alley` |\n| Tilt | `tilt up from hands to face` |\n| Track / truck | `shoulder-height tracking shot beside the subject` |\n| Crane | `slow crane down past neon signs` |\n| Static + atmosphere | `locked camera, steam rises, light shifts` |\n| Handheld | `subtle handheld drift` (use sparingly) |\n\nAvoid whip pans and stacked contradictory moves in one short clip.\n\n## Motivated lighting\n\nPrefer **named sources** over “beautiful lighting”:\n\n| Source | Example cue |\n|--------|-------------|\n| Window / dawn | `dawn light spreads across the desk` |\n| Practical | `warm lamp spill, cool window fill` |\n| Neon / gel | `magenta-cyan neon rim, wet reflections` |\n| Overhead institutional | `cold fluorescent flicker` |\n| Fire / candle | `candle flicker on faces` (avatar: often too transition-y — prefer steady) |\n| Overcast soft | `soft overcast skylight, low contrast` |\n\nLog as `lighting_tag`. Hex in stills when brand colors matter (`#0d3d2d rim`).\n\n## Palette cues (one look)\n\nPick one coherent palette phrase: `teal-magenta night`, `warm tungsten interior`, `bleached noon desert`, `desaturated documentary`.\n\nDo not stack competing genre looks in one prompt.\n\n## Avatar-friendly defaults\n\nTalking heads: **MCU**, one slow push-in or static, **steady light**, mouth visible. Variety across scenes = change angle/background still, not five camera moves mid-line. See [p-video-avatar-prompting.md](./p-video-avatar-prompting.md).\n\nFile v1.0.11:references/clip-chaining.md\n\n# Clip chaining (multi-scene video)\n\nWhen and how to continue motion across `p-video` clips. Plan JSON examples stay in [scene-anchor-pair.md](./scene-anchor-pair.md) and [scene-anchor-triple.md](./scene-anchor-triple.md); this page is the decision tree + prompt rules.\n\nWorkflows: `visual-transition-reel` · `narrated-multi-scene`.\n\n## Decision tree\n\n```text\nDoes motion continue in the same place/moment (no time jump)?\n  NO  → chain_from_previous: false — hard cut; compose a new OPENING still\n  YES → chain_from_previous: true\n        Prefer frame_chain_mode: extract_last_frame (sequential renders)\n        Only use planned_stills if you accept possible cut jumps\n```\n\n| Situation | `chain_from_previous` | Join |\n|-----------|----------------------|------|\n| Continuous action (run → leap) | `true` | Short crossfade ~0.12–0.15s after extract |\n| New beat / location / pause | `false` | Hard cut (0 crossfade) |\n| First scene | `false` | — |\n| Montage vignettes (no shared motion) | `false` + `parallel_vignettes` | Hard cuts; parallel renders OK |\n\n| `frame_chain_mode` | Next scene `image` | Render order |\n|--------------------|--------------------|--------------|\n| **`extract_last_frame`** | ffmpeg last frame of prior clip | **Sequential** when any scene chains |\n| **`parallel_vignettes`** | each scene’s own start still | **Parallel** |\n| **`planned_stills`** | prior scene end still URL | Parallel once stills exist — higher jump risk |\n\n**Why extract?** Planned end stills often differ from the model’s actual last frame → visible jump.\n\n## Prompt rules for chained beats\n\n1. **Same subject language** — repeat “same [character]” in OPEN/MID/CLOSE.  \n2. **No teleport** — ban `cut to`, `suddenly in`, `walls disappear`; use `gradually`, `walks through`, `ease into`.  \n3. **Match lighting era** — chained clips share `style_bible` and time-of-day.  \n4. **Exit / enter continuity** — if scene 1 CLOSE faces right, scene 2 OPEN should not hard-flip screen direction without a motivated turn.  \n5. **Hard-cut scenes** — treat as fresh OPENING; do not assume prior pose.\n\n## Assembly notes\n\n1. Concat in scene order (ffmpeg concat — see the workflow skill).  \n2. Per-join `crossfades`: chain ~0.12–0.15s; hard cuts 0.  \n3. Normalize audio (48 kHz stereo) when mixing formats.  \n4. Optional bed under native SFX — `audio-prompting`.\n\n## Intake checklist\n\n- [ ] Each scene: chain flag only if motion truly continues  \n- [ ] `frame_chain_mode` chosen  \n- [ ] Chained prompts pass Details Law ([prompt-dramaturgy.md](./prompt-dramaturgy.md))  \n- [ ] Physics tier OK ([physics-safe-motion.md](./physics-safe-motion.md))\n\nFile v1.0.11:references/p-video-animate-prompting.md\n\n# p-video-animate prompting\n\nMotion-transfer craft for `p-video-animate`. QA: [p-video-animate-quality-checklist.md](./p-video-animate-quality-checklist.md). Mixed reels: `avatar-multi-scene`.\n\n**Appearance from `image`, motion from `video`.** Wrong tool for identity swap on real footage → [p-video-replace-prompting.md](./p-video-replace-prompting.md).\n\n## Pairing gates (before every job)\n\nAsk:\n\n1. **Framing** — same body region (head-and-shoulders / medium / full body)?  \n2. **Pose** — facing and limb position roughly aligned with the template’s first frame?  \n3. **Visibility** — same body parts visible; no crop the video lacks?\n\n| Factor | Guidance |\n|--------|----------|\n| Shot size | Match close-up / medium / full |\n| Facing | Front still + profile motion → artifacts |\n| Limbs | If template waves arms, still must show arms |\n| Proportions | Human full-body dance on chibi often breaks gait |\n| Speaking templates | Mouth clear and large when source has dialogue |\n\n**Pairing failure:** head-and-shoulders still + full-body dance → model does **not** invent limbs; choreography is lost. Repose with `p-image-edit` or pick a closer template.\n\n## `instruction_prompt`\n\nOptional. Overrides **behavior**, not identity. **Leave blank** when source motion is already right.\n\n**Useful** — one specific end beat:\n\n```text\nAt the very end of the clip, just after her last gesture, she gives a clear thumbs-up toward the camera. Keep the source motion otherwise.\n```\n\n**Less useful** — redescribes the still:\n\n```text\nA confident woman in a charcoal blazer speaks to the camera in a modern office.\n```\n\n## Style variety\n\nPhotoreal, cartoon, 3D, and mascot stills can share one template when framing aligns — the still’s render style carries through.\n\n## Speaking motion sources\n\nWhen this template feeds lip-sync showcases, the source clip (often from `p-video-avatar`) must show clear speaking / lip movement. See [p-video-avatar-prompting.md](./p-video-avatar-prompting.md) and animate-beats.\n\n## Pre-send\n\n- [ ] Framing / pose / limbs match  \n- [ ] `instruction_prompt` blank or one concrete beat  \n- [ ] Not using animate for in-place replace  \n- [ ] Long templates split (~5s compute per 1s video)\n\nFile v1.0.11:references/p-video-animate-quality-checklist.md\n\n# p-video-animate quality checklist\n\nAfter each animate job, **open the source video, reference image, and output clip** and review them against this list (agent vision review — see `generation-diversity`).\n\n## Applies to\n\nSee the canonical mapping in `generation-diversity`.\n\n## Input gate (pre-render)\n\n- Source **`video`** is the intended motion template (camera path, acting, timing, scene structure); motion is **clear and readable** (not blurry, fast-cut, or low-contrast).\n- Reference **`image`** clearly shows the subject to animate (face/body unobstructed, rights cleared).\n- **First-frame alignment:** framing, pose, and visible limbs match the **first frame** of the source video (or repose with **`p-image-edit`** first).\n- **Mismatch risk:** head-and-shoulders still + full-body template → expect lost choreography, not full-body motion.\n- **Proportion fit:** human full-body motion on meme/mascot/chibi subjects often breaks legs, arms, and contact points—flag before generate.\n- **`instruction_prompt`** (if used) describes **behavior overrides only** — not a repeat of the image description.\n- **`resolution`** and **`target_fps`** match delivery spec.\n- Source longer than budget: plan to **split** the template and animate segments (~5 s compute per 1 s video).\n\n## Motion transfer fidelity\n\n- Output preserves source motion, timing, and camera movement (not a generic re-enactment).\n- Acting beats and scene structure track the reference video.\n- Subject identity and style come from the reference image, not the source video's actor.\n\n## Technical quality\n\n- No severe flicker, warping, or unstable anatomy during motion.\n- Audio (when `save_audio` is true) stays aligned with visual motion.\n- Output duration matches the source video length.\n\n## Clean delivery\n\n- No accidental overlays, stray text, or watermark-like artifacts unless requested.\n- Clip is ready for downstream edit, concat, or platform upload.\n\nFile v1.0.11:references/p-video-avatar-prompting.md\n\n# p-video-avatar prompting\n\nTalking-head prompt craft for `p-video-avatar`. Templates: `avatar-multi-scene`. Camera: [camera-lighting-vocabulary.md](./camera-lighting-vocabulary.md). Physics: [physics-safe-motion.md](./physics-safe-motion.md).\n\n**Do not** use OPEN/MID/CLOSE — the model treats beats as cuts and the clip feels cutty.\n\n## Three-layer stack\n\n| Layer | Field | Job |\n|-------|-------|-----|\n| 1. Plate | `image` | Locked approved still — mouth visible; quality caps the avatar |\n| 2. Voice | `voice_script` + `voice_prompt` | What they say + how they sound |\n| 3. Motion | `video_prompt` | Unique camera/gesture per clip |\n\nNever ship multi-scene reels where every row reuses `medium close-up, gentle dolly push-in`.\n\n## Field hygiene\n\n| Field | Write | Never |\n|-------|-------|-------|\n| **`voice_script`** | Natural spoken lines | Brochure / slogan paste as the only line without human rhythm |\n| **`voice_prompt`** | Short delivery: pacing, warmth, archetype | Product names, script lines, long scene descriptions |\n| **`video_prompt`** | MCU, one slow push-in or static, speaks to camera | OPEN/MID/CLOSE; walk across room; hold up documents; wild gestures |\n\nStylized hosts: match energy to medium (anime slightly more expressive; documentary restrained). Separate hero stills per `visual_style_tag`.\n\n## Micro-actions (talking-head Details Law)\n\nPrefer face/eye micro-moves over locomotion:\n\n- eyes lift to lens, slight nod, natural blink rhythm, subtle lean  \n- one small hand-to-chest max  \n\nAvoid physics traps while speaking ([physics-safe-motion.md](./physics-safe-motion.md) avatar subsection).\n\n## Negative prompt (experimental)\n\nAPI `negative_prompt` = **noun suppression list** (subtitles, captions, watermark…), not creative wording. Strength ~0.3–0.4. Primary fix: positive-only stills (`plain unmarked walls`). See SKILL for defaults.\n\n## Good / bad triples\n\n**Good**\n\n```text\nvoice_script: \"So we tried something weird last quarter — and it actually worked.\"\nvoice_prompt: Natural conversational tone, relaxed pacing, real pauses, honest not salesy.\nvideo_prompt: Medium close-up speaking directly to lens, one very slow push-in, steady light, natural head motion, no cuts.\n```\n\n**Bad**\n\n```text\nvoice_prompt: Mention Pruna and our 10x faster inference in an exciting cinematic way.\nvideo_prompt: OPEN: hold. MID: she walks across the room waving a laptop. CLOSE: product hero shot.\n```\n\n## Pre-send\n\n- [ ] Plate approved; mouth visible  \n- [ ] `voice` locked per character across scenes  \n- [ ] Unique `video_prompt` per clip  \n- [ ] No OPEN/MID/CLOSE  \n- [ ] `voice_prompt` has no script/product paste  \n\nQA: [p-video-avatar-quality-checklist.md](./p-video-avatar-quality-checklist.md).\n\nFile v1.0.11:references/p-video-avatar-quality-checklist.md\n\n# p-video-avatar quality checklist\n\nBefore calling the model and after each avatar clip, **open the still or video and review it visually** against this checklist (agent vision review — see `generation-diversity`).\n\n## Applies to\n\nSee the canonical mapping in `generation-diversity`.\n\n## Input still gate (pre-render)\n\n- Face and mouth/beak are large and clear enough for lip-sync.\n- Mouth/beak and eyes are unobstructed (no hair/props/foreground clutter crossing them).\n- Head pose is speaking-friendly (avoid extreme angles, tiny head crop, or chin cutoff).\n- Identity/style match cast bible and scene continuity.\n- **Photoreal path:** skin reads natural (not mushy/waxy); plate matches `image-prompting` intent.\n- **Try-on → avatar path:** try-on preservation passed; outfit details visible if script references them.\n\n## Speech and performance\n\n- Spoken output matches intended script/audio content.\n- Voice choice is consistent for recurring characters.\n- Delivery tone matches brief; `voice_prompt` is short and does not leak unintended text.\n- **`voice_script`** reads as speakable human dialogue — not brochure/marketing copy.\n\n## Motion and scene dynamism\n\n- **`video_prompt`** is **unique to this clip** — not duplicated from other scenes in the same project.\n- Motion grammar matches the still (props, setting, angle) — e.g. glance targets exist in plate.\n- Multi-scene reels vary camera angle and movement — not every clip `medium close-up, gentle dolly push-in`.\n- **Stylized clips:** motion energy matches `visual_style_tag` (anime vs documentary vs clay).\n- **Motion templates:** when the clip is a source for `p-video-animate`, verify audible speech and visible lip sync — reject smile/wave-only outputs with no dialogue motion.\n\n## Lip-sync and visual stability\n\n- Mouth movement is plausible and synchronized.\n- No facial warping, jitter, or unstable eye/teeth regions.\n- Hands/props near face do not cause ambiguous anatomy artifacts.\n\n## Clean delivery\n\n- `video_prompt` results in clean framing/motion without prompt side effects.\n- No accidental overlays, stray text, or watermark-like artifacts unless requested.\n- For explainers / any text-prone still: `negative_prompt` + `negative_prompt_strength` > 0 on `p-video-avatar` (runner default or plan `defaults.avatar_negative_*`). Tune strength up only if artifacts persist — high values can harm identity/motion.\n- Still lines stayed free of signage/label triggers; `style_bible` holds negations, not `edit_prompt`.\n- Clip is ready for assembly with consistent style/voice across adjacent scenes.\n\nFile v1.0.11:references/p-video-edit-prompting.md\n\n# p-video-edit prompting\n\nSurgical edit prompts for `p-video-edit`. QA: [p-video-edit-quality-checklist.md](./p-video-edit-quality-checklist.md). Identity/slot swap on footage → [p-video-replace-prompting.md](./p-video-replace-prompting.md).\n\n**`prompt`** is the edit instruction. Optional **`images`** (0–4) guide product, accessory, or style when the brief names a reference.\n\n## Core formula\n\nOne principal change per run. Describe the desired final state, then name what must stay unchanged.\n\n```text\nChange only [specific thing]. Preserve [geometry / motion / camera / lighting / unmentioned subjects].\n```\n\n| Do | Don't |\n|----|-------|\n| `Change only the jacket from blue to red. Preserve its cut, folds and motion.` | `Make it cinematic` |\n| Name the source element (`SUV body paint`, `text overlay \"Miam\"`, `plants in the corner`) | Vague `edit the video` |\n| Point at refs: `Add the cargo box from the first reference to the roof.` | `Use the reference` with no slot |\n| Keep-list for camera, motion, and subjects you are not changing | Hope the model leaves them alone |\n\n## Decide edit intent first\n\n| Intent | Prompt must |\n|--------|-------------|\n| **Attribute** | Name the property (color, material, shade) and keep cut/geometry/motion |\n| **Remove** | Name only the object to delete; reconstruct the occluded surface |\n| **Add** | Name what to add and where it attaches; keep the rest |\n| **Environment** | Change setting or walls; preserve subject, furniture, camera path |\n| **Relight** | Change atmosphere/palette only; keep faces, products, blocking |\n| **Text** | Quote the exact string to add, replace, or remove |\n| **Reference-guided** | Map each image to a source slot (`first reference` → roof box / sunglasses) |\n\n## Weak jobs (warn before pay)\n\n- A brand-new scene, plot, or video rather than an edit of this clip\n- Adding an object with its own independent motion — especially a new in-hand shape\n- Changing camera angle, camera motion, or zoom\n\nSplit those into a new `p-video` generation, or keep the camera locked and edit only appearance.\n\n## Patterns (rewrite for the brief — do not paste as the user's prompt)\n\n**Attribute**\n\n```text\nChange only the SUV body paint to deep metallic red. Preserve the vehicle geometry, glass, trim, headlights, tires, reflections and shadows. Keep the environment, lighting and camera movement unchanged.\n```\n\n**Remove**\n\n```text\nRemove only the potted plant from the countertop. Reconstruct the countertop and wall naturally. Keep the person, bottle, camera and lighting unchanged.\n```\n\n**Add (optional reference)**\n\n```text\nAdd the roof cargo box shown in the first reference to the SUV. Match its shape, matte-black material and proportions. Keep it rigidly attached and correctly aligned to the vehicle roof throughout the camera movement. Preserve the vehicle body, windows, trim, wheels, environment and lighting.\n```\n\n**Environment**\n\n```text\nChange only the room wall from beige plaster to deep sage-green plaster. Preserve the room geometry and shadows. Keep the furniture and camera motion unchanged.\n```\n\n**Text**\n\n```text\nRemove only the text \"Miam\". Reconstruct the underlying image naturally where the text was located.\n```\n\n## Reference stills\n\nUse when the user supplied a product, accessory, or SKU to match. Bare packshots — no extra hands or scene props. Index refs in `prompt` (`first reference`, `image 2`). Max 4 images (`jpg`, `jpeg`, `png`, `webp`).\n\n## Pre-send\n\n- [ ] Edit intent chosen (one principal change)\n- [ ] Source element named + keep-list present\n- [ ] Source `video` ≤ 15 seconds\n- [ ] Refs (if any) indexed and ≤ 4\n- [ ] Not a new scene, independent in-hand motion, or camera rewrite\n- [ ] `draft` / `save_audio` decided\n\nFile v1.0.11:references/p-video-edit-quality-checklist.md\n\n# p-video-edit quality checklist\n\nAfter each edit job, **open the source video, optional reference images, and output clip** and review them against this checklist (agent vision review — see `generation-diversity`).\n\n## Applies to\n\nSee the canonical mapping in `generation-diversity`.\n\n## Input gate (pre-render)\n\n- Source **`video`** is the intended scene (≤ **15 seconds**; motion, audio, and framing to keep).\n- **`prompt`** names one principal change and includes a keep-list (camera, motion, unmentioned subjects).\n- Edit intent is explicit: attribute, remove, add, environment, relight, text, or reference-guided.\n- **`images`** (if used) has **1–4** clear references (rights cleared); `prompt` maps each image to a source slot.\n- Product/object refs are bare packshots — no extra hands or scene props in frame.\n- Job is not a brand-new scene/plot, independently moving new in-hand object, or camera angle/zoom rewrite.\n- **`draft`** and **`save_audio`** match the brief (draft for preview; `save_audio: true` to keep source audio).\n- Optional API **`seed`** only when the user requested a reproducible rerun.\n\n## Edit fidelity\n\n- Requested change is present and obvious in every relevant frame.\n- Locked regions remain stable (identity, product geometry, camera path, unmentioned props).\n- \"Change only X\" constraints are respected; unrelated regions do not drift.\n- Reference-guided jobs: added or swapped element reads from the still (shape, material, proportions) and stays attached through the camera move.\n- Text jobs: named string is added, replaced, or removed; underlying pixels reconstruct cleanly.\n- Audio (when `save_audio` is true) stays aligned with the source clip.\n\n## Technical quality\n\n- No severe flicker, warping, or unstable anatomy on edited regions.\n- Output duration matches the source video length.\n- Draft vs standard quality matches the chosen `draft` flag.\n\n## Clean delivery\n\n- No accidental overlays, stray text, or watermark-like artifacts unless requested.\n- Clip is ready for downstream edit, concat, or platform upload.\n\nFile v1.0.11:references/p-video-quality-checklist.md\n\n# p-video quality checklist\n\nAfter each `p-video` output is saved, **open the clip and review it visually** against this checklist (agent vision review — see `generation-diversity`).\n\n## Applies to\n\nSee the canonical mapping in `generation-diversity`.\n\n## Motion and story fidelity\n\n- Video follows the prompt beat and intended camera grammar.\n- Motion is temporally coherent (no sudden identity/scene jumps).\n- Runtime and pacing fit the requested duration/use case.\n\n## Technical quality\n\n- Output `resolution` / `fps` meet the brief.\n- No severe flicker, frame tearing, or unstable object geometry.\n- If image-to-video: subject identity and core composition remain anchored to the input still.\n\n## Scene anchor pair (visual transitions)\n\nWhen using [scene-an\n\nArchive v1.0.10: 18 files, 31458 bytes\n\nFiles: references/audio-in-video-prompting.md (3342b), references/camera-lighting-vocabulary.md (2812b), references/clip-chaining.md (2668b), references/p-video-animate-prompting.md (2230b), references/p-video-animate-quality-checklist.md (1940b), references/p-video-avatar-prompting.md (2733b), references/p-video-avatar-quality-checklist.md (2586b), references/p-video-quality-checklist.md (2507b), references/p-video-replace-prompting.md (3103b), references/p-video-replace-quality-checklist.md (3852b), references/physics-safe-motion.md (2859b), references/prompt-dramaturgy.md (3959b), references/scene-anchor-pair.md (10563b), references/scene-anchor-triple.md (7326b), skill-card.md (3457b), skill.manifest.json (511b), SKILL.md (6051b), _meta.json (135b)\n\nArchive v1.0.9: 18 files, 31277 bytes\n\nFiles: references/audio-in-video-prompting.md (3342b), references/camera-lighting-vocabulary.md (2812b), references/clip-chaining.md (2668b), references/p-video-animate-prompting.md (2230b), references/p-video-animate-quality-checklist.md (1940b), references/p-video-avatar-prompting.md (2733b), references/p-video-avatar-quality-checklist.md (2586b), references/p-video-quality-checklist.md (2507b), references/p-video-replace-prompting.md (3103b), references/p-video-replace-quality-checklist.md (3852b), references/physics-safe-motion.md (2859b), references/prompt-dramaturgy.md (3959b), references/scene-anchor-pair.md (10563b), references/scene-anchor-triple.md (7326b), skill-card.md (3058b), skill.manifest.json (511b), SKILL.md (6050b), _meta.json (134b)\n\nArchive v1.0.8: 18 files, 31210 bytes\n\nFiles: references/audio-in-video-prompting.md (3342b), references/camera-lighting-vocabulary.md (2812b), references/clip-chaining.md (2668b), references/p-video-animate-prompting.md (2230b), references/p-video-animate-quality-checklist.md (1940b), references/p-video-avatar-prompting.md (2733b), references/p-video-avatar-quality-checklist.md (2586b), references/p-video-quality-checklist.md (2507b), references/p-video-replace-prompting.md (3103b), references/p-video-replace-quality-checklist.md (3852b), references/physics-safe-motion.md (2859b), references/prompt-dramaturgy.md (3959b), references/scene-anchor-pair.md (10563b), references/scene-anchor-triple.md (7326b), skill-card.md (2989b), skill.manifest.json (511b), SKILL.md (6050b), _meta.json (134b)\n\nArchive v1.0.7: 18 files, 31298 bytes\n\nFiles: references/audio-in-video-prompting.md (3342b), references/camera-lighting-vocabulary.md (2812b), references/clip-chaining.md (2668b), references/p-video-animate-prompting.md (2230b), references/p-video-animate-quality-checklist.md (1940b), references/p-video-avatar-prompting.md (2733b), references/p-video-avatar-quality-checklist.md (2586b), references/p-video-quality-checklist.md (2507b), references/p-video-replace-prompting.md (3103b), references/p-video-replace-quality-checklist.md (3852b), references/physics-safe-motion.md (2859b), references/prompt-dramaturgy.md (3959b), references/scene-anchor-pair.md (10563b), references/scene-anchor-triple.md (7326b), skill-card.md (3451b), skill.manifest.json (511b), SKILL.md (5917b), _meta.json (134b)","readmeExcerpt":"Skill: video-prompting Owner: pruna-ai Summary: Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. Tags: ai:1.0.14, generative:1.0.14, latest:1.0.14, pruna:1.0.14 Version history: v1.0.14 | 2026-09-29T15:31:08.407Z | auto - Updated to version 1.0.14 - Internal documentation and quality checklist references updated - Sample/re","codeSnippets":[],"executableExamples":[{"language":"text","snippet":"MID: [same subject] says \"[LINE]\" — mouth open, [one gesture toward target]; [optional diegetic SFX cue]"},{"language":"text","snippet":"OPEN: hold wide on dog in tall grass, warm afternoon light.\nMID: gentle push-in as he searches; tail motion matches narrator energy; grass sways.\nCLOSE: settle on end pose — curious head tilt."},{"language":"text","snippet":"Does motion continue in the same place/moment (no time jump)?\n  NO  → chain_from_previous: false — hard cut; compose a new OPENING still\n  YES → chain_from_previous: true\n        Prefer frame_chain_mode: extract_last_frame (sequential renders)\n        Only use planned_stills if you accept possible cut jumps"},{"language":"text","snippet":"A matte stainless pour-over kettle sits on a seamless light-gray studio sweep. Thin steam rises from the spout."},{"language":"text","snippet":"A narrative film scene, 1970s Roman trattoria at night, warm tungsten, cigarette haze, Super-8 grain. A man and a woman in period clothes sit at a small table with wine. She leans in and quietly says, \"Then we leave before sunrise.\" The camera slowly dollies around the table. Audio: Italian radio pop from a small speaker, plates, low room tone, her line clear and close."},{"language":"text","snippet":"A dramatic cinematic shot of an isolated lighthouse standing on a rocky cliff during an enormous Atlantic storm at dusk. The camera begins close to the lighthouse keeper standing outside near the railing as violent wind pulls at his coat. He turns toward the ocean just as a massive wave crashes against the cliff and sends water high above the lighthouse. The camera rapidly pulls back to reveal the scale of the storm. Photorealistic water physics, heavy rain, turbulent clouds, realistic human movement, and epic scale."}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: video-prompting\ndescription: Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining.\nlicense: MIT\nmetadata:\n  version: \"1.0.14\"\n  package: pruna-skills\n---\n\n# Video prompting\n\nVendor-neutral craft for **short video / motion** generation. Works with Pruna `p-video-2-pro` / `p-video-2` / `p-video` family, Runway, Kling, Luma, Veo, and similar APIs.\n\n## Install\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `video-prompting` | Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. | `npx skills add PrunaAI/pruna-skills@video-prompting -y` |\n| `generation-diversity` | Use when writing any generative prompt — ritual seed, explicit structure, scenario axes, and quality gates before paid API calls. | `npx skills add PrunaAI/pruna-skills@generation-diversity -y` |\n\n## When to use\n\n- Text-to-video or image-to-video prompts\n- Start/end frame (anchor pair) or narrated beat (anchor triple) specs\n- Camera and lighting vocabulary in motion lines\n- Physics-safe subject motion\n- Multi-clip continuity / clip chaining\n- Talking-head, motion-transfer, slot-replace, or instruction-based video-edit prompts\n\n## Works with\n\nPruna `p-video-2-pro` / `p-video-2` / `p-video` / `p-video-avatar` / `p-video-animate` / `p-video-replace` / `p-video-edit`, Runway Gen-3, Kling, Luma Dream Machine, Veo, and other video models. Cinematic generation (generated audio, first/last frame): `p-video-2-pro`. 1080p / imported audio / draft: `p-video-2`. Cheaper, faster drafts: `p-video`.\n\n## When NOT to use\n\nUse a different skill instead:\n\n| Skill | Description | Install |\n| --- | --- | --- |\n| `image-prompting` | Use when crafting still-image prompts for any generative model — composition, identity sheets, edits, try-on, and photoreal personas. | `npx skills add PrunaAI/pruna-skills@image-prompting -y` |\n| `audio-prompting` | Use when crafting TTS, music, or bed prompts for any generative audio model — director style, song structure, and post-production layering. | `npx skills add PrunaAI/pruna-skills@audio-prompting -y` |\n| `music-video` | Use when someone wants a full music video — original song or vocals, performance clips, B-roll, and lyric-synced edits. | `npx skills add PrunaAI/pruna-skills@music-video -y` |\n| `narrated-multi-scene` | Use when someone wants a multi-part story with voiceover — episodic B-roll, chaptered promo, or several linked video scenes without on-camera dialogue. | `npx skills add PrunaAI/pruna-skills@narrated-multi-scene -y` |\n| `p-video-2-pro` | Use when someone wants a cinematic clip from text or start/end frames — product ads, documentary shots, or dialogue with generated audio. Not for 1080p, imported audio tracks, or talking-head-only hosts. | `npx skills add PrunaAI/pruna-skills@p-video-2-pro -y` |\n| `p-video-2` | Use when someone wants a polished short clip fro"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7cagwf7q3t0cxrgteb7xk0bh81j0eb\",\n  \"slug\": \"video-prompting\",\n  \"version\": \"1.0.14\",\n  \"publishedAt\": 1790695868407\n}"},{"path":"references/audio-in-video-prompting.md","content":"# Audio-in-video prompting (`p-video`)\n\nHow to **write prompts** when sound matters on `p-video`. Layering / tool picker: `audio-prompting`. Talking heads: [p-video-avatar-prompting.md](./p-video-avatar-prompting.md).\n\n## Three modes\n\n| Mode | API | Prompt job |\n|------|-----|------------|\n| **A — Native SFX / dialogue** | `prompt` + optional `save_audio`; use `duration` | Name **diegetic** sounds the picture should emit |\n| **B — Uploaded audio (preferred for VO/music)** | `audio` URL; **omit `duration`**; `save_audio: true` | Motion matches **mood/beats** of the track — do not paste VO text into `prompt` |\n| **C — Post bed** | Stable Audio mixed under VO in ffmpeg | Bed prompt is separate (`stable-audio-2.5` + `audio-prompting`); video prompt ignores the bed |\n\nNever generate silent `p-video` and post-mux narration unless re-render is impossible (truncation risk). Probe TTS ≤ ~19s before Mode B.\n\n## Mode A — native SFX / dialogue\n\n**When:** `save_audio: true`, no uploaded `audio`, user wants diegetic SFX and/or spoken lines in the clip.\n\n### Diegetic SFX\n\nBe concrete; skip `cinematic soundscape`:\n\n| Bad | Good |\n|-----|------|\n| `epic soundtrack vibe` | `rain ticks on the awning, distant train horn once` |\n| `dramatic music` | `crowd murmur swells, single glass clink` |\n\nKeep cues short — the model invents audio from visual+prompt context when `save_audio` is on.\n\n### Native dialogue\n\nPut **exact spoken words in double quotes** inside the motion `prompt` (Mode A only). Placeholders in square brackets below are docs notation, not literal syntax — do not confuse with Gemini TTS `[tags]`.\n\n**Template** (inside OPEN/MID/CLOSE):\n\n```text\nMID: [same subject] says \"[LINE]\" — mouth open, [one gesture toward target]; [optional diegetic SFX cue]\n```\n\nRules:\n\n- Exact spoken words in **double quotes** — never paraphrase the line\n- Name **who** speaks — repeat the subject label for continuity (`same [role]`)\n- Pair every line with **mouth state** + **one gesture** (point, turn to camera, hand on prop)\n- Keep lines **short** (1–2 per beat)\n- Diegetic SFX can sit beside dialogue — stay concrete (table above)\n- **Mode B:** when `input.audio` is set, do **not** put the VO transcript in `prompt` — motion matches mood only\n\n## Mode B — motion matches uploaded audio\n\n```text\nOPEN: hold wide on dog in tall grass, warm afternoon light.\nMID: gentle push-in as he searches; tail motion matches narrator energy; grass sways.\nCLOSE: settle on end pose — curious head tilt.\n```\n\nRules:\n\n- Describe **picture motion**, not the spoken words.  \n- Match energy: tense VO → tighter push-in; calm story → slow drift.  \n- Optional: `motion matches narrator mood` once — not a transcript.  \n- Triple anchors: [scene-anchor-triple.md](./scene-anchor-triple.md).\n\n## Avatar path (not Mode B fields)\n\n`p-video-avatar` uses `voice_script` + `voice_prompt` + `video_prompt` — not `p-video` `input.audio` for the spoken line. See avatar prompting ref. Do not paste script lines into `voice_prompt`"},{"path":"references/camera-lighting-vocabulary.md","content":"# Camera and lighting vocabulary\n\nShared lexicon for `camera_tag` / `lighting_tag` (stills) and motion lines in `p-video` / `p-video-avatar` prompts. Diversity axes: generation-diversity.md#visual-variety (`generation-diversity`). Dramaturgy: [prompt-dramaturgy.md](./prompt-dramaturgy.md).\n\n**Sources:** patterns adapted from [smixs/visual-skills](https://github.com/smixs/visual-skills) and [inference-sh/skills](https://github.com/inference-sh/skills) (MIT); rewritten for Pruna.\n\n## Framing ladder\n\n| Term | Use when |\n|------|----------|\n| ECU (extreme close-up) | Eyes, hands, product detail |\n| CU (close-up) | Face, emotion |\n| MCU (medium close-up) | Talking head default |\n| MS (medium shot) | Waist-up action |\n| MLS / FS | Full body travel |\n| WS / EWS | Environment as character |\n\nLog as `camera_tag`, e.g. `medium close-up, slight low angle`.\n\n## Lens roles (optional but sharp)\n\n| Lens | Feel |\n|------|------|\n| 24mm | Wide, immersive, exaggerated space |\n| 35mm | Documentary natural |\n| 50mm | Intimate human perspective |\n| 85mm | Portrait, compressed background |\n| Macro | Texture, product detail |\n\nExample: `shot on 50mm, eye-level`.\n\n## Camera moves (pick one for MID)\n\n| Move | Prompt cue |\n|------|------------|\n| Dolly / push-in | `slow dolly in`, `gentle push-in` |\n| Dolly out | `slow pull back revealing the room` |\n| Pan | `gentle pan left across the alley` |\n| Tilt | `tilt up from hands to face` |\n| Track / truck | `shoulder-height tracking shot beside the subject` |\n| Crane | `slow crane down past neon signs` |\n| Static + atmosphere | `locked camera, steam rises, light shifts` |\n| Handheld | `subtle handheld drift` (use sparingly) |\n\nAvoid whip pans and stacked contradictory moves in one short clip.\n\n## Motivated lighting\n\nPrefer **named sources** over “beautiful lighting”:\n\n| Source | Example cue |\n|--------|-------------|\n| Window / dawn | `dawn light spreads across the desk` |\n| Practical | `warm lamp spill, cool window fill` |\n| Neon / gel | `magenta-cyan neon rim, wet reflections` |\n| Overhead institutional | `cold fluorescent flicker` |\n| Fire / candle | `candle flicker on faces` (avatar: often too transition-y — prefer steady) |\n| Overcast soft | `soft overcast skylight, low contrast` |\n\nLog as `lighting_tag`. Hex in stills when brand colors matter (`#0d3d2d rim`).\n\n## Palette cues (one look)\n\nPick one coherent palette phrase: `teal-magenta night`, `warm tungsten interior`, `bleached noon desert`, `desaturated documentary`.\n\nDo not stack competing genre looks in one prompt.\n\n## Avatar-friendly defaults\n\nTalking heads: **MCU**, one slow push-in or static, **steady light**, mouth visible. Variety across scenes = change angle/background still, not five camera moves mid-line. See [p-video-avatar-prompting.md](./p-video-avatar-prompting.md)."},{"path":"references/clip-chaining.md","content":"# Clip chaining (multi-scene video)\n\nWhen and how to continue motion across `p-video` clips. Plan JSON examples stay in [scene-anchor-pair.md](./scene-anchor-pair.md) and [scene-anchor-triple.md](./scene-anchor-triple.md); this page is the decision tree + prompt rules.\n\nWorkflows: `visual-transition-reel` · `narrated-multi-scene`.\n\n## Decision tree\n\n```text\nDoes motion continue in the same place/moment (no time jump)?\n  NO  → chain_from_previous: false — hard cut; compose a new OPENING still\n  YES → chain_from_previous: true\n        Prefer frame_chain_mode: extract_last_frame (sequential renders)\n        Only use planned_stills if you accept possible cut jumps\n```\n\n| Situation | `chain_from_previous` | Join |\n|-----------|----------------------|------|\n| Continuous action (run → leap) | `true` | Short crossfade ~0.12–0.15s after extract |\n| New beat / location / pause | `false` | Hard cut (0 crossfade) |\n| First scene | `false` | — |\n| Montage vignettes (no shared motion) | `false` + `parallel_vignettes` | Hard cuts; parallel renders OK |\n\n| `frame_chain_mode` | Next scene `image` | Render order |\n|--------------------|--------------------|--------------|\n| **`extract_last_frame`** | ffmpeg last frame of prior clip | **Sequential** when any scene chains |\n| **`parallel_vignettes`** | each scene’s own start still | **Parallel** |\n| **`planned_stills`** | prior scene end still URL | Parallel once stills exist — higher jump risk |\n\n**Why extract?** Planned end stills often differ from the model’s actual last frame → visible jump.\n\n## Prompt rules for chained beats\n\n1. **Same subject language** — repeat “same [character]” in OPEN/MID/CLOSE.  \n2. **No teleport** — ban `cut to`, `suddenly in`, `walls disappear`; use `gradually`, `walks through`, `ease into`.  \n3. **Match lighting era** — chained clips share `style_bible` and time-of-day.  \n4. **Exit / enter continuity** — if scene 1 CLOSE faces right, scene 2 OPEN should not hard-flip screen direction without a motivated turn.  \n5. **Hard-cut scenes** — treat as fresh OPENING; do not assume prior pose.\n\n## Assembly notes\n\n1. Concat in scene order (ffmpeg concat — see the workflow skill).  \n2. Per-join `crossfades`: chain ~0.12–0.15s; hard cuts 0.  \n3. Normalize audio (48 kHz stereo) when mixing formats.  \n4. Optional bed under native SFX — `audio-prompting`.\n\n## Intake checklist\n\n- [ ] Each scene: chain flag only if motion truly continues  \n- [ ] `frame_chain_mode` chosen  \n- [ ] Chained prompts pass Details Law ([prompt-dramaturgy.md](./prompt-dramaturgy.md))  \n- [ ] Physics tier OK ([physics-safe-motion.md](./physics-safe-motion.md))"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. Skill: video-prompting Owner: pruna-ai Summary: Use when crafting video or motion prompts for any generative model — dramaturgy, camera, physics-safe motion, frame anchors, and clip chaining. Tags: ai:1.0.14, generative:1.0.14, latest:1.0.14, pruna:1.0.14 Version history: v1.0.14 | 2026-09-29T15:31:08.407Z | auto - Updated to version 1.0.14 - Internal documentation and quality checklist references updated - Sample/re","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":2150,"uniquenessScore":46,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T17:42:52.136Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T17:42:52.136Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T20:58:44.205Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}