{"id":"ea2cd4a7-6e17-4b32-b4a6-230069d9dae6","entityType":"agent","slug":"clawhub-heygen-com-media-use","name":"media-use","canonicalUrl":"https://www.xpersona.co/agent/clawhub-heygen-com-media-use","canonicalPath":"/agent/clawhub-heygen-com-media-use","generatedAt":"2026-10-09T16:13:17.275Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T05:10:42.588Z","emptyReason":null},"description":"Agent Media OS for a HyperFrames project. Resolve BGM, SFX, image, icon, brand logo, voice, color grade, or LUT into a frozen local file or paste-ready block + ledger record (one verb, `resolve`); generate via TTS / music / image models when the catalog misses; produce voiceover, transcription, captions, and background removal through one shared audio engine; operate on media (cut / reframe / transform); and reuse assets across projects. Also use for vague feedback that real footage looks dark, flat, boring, should feel retro/camcorder/print/ASCII, needs privacy, or needs a media reveal. When the host app provides its own music or sound-effect tools, use those for music and sound effects; `resolve --type bgm|sfx` needs the heygen CLI. When `HEYGEN_API_BASE` is set, HeyGen calls go through that host with no CLI sign-in.","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 4.6K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s17fpgb0p797dzkbtbrxw5x1hh89qs64:media-use","sourceUrl":"https://clawhub.ai/heygen-com/media-use","homepage":"https://clawhub.ai/heygen-com/skills/media-use","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/heygen-com/media-use","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/heygen-com/skills/media-use","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":73,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"media-use technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T05:10:42.588Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T05:10:42.588Z","emptyReason":null},"stars":null,"forks":null,"downloads":4575,"packageName":null,"latestVersion":"1.0.75","tractionLabel":"4.6K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T05:10:42.571Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T05:10:42.588Z","lastCrawledAt":"2026-10-09T05:10:42.571Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T05:10:42.571Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.75","createdAt":"2026-10-08T21:03:34.370Z","changelog":"Synced from b1aa71d (main)","fileCount":91,"zipByteSize":211111},{"version":"1.0.74","createdAt":"2026-10-08T15:41:21.947Z","changelog":"Synced from 8dae5ef (main)","fileCount":91,"zipByteSize":211147},{"version":"1.0.73","createdAt":"2026-10-07T18:34:48.007Z","changelog":"Synced from 21d14b2 (main)","fileCount":91,"zipByteSize":211035},{"version":"1.0.72","createdAt":"2026-10-07T07:30:27.419Z","changelog":"Synced from 4fcad1e (main)","fileCount":91,"zipByteSize":208502},{"version":"1.0.71","createdAt":"2026-10-04T15:10:07.259Z","changelog":"Synced from 2497402 (main)","fileCount":89,"zipByteSize":206029},{"version":"1.0.70","createdAt":"2026-10-04T10:38:23.757Z","changelog":"Synced from 981bcf4 (main)","fileCount":86,"zipByteSize":199856},{"version":"1.0.69","createdAt":"2026-10-04T08:59:18.542Z","changelog":"Synced from b04081b (main)","fileCount":86,"zipByteSize":199269},{"version":"1.0.68","createdAt":"2026-10-04T08:06:49.417Z","changelog":"Synced from 32d2bc6 (main)","fileCount":86,"zipByteSize":199022}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17fpgb0p797dzkbtbrxw5x1hh89qs64:media-use","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s17fpgb0p797dzkbtbrxw5x1hh89qs64:media-use` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/heygen-com/media-use before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-heygen-com-media-use/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-heygen-com-media-use/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-heygen-com-media-use/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-heygen-com-media-use/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-heygen-com-media-use/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-heygen-com-media-use/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-09T16:13:17.271Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-heygen-com-media-use/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-heygen-com-media-use/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-heygen-com-media-use/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-heygen-com-media-use/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T05:10:42.588Z","emptyReason":null},"readme":"Skill: media-use\n\nOwner: heygen-com\n\nSummary: Agent Media OS for a HyperFrames project. Resolve BGM, SFX, image, icon, brand logo, voice, color grade, or LUT into a frozen local file or paste-ready block + ledger record (one verb, `resolve`); generate via TTS / music / image models when the catalog misses; produce voiceover, transcription, captions, and background removal through one shared audio engine; operate on media (cut / reframe / transform); and reuse assets across projects. Also use for vague feedback that real footage looks dark, flat, boring, should feel retro/camcorder/print/ASCII, needs privacy, or needs a media reveal. When the host app provides its own music or sound-effect tools, use those for music and sound effects; `resolve --type bgm|sfx` needs the heygen CLI. When `HEYGEN_API_BASE` is set, HeyGen calls go through that host with no CLI sign-in.\n\nTags: latest:1.0.75\n\nVersion history:\n\nv1.0.75 | 2026-10-08T21:03:34.370Z | user\n\nSynced from b1aa71d (main)\n\nv1.0.74 | 2026-10-08T15:41:21.947Z | user\n\nSynced from 8dae5ef (main)\n\nv1.0.73 | 2026-10-07T18:34:48.007Z | user\n\nSynced from 21d14b2 (main)\n\nv1.0.72 | 2026-10-07T07:30:27.419Z | user\n\nSynced from 4fcad1e (main)\n\nv1.0.71 | 2026-10-04T15:10:07.259Z | user\n\nSynced from 2497402 (main)\n\nv1.0.70 | 2026-10-04T10:38:23.757Z | user\n\nSynced from 981bcf4 (main)\n\nv1.0.69 | 2026-10-04T08:59:18.542Z | user\n\nSynced from b04081b (main)\n\nv1.0.68 | 2026-10-04T08:06:49.417Z | user\n\nSynced from 32d2bc6 (main)\n\nv1.0.67 | 2026-10-04T07:32:26.887Z | user\n\nSynced from 5f3191b (main)\n\nv1.0.66 | 2026-10-03T03:38:30.650Z | user\n\nSynced from 835e0c1 (main)\n\nv1.0.65 | 2026-10-02T14:32:23.923Z | user\n\nSynced from 9465048 (main)\n\nv1.0.64 | 2026-10-02T12:53:06.395Z | user\n\nSynced from f1b79d8 (main)\n\nv1.0.63 | 2026-09-29T01:57:49.558Z | user\n\nSynced from 2195db5 (main)\n\nv1.0.62 | 2026-09-29T00:21:35.788Z | user\n\nSynced from e07ea1f (main)\n\nv1.0.61 | 2026-09-27T23:32:38.525Z | user\n\nSynced from 471e90c (main)\n\nv1.0.60 | 2026-09-27T22:11:21.793Z | user\n\nSynced from 0a5e3c7 (main)\n\nv1.0.59 | 2026-09-27T21:21:28.493Z | user\n\nSynced from ff6e210 (main)\n\nv1.0.58 | 2026-09-27T03:23:48.946Z | user\n\nSynced from 8dfdceb (main)\n\nv1.0.57 | 2026-09-27T01:49:56.983Z | user\n\nSynced from d290c05 (main)\n\nv1.0.56 | 2026-09-26T14:24:59.842Z | user\n\nSynced from c238abf (main)\n\nv1.0.55 | 2026-09-24T01:23:59.289Z | user\n\nSynced from 01601d1 (main)\n\nv1.0.54 | 2026-09-23T23:10:49.516Z | user\n\nSynced from 29fc953 (main)\n\nv1.0.53 | 2026-09-23T20:57:38.225Z | user\n\nSynced from d49e7b9 (main)\n\nv1.0.52 | 2026-09-23T00:29:58.718Z | user\n\nSynced from dd97478 (main)\n\nv1.0.51 | 2026-09-21T22:37:05.551Z | user\n\nSynced from 3091dcb (main)\n\nv1.0.50 | 2026-09-19T03:22:35.240Z | user\n\nSynced from 2db126d (main)\n\nv1.0.49 | 2026-09-16T16:39:28.268Z | user\n\nSynced from 3704863 (main)\n\nv1.0.48 | 2026-09-14T01:25:34.476Z | user\n\nSynced from 95bea16 (main)\n\nv1.0.47 | 2026-09-10T10:17:54.339Z | user\n\nSynced from a7e2e38 (main)\n\nv1.0.46 | 2026-09-10T03:24:46.899Z | user\n\nSynced from 0f8eb89 (main)\n\nv1.0.45 | 2026-09-08T23:30:57.744Z | user\n\nSynced from efe2404 (main)\n\nv1.0.44 | 2026-09-08T21:36:41.205Z | user\n\nSynced from a7bf0d2 (main)\n\nv1.0.43 | 2026-09-08T20:41:47.556Z | user\n\nSynced from a92dc68 (main)\n\nv1.0.42 | 2026-09-05T16:30:06.628Z | user\n\nSynced from eec0434 (main)\n\nv1.0.41 | 2026-08-26T23:17:04.243Z | user\n\nSynced from 8392e84 (main)\n\nv1.0.40 | 2026-08-18T01:47:41.909Z | user\n\nSynced from f8a1e2d (main)\n\nv1.0.39 | 2026-08-10T17:36:39.897Z | user\n\nSynced from 6de29f5 (main)\n\nv1.0.38 | 2026-08-05T19:30:01.716Z | user\n\nSynced from b793e42 (main)\n\nv1.0.37 | 2026-08-04T17:45:25.658Z | user\n\nSynced from f9ec934 (main)\n\nv1.0.36 | 2026-07-29T22:27:58.100Z | user\n\nSynced from 30900c3 (main)\n\nv1.0.35 | 2026-07-28T21:48:09.736Z | user\n\nSynced from a996e91 (main)\n\nv1.0.34 | 2026-07-26T10:24:36.588Z | user\n\nSynced from e2e61b0 (main)\n\nv1.0.33 | 2026-07-25T02:35:17.053Z | user\n\nSynced from 3978b8e (main)\n\nv1.0.32 | 2026-07-24T16:59:19.697Z | user\n\nSynced from 9e7ddb0 (main)\n\nv1.0.31 | 2026-07-21T16:45:36.796Z | user\n\nSynced from 696cbdb (main)\n\nv1.0.30 | 2026-07-20T15:21:35.107Z | user\n\nSynced from 6ad738b (main)\n\nv1.0.29 | 2026-07-18T08:37:49.671Z | user\n\nSynced from 49113eb (main)\n\nv1.0.28 | 2026-07-17T23:09:17.963Z | user\n\nSynced from 4a2c9ee (main)\n\nv1.0.27 | 2026-07-16T22:21:31.902Z | user\n\nSynced from 428e571 (main)\n\nv1.0.26 | 2026-07-15T14:25:19.457Z | user\n\nSynced from 7d21cc9 (main)\n\nArchive index:\n\nArchive v1.0.75: 91 files, 211111 bytes\n\nFiles: audio/assets/sfx/CREDITS.md (1183b), audio/assets/sfx/manifest.json (3721b), audio/references/bgm.md (8141b), audio/references/captions/authoring.md (9545b), audio/references/captions/motion.md (5656b), audio/references/captions/transcript-handling.md (5439b), audio/references/remove-background.md (8472b), audio/references/requirements.md (4347b), audio/references/sfx.md (3678b), audio/references/transcribe.md (2783b), audio/references/tts-to-captions.md (1755b), audio/references/tts.md (14011b), audio/scripts/audio.mjs (13474b), audio/scripts/audio.test.mjs (5930b), audio/scripts/gemini-pipeline.test.mjs (4932b), audio/scripts/heygen-tts.mjs (4388b), audio/scripts/heygen-tts.test.mjs (1408b), audio/scripts/heygen-voice.mjs (3868b), audio/scripts/heygen-voice.test.mjs (8536b), audio/scripts/lib/audio-meta.mjs (1166b), audio/scripts/lib/audio-meta.test.mjs (4586b), audio/scripts/lib/bgm-volume.mjs (236b), audio/scripts/lib/bgm.mjs (11558b), audio/scripts/lib/bgm.test.mjs (2639b), audio/scripts/lib/concurrency.mjs (527b), audio/scripts/lib/concurrency.test.mjs (1654b), audio/scripts/lib/gemini-auth_test.py (4430b), audio/scripts/lib/gemini-auth.mjs (1610b), audio/scripts/lib/gemini-auth.py (1981b), audio/scripts/lib/gemini-auth.test.mjs (3097b), audio/scripts/lib/gemini-tts.mjs (4711b), audio/scripts/lib/gemini-tts.test.mjs (9018b), audio/scripts/lib/heygen.mjs (10925b), audio/scripts/lib/heygen.test.mjs (10191b), audio/scripts/lib/host-audio.mjs (1470b), audio/scripts/lib/host-audio.test.mjs (4036b), audio/scripts/lib/media-record.mjs (4028b), audio/scripts/lib/media-record.test.mjs (7057b), audio/scripts/lib/python.mjs (3290b), audio/scripts/lib/python.test.mjs (5282b), audio/scripts/lib/sfx.mjs (6577b), audio/scripts/lib/sfx.test.mjs (4698b), audio/scripts/lib/tts.mjs (17162b), audio/scripts/lib/tts.spawn.test.mjs (5249b), audio/scripts/lib/tts.test.mjs (8282b), audio/scripts/lyria-recipe.py (5023b), audio/scripts/wait-bgm.mjs (5813b), audio/scripts/wait-bgm.test.mjs (3336b), luts/index.json (1997b), luts/README.md (1087b), references/audio.md (2234b), references/grading.md (6892b), references/media-treatment-recipes.md (35868b), references/media-treatments.md (13920b), references/memory.md (3498b), references/meta.md (6780b), references/operations.md (14823b), references/resolve.md (9605b), references/setup-providers.md (8542b), references/telemetry-dashboard.md (4200b), scripts/audio-duck.mjs (4069b), scripts/compatibility.test.mjs (3804b), scripts/dither.mjs (9969b), scripts/dither.test.mjs (4723b), scripts/eval.mjs (14891b), scripts/lib/config-lock.mjs (1520b), scripts/lib/cutlist.mjs (6095b), scripts/lib/duck.mjs (3345b), scripts/lib/error-diffusion.mjs (6631b), scripts/lib/index-gen.mjs (1924b), scripts/lib/manifest.mjs (9313b), scripts/lib/media-fetch.mjs (2838b), scripts/lib/media-home.mjs (730b), scripts/lib/npx-sync.mjs (2328b), scripts/lib/parakeet-words.mjs (1149b), scripts/lib/prefs-store.mjs (6617b), scripts/lib/recipe-store.mjs (13033b), scripts/lib/telemetry.mjs (7107b), scripts/lib/transcriptCutFade.mjs (911b), scripts/lib/words.mjs (728b)\n\nFile v1.0.75:SKILL.md\n\n---\nname: media-use\ndescription: Agent Media OS for a HyperFrames project. Resolve BGM, SFX, image, icon, brand logo, voice, color grade, or LUT into a frozen local file or paste-ready block + ledger record (one verb, `resolve`); generate via TTS / music / image models when the catalog misses; produce voiceover, transcription, captions, and background removal through one shared audio engine; operate on media (cut / reframe / transform); and reuse assets across projects. Also use for vague feedback that real footage looks dark, flat, boring, should feel retro/camcorder/print/ASCII, needs privacy, or needs a media reveal. When the host app provides its own music or sound-effect tools, use those for music and sound effects; `resolve --type bgm|sfx` needs the heygen CLI. When `HEYGEN_API_BASE` is set, HeyGen calls go through that host with no CLI sign-in.\n---\n\n**Plugin installs:** Before setup or freshness commands, follow [plugin execution rules](../hyperframes/references/plugin-installation.md) when this skill is inside a HyperFrames plugin. Standalone installs keep the update instructions below.\n\n# media-use\n\nThe media OS for HyperFrames: resolve · generate · operate · remember — every media type, one skill, zero context noise.\n\nOnly when `HEYGEN_API_BASE` is set in your environment (a host app set it and pays for HeyGen with its own key): HeyGen media is already paid for. Do not ask the person to install or sign in to the `heygen` CLI and do not offer its OAuth allowance; catalog search, TTS and avatar calls here go through the host. When that same host also gives you its own HeyGen tools, use those first. When a call through the host is refused, tell the person the host's message as written (it names the fix, such as adding or replacing the key in the app's Settings) and stop; do not switch to another provider unless they ask.\n\nFirst run otherwise (no `HEYGEN_API_BASE`), when you will use HeyGen media (catalog search, TTS, avatar video): install and sign in to the `heygen` CLI (the free-usage path), then verify with `npx hyperframes media-use resolve --doctor`. Setup and providers: `references/setup-providers.md`.\n\nMusic and sound effects inside a host app: when the app you run in gives you its own music or sound-effect tools, use those. Without `HEYGEN_API_BASE`, `resolve --type bgm` and `--type sfx` search the HeyGen catalog through the `heygen` CLI; without it they fail and say so (`sfx` still answers from its bundled library).\n\nWithout `HEYGEN_API_BASE`, before generating a voiceover or an avatar video, tell the person: signing in to the heygen CLI with OAuth (`heygen auth login --oauth`) gives a free allowance for TTS voiceover and avatar videos, while an API key bills API credits.\n\n## Resolve — the one verb\n\n```bash\nnpx hyperframes media-use resolve --type <type> --intent \"<description>\" --project <dir>\n```\n\nReturns one line: `resolved <id> → <path> (<type>, <metadata>)`. All search noise stays on disk.\n\n| Type    | One-line intent                                                                  |\n| ------- | -------------------------------------------------------------------------------- |\n| `bgm`   | background music (HeyGen catalog via the `heygen` CLI, 10k+ tracks)              |\n| `sfx`   | sound effects (bundled 19-file library + catalog via the `heygen` CLI)           |\n| `image` | photos, backgrounds (HeyGen asset search, 75k+ vectors)                          |\n| `icon`  | icons, symbols (transparent)                                                     |\n| `logo`  | official brand marks (theSVG → GitHub avatar → favicon; never redrawn)           |\n| `voice` | TTS voiceover (HeyGen free-usage path; optional local Kokoro)                    |\n| `grade` | measured correction candidate; broad polish/stylization follows Media Treatments |\n| `lut`   | user-provided or explicitly chosen reusable validated `.cube` file               |\n\nBefore resolving fresh, list reusable candidates with `--candidates` and judge fit yourself — reuse rules, all flags, ingest (`--from`), and adopt are in `references/resolve.md`.\n\n## Treat broad visual feedback as media intent\n\nWhen a user explicitly asks to fix, polish, stylize, obscure, emphasize, or\nreveal photographic media, read `references/media-treatments.md` even if they\ndo not name color grading or an effect. Inspect the real `<img>`/`<video>`,\nchoose one primary intent, then use deterministic persistence and verification.\nUse a matching recipe as an optional tested seed, or inspect\n`hyperframes media-treatment --capabilities --json`, then request one relevant\nfamily/effect with `--capability <id>` and assemble a custom treatment from\ncanonical controls. Never load `--all` for ordinary authoring. A treatment may\ncompose correction, a preset, finishing, compatible shader effects, supported\nkeyframes, and optional Registry overlays. Add only source-justified bounded\ntuning and compatible parts, never effects merely to make the result look more\nsophisticated. Persist the final combined payload with\n`hyperframes media-treatment`.\n\nUse one progressively escalating workflow. For video, inspect one labeled\nearly/middle/late contact sheet rather than reading frames separately. Apply one\ncandidate and inspect one after-sheet for ordinary correction or polish.\nEscalate to individual frames or moving draft evidence only when the result is\nambiguous, temporal, stylized, LUT-based, HDR/LOG-sensitive, private, or\nbrand-critical.\n\nFor ordinary correction or polish, persist the final treatment's\npreset/adjustment JSON.\nDo not generate a `.cube` LUT merely to encode exposure, shadows, contrast, or\nwarmth. Use a LUT only when the user supplies one or the selected treatment\nexplicitly owns one. `resolve --type grade --for ... --analyze` is measurement\nevidence, not permission to replace the chosen treatment with a generated LUT.\nDo not recreate supported vignette, grain, blur, pixelate, color, or treatment\neffects with CSS/SVG overlays; that bypasses Studio controls and the canonical\npreview/render shader path.\n\n## Be proactive — run a media opportunity pass\n\nThe human usually can't tell which media would lift the piece. You can. When you build or review a composition, do **one** grounded scan and then **ask once** — don't silently add, and don't nag per asset.\n\nSurface an opportunity only when a concrete signal is present:\n\n| Signal detected                                          | Offer                                                                                                  |\n| -------------------------------------------------------- | ------------------------------------------------------------------------------------------------------ |\n| On-screen text / a script with no voiceover              | TTS voiceover (audio engine)                                                                           |\n| Emoji or a `<div>` styled as an icon                     | resolve real `icon`s                                                                                   |\n| Image that is a placeholder, tiny, or upscaled-looking   | a better `image` (and/or upscale — see `references/operations.md`)                                     |\n| Hard scene cuts / transitions with no sound              | transition `sfx`                                                                                       |\n| A piece over ~10s with no music bed                      | `bgm`                                                                                                  |\n| Footage that reads under/over-exposed or color-cast      | a corrective grade (inspect it with `hyperframes media-treatment --selector '#hero' --analyze --json`) |\n| Photographic media that feels visually flat or off-topic | one specific source-appropriate preset or custom treatment, with the intended target named             |\n| A meaningful media entrance/reveal that feels static     | one supported seek-safe treatment animation; preserve color unless the request also justifies a preset |\n\nRules that keep this a help, not nagware: **grounded, not generic** (no signal → no suggestion); **opinionated + concrete** (propose the specific fix with defaults chosen — the human approves **all / some / none**); **once per project** (one consolidated ask; respect \"leave it\"); **surface, never silently mutate** (color grades especially: propose and preview — a gray-world \"correction\" ruins an intentional sunset or neon look).\n\n## Where to look — read only the file your task needs\n\n| Task                                                                      | Read                             |\n| ------------------------------------------------------------------------- | -------------------------------- |\n| resolve / reuse / adopt / ingest, flags, cascade, inventory               | `references/resolve.md`          |\n| color grading, LUTs, smart grade (`--for`), grade-compare                 | `references/grading.md`          |\n| voiceover / TTS, music, SFX, captions, transcription (audio engine)       | `references/audio.md`            |\n| cut / reframe / transform existing media, exact error diffusion, HEVC     | `references/operations.md`       |\n| source-aware creative treatments, realtime effects, overlays, reveals     | `references/media-treatments.md` |\n| install + auth, provider table, RAM ladders, `--local-only`, `--provider` | `references/setup-providers.md`  |\n| remembered preferences + frozen recipes (user memory)                     | `references/memory.md`           |\n| ownership matrix, usage stats, telemetry, privacy (maintainer-facing)     | `references/meta.md`             |\n\nFile v1.0.75:luts/README.md\n\n# LUT library (authoring)\n\n`index.json` is the agent-consumed catalog of color-grade looks. Each entry resolves\non demand — no `.cube` bodies are committed to the repo.\n\nEach look has:\n\n- `id`, `description`, `tags`, `intensity` — matching + application metadata.\n- `url` (optional) — a hosted `.cube` downloaded, validated, and frozen at resolve\n  time, exactly like bgm/image assets.\n- `params` (optional) — a deterministic `buildCube` spec used offline (`--local-only`)\n  or as a fallback if the `url` download/validation fails.\n\nAn entry needs at least one of `url` or `params`; prefer both (CDN url with a params\nfallback) so resolution is never blocked on the network.\n\n## Hosting a new look (operators)\n\n1. Generate the `.cube` (e.g. `resolve -t lut --params '{...}'` or a graded export).\n2. Upload it to the public CDN origin bucket:\n\n   ```\n   aws s3 cp <id>.cube s3://heygen-public/luts/<id>.cube\n   ```\n\n   It is then served at `https://static.heygen.ai/luts/<id>.cube` (CloudFront).\n\n3. Add an entry to `index.json` with that `url` (and ideally a `params` fallback).\n\nFile v1.0.75:_meta.json\n\n{\n  \"ownerId\": \"kn77d06grj6xqp3dqwkk4bavhn89pegt\",\n  \"slug\": \"media-use\",\n  \"version\": \"1.0.75\",\n  \"publishedAt\": 1791493414370\n}\n\nFile v1.0.75:audio/references/bgm.md\n\n# Background music (BGM)\n\nOne music bed per composition, produced by the shared audio engine (`scripts/audio.mjs` → `scripts/lib/bgm.mjs`). Two routes, chosen by the engine's one switch — whether a HeyGen credential is present:\n\n- **HeyGen retrieval — the default when credentialed.** Search HeyGen's music catalog by mood, download the top track. No generation; same `~/.heygen` / `$HEYGEN_API_KEY` credential as TTS.\n- **Local generation (Lyria → MusicGen) — the fallback when there is no credential** (or when asked for explicitly). Generate a WAV from a mood prompt. There is **no `npx hyperframes bgm` command**; the engine spawns `scripts/lyria-recipe.py` or an inline MusicGen script directly.\n\n> **Run the Preflight first — no credential is not a green light to silently generate locally.** Before generating, complete the sign-in **Preflight** (see `../../SKILL.md` → Preflight): run `npx hyperframes auth status`, recommend signing in, and **STOP for the user's choice** (sign in for HeyGen's music library, or continue offline with local generation). This applies to a one-off \"generate a BGM\" request just as much as inside a full workflow.\n\n## Driving it from the request\n\n`audio_request.json` → `bgm: { mode?, query?, prompt? }`:\n\n- **`mode`** — `retrieve | generate | none`. Omit for **auto** (retrieve when credentialed, else generate). An **explicit** `retrieve` is strict: no credential ⇒ skip, never a detached generate (so a caller with no `wait-bgm` step, e.g. product-launch, can't get a pending job it won't await).\n- **`query`** — the mood, used for retrieval and as a fallback prompt seed (e.g. a storyboard's `music:` field, falling back to `message` → `arc` → `\"calm cinematic underscore\"`).\n- **`prompt`** — an explicit full prompt for generation; omit and the engine infers one (see Mood inference). Optional `blob` / `archetype` / `arc` feed that inference.\n\nBoth routes keep a file of yours already at the output name: the engine writes the next free name (`track-2.mp3`), reports it as an anomaly, and `bgm.path` carries the real path.\n\n## HeyGen retrieval (default)\n\n`searchSounds(query, \"music\", { limit: 5 })` → `GET /audio/sounds?query=<mood>&type=music&limit=5`. Take the top result (ranked by `score`), download its presigned `audio_url` → `assets/bgm/track.mp3`. Synchronous. No match → skip (BGM is optional; never fail the render over it). Cue written to `audio_meta.json`:\n\n```jsonc\n{\n  \"path\": \"assets/bgm/track.mp3\",\n  \"volume\": 0.12,\n  \"mode\": \"retrieve\",\n  \"query\": \"calm cinematic underscore\",\n  \"duration_s\": 42.0,\n}\n```\n\n`volume` comes from the engine's `bgmDefaultVolume()`: `BGM_BED_VOLUME` (currently `0.12` ≈ -18 dB — a bed under the voice) under narration, `BGM_SILENT_VOLUME` (currently `0.9`) for a silent film (no voice). Tune those constants in `scripts/lib/bgm.mjs`, not call sites. An explicit `volume` in `audio_meta.json` always overrides this default. `bgm_pending` is `false` — the file is on disk when the engine returns.\n\nFor short launch videos, do not assume the beginning of the retrieved file is the best edit point. Check the opening against later five-second sections. If the track starts with a quiet build but a later section has a stronger, clean musical entrance, trim from that section and apply a short fade-in and longer fade-out. Repeat this check whenever the composition duration changes; the final music file must cover the full cut without a silent tail.\n\n## Local generation (fallback) — Lyria → MusicGen\n\nSpawned **detached** so voice work isn't blocked; `audio_meta.bgm_pending: true` and `bgm_pid` / `bgm_log` are set until it finishes. **Run `scripts/wait-bgm.mjs` before assembling** — it polls the output file / process / log, detects crashes, and writes `bgm_status.json` (`status: ready | failed | timeout | disabled`). A failed/absent track is simply omitted; it never blocks voice/SFX.\n\n| Order | Provider                             | Env / deps                                                                            | Speed                                   | Quality                     |\n| ----- | ------------------------------------ | ------------------------------------------------------------------------------------- | --------------------------------------- | --------------------------- |\n| 1     | Google Lyria RealTime                | `$GEMINI_API_KEY` or `$GOOGLE_API_KEY` + `google-genai` (auto-installed on demand)    | Real-time stream (≈ requested duration) | Production-grade            |\n| 2     | MusicGen (`facebook/musicgen-small`) | Python `transformers + torch + soundfile + numpy` (~300 MB first run; auto-installed) | Slow on CPU; fast on Apple MPS / CUDA   | Decent; prompt-only control |\n\nOutput → `assets/bgm/track.wav`, target = total voice duration. MusicGen generates **one** seed clip (≤28–30s, under the decoder's positional limit) then crossfade-loops it up to the target (or trims down if shorter), avoiding per-segment seams. Backend selection is by what can actually **run**: Lyria only when `import google.genai` succeeds, else MusicGen; if neither can be made to run, BGM is skipped (voice + SFX still render).\n\n## Mood inference (the generate prompt)\n\n`inferBgmPrompt()` in `scripts/lib/bgm.mjs`: an explicit `prompt` wins; otherwise industry-keyword **base** → narrative-**archetype** shape → emotional-**arc** tiebreaker.\n\n| Match in `blob` / `query`                              | Base prompt                                                                 | BPM |\n| ------------------------------------------------------ | --------------------------------------------------------------------------- | --- |\n| `crypto / nft / web3 / defi / token / blockchain`      | atmospheric electronic, deep bass, futuristic synths, restrained percussion | 100 |\n| `finance / fintech / bank / payment / invest / wealth` | calm cinematic, soft strings, subtle piano, restrained percussion           | 92  |\n| `creative / agency / design / studio / art / brand`    | playful electronic, warm pads, light percussion                             | 115 |\n| _(default: SaaS / tech / platform)_                    | uplifting corporate tech, bright modern piano with synth pads               | 108 |\n\nArchetype then reshapes the arc — PAS → \"MINOR to MAJOR\" build; BAB / future-pacing → aspirational rising; feature-cascade → +10 BPM driving; demo-loop → −8 BPM minimal. The emotional arc breaks remaining ties (tension→relief, excitement, trust/reassurance).\n\n## Lyria knobs (direct recipe use)\n\nThe engine bakes BPM / scale into the **prompt text** (via the inference above) and passes only `--output` / `--duration` / `--prompt` to the recipe. If you invoke `scripts/lyria-recipe.py` directly you can also set: `--bpm` (90–110 calm, 110–130 energetic), `--brightness` (0–1, ≥0.7 promotional), `--density` (0–1, higher = fuller), `--scale` (`MAJOR` / `MINOR` / `PENTATONIC` / …), `--negative-prompt` (styles to exclude). MusicGen ignores all of these — put the mood in the prompt.\n\n## Failure modes\n\n| Failure                                       | Behavior                                                                                 |\n| --------------------------------------------- | ---------------------------------------------------------------------------------------- |\n| No music match (retrieve)                     | `bgm: null`, anomaly logged. Render proceeds without BGM.                                |\n| Explicit `retrieve`, no credential            | Skipped (no silent generate fallback). Use `mode: generate` or omit `mode` for auto.     |\n| Neither Lyria nor MusicGen can run (generate) | `bgm` disabled with a `pip install …` hint. Voice + SFX still render.                    |\n| Generate still rendering at assemble time     | `bgm_pending: true`; `wait-bgm.mjs` waits/checks and writes `bgm_status.json` first.     |\n| Generate crashed                              | `wait-bgm.mjs` → `bgm_status.json { status: \"failed\" }`; the `<audio>` track is omitted. |\n\nBGM failure never blocks a render.\n\nFile v1.0.75:audio/references/captions/authoring.md\n\n# Captions\n\n<!-- registry-items: allow=max-width,data-composition-src,hyperframes-registry,blend-mode,caption-style,font-family -->\n\n**The live search is the source of truth for what the registry has.** The table(s) below are a hand-maintained sample and under-cover by design: run `npx hyperframes catalog --query \"<what you want>\" --json` — it needs nothing installed — before concluding the registry lacks something. Item names here are checked against `registry/registry.json` by `bun run lint:skills`.\n\nBefore authoring: confirm the transcript came from the right Whisper model. CLI default `small.en` silently translates non-English audio — see [`../transcribe.md`](../transcribe.md) → \"Language Rule\" and [`transcript-handling.md`](transcript-handling.md) for the mandatory quality check.\n\nAnalyze spoken content to determine caption style. If user specifies a style, use that. Otherwise, detect tone from the transcript.\n\n## Transcript Source\n\n```json\n[\n  { \"id\": \"w0\", \"text\": \"Hello\", \"start\": 0.0, \"end\": 0.5 },\n  { \"id\": \"w1\", \"text\": \"world.\", \"start\": 0.6, \"end\": 1.2 }\n]\n```\n\n`id` (`w0`, `w1`, …) is the stable reference for per-word overrides and is added by `hyperframes transcribe`. It's optional for backwards compatibility with hand-authored transcripts. See [`../transcribe.md`](../transcribe.md) → \"Output Shape\" for how this is produced, and [`transcript-handling.md`](transcript-handling.md) for cleanup before consumption.\n\n## Style Detection (When No Style Specified)\n\nRead the full transcript before choosing. Four dimensions:\n\n**1. Visual feel** — corporate→clean; energetic→bold; storytelling→elegant; technical→precise; social→playful.\n\n**2. Color palette** — dark+bright for energy; muted for professional; high contrast for clarity; one accent color.\n\n**3. Font mood** — heavy/condensed for impact; clean sans for modern; rounded for friendly; serif for elegance.\n\n**4. Animation character** — scale-pop for punchy; gentle fade for calm; word-by-word for emphasis; typewriter for technical.\n\n## Per-Word Styling\n\nScan for words deserving distinct treatment:\n\n- **Brand/product names** — larger size, unique color\n- **ALL CAPS** — scale boost, flash, accent color\n- **Numbers/statistics** — bold weight, accent color\n- **Emotional keywords** — exaggerated animation (overshoot, bounce)\n- **Call-to-action** — highlight, underline, color pop\n- **Marker highlight** — for beyond-color emphasis (highlight sweep, circle, burst, scribble, sketchout), see `hyperframes-animation/rules/css-marker-patterns.md`.\n\n## Script-to-Style Mapping\n\n| Tone         | Font mood                | Animation                          | Color                       | Size    |\n| ------------ | ------------------------ | ---------------------------------- | --------------------------- | ------- |\n| Hype/launch  | Heavy condensed, 800-900 | Scale-pop, back.out(1.7), 0.1-0.2s | Bright on dark              | 72-96px |\n| Corporate    | Clean sans, 600-700      | Fade+slide, power3.out, 0.3s       | White/neutral, muted accent | 56-72px |\n| Tutorial     | Mono/clean sans, 500-600 | Typewriter/fade, 0.4-0.5s          | High contrast, minimal      | 48-64px |\n| Storytelling | Serif/elegant, 400-500   | Slow fade, power2.out, 0.5-0.6s    | Warm muted tones            | 44-56px |\n| Social       | Rounded sans, 700-800    | Bounce, elastic.out, word-by-word  | Playful, colored pills      | 56-80px |\n\n## Word Grouping\n\n- **High energy:** 2-3 words. Quick turnover.\n- **Conversational:** 3-5 words. Natural phrases.\n- **Measured/calm:** 4-6 words. Longer groups.\n\nBreak on sentence boundaries, 150ms+ pauses, or max word count.\n\n## Positioning\n\n- **Landscape (1920x1080):** Bottom 80-120px, centered\n- **Portrait (1080x1920):** Lower middle ~600-700px from bottom, centered\n- Never cover the subject's face\n- `position: absolute` — never relative\n- One caption group visible at a time\n\n## Text Overflow Prevention\n\nUse `window.__hyperframes.fitTextFontSize()`:\n\n```js\nvar result = window.__hyperframes.fitTextFontSize(group.text.toUpperCase(), {\n  fontFamily: \"Outfit\",\n  fontWeight: 900,\n  maxWidth: 1600,\n});\nel.style.fontSize = result.fontSize + \"px\";\n```\n\nOptions: `maxWidth` (1600 landscape, 900 portrait), `baseFontSize` (78), `minFontSize` (42), `fontWeight`, `fontFamily`, `step` (2).\n\nCSS safety nets: `max-width` on container, `overflow: visible` (**not** `hidden` — hidden clips scaled emphasis words and glow effects), `position: absolute`, explicit `height`. When per-word styling uses `scale > 1.0`, compute `maxWidth = safeWidth / maxScale` to leave headroom.\n\n**Container pattern:** Full-width absolute container, centered. Do **not** use `left: 50%; transform: translateX(-50%)` — causes clipping at composition edges.\n\n## Caption Exit Guarantee\n\nEvery group **must** have a hard kill after exit animation:\n\n```js\ntl.to(groupEl, { opacity: 0, scale: 0.95, duration: 0.12, ease: \"power2.in\" }, group.end - 0.12);\n// `tl.set` is an instant flip, not a tween — safe to set `visibility` here (core's \"no animating\n// visibility\" rule applies to tweens, which can't smoothly interpolate non-numeric values anyway).\ntl.set(groupEl, { opacity: 0, visibility: \"hidden\" }, group.end);\n```\n\nSelf-lint after building timeline — place **before** `window.__timelines[id] = tl` so it runs at composition init:\n\n```js\nGROUPS.forEach(function (group, gi) {\n  var el = document.getElementById(\"cg-\" + gi);\n  if (!el) return;\n  tl.seek(group.end + 0.01);\n  var computed = window.getComputedStyle(el);\n  if (computed.opacity !== \"0\" && computed.visibility !== \"hidden\") {\n    console.warn(\n      \"[caption-lint] group \" + gi + \" still visible at t=\" + (group.end + 0.01).toFixed(2) + \"s\",\n    );\n  }\n});\ntl.seek(0);\n```\n\n## Pre-Built Caption Components\n\nBefore building caption styles from scratch, check the registry — 15 ready-to-use caption components cover the most common styles. Install with `npx hyperframes add <name>` and wire as a sub-composition via `data-composition-src` (see `hyperframes-registry`).\n\n```bash\nnpx hyperframes catalog --tag caption-style   # list all caption components\nnpx hyperframes add caption-highlight         # install a specific one\n```\n\n| Style                     | Component                    | Best for                     |\n| ------------------------- | ---------------------------- | ---------------------------- |\n| TikTok-style highlight    | `caption-highlight`          | Social, high-energy          |\n| Karaoke pill              | `caption-pill-karaoke`       | Music, lyric videos          |\n| Cinematic editorial       | `caption-editorial-emphasis` | Documentary, storytelling    |\n| Glitch / cyber            | `caption-glitch-rgb`         | Tech, gaming                 |\n| Full-screen slam          | `caption-kinetic-slam`       | Hype, announcements          |\n| Neon glow                 | `caption-neon-glow`          | Night, club, neon aesthetics |\n| Neon accent (multi-color) | `caption-neon-accent`        | Colorful, playful            |\n| Wipe reveal               | `caption-clip-wipe`          | Clean, modern                |\n| Gradient fill             | `caption-gradient-fill`      | Vibrant, eye-catching        |\n| Matrix decode             | `caption-matrix-decode`      | Sci-fi, tech reveals         |\n| Emoji pop                 | `caption-emoji-pop`          | Social, casual               |\n| Parallax layers           | `caption-parallax-layers`    | Depth, cinematic             |\n| Particle burst            | `caption-particle-burst`     | Celebration, impact keywords |\n| Lava texture              | `caption-texture`            | Bold, dramatic               |\n| Weight shift              | `caption-weight-shift`       | Elegant, typographic         |\n\nRelated: `caption-blend-difference` (tagged `text` / `blend-mode`, not `caption-style`, so it won't appear under the filter above) auto-inverts text against any background via `mix-blend-mode: difference` — useful when the background is busy or unpredictable.\n\nBrowse all with previews: [hyperframes.heygen.com/catalog](https://hyperframes.heygen.com/catalog)\n\nCaption components ship with transparent backgrounds — they're pure overlays. If the underlying video is bright or busy, add a contrast layer (e.g. a semi-transparent dark div) in the host composition beneath the caption sub-composition, not inside the component itself.\n\n## Further References\n\n- [`motion.md`](motion.md) — karaoke, marker effects, audio-reactive modulation, scatter exits.\n- [`transcript-handling.md`](transcript-handling.md) — input formats, quality checks, cleaning, external API fallback.\n- `hyperframes-animation/rules/css-marker-patterns.md` — marker highlighting (deterministic, fully seekable).\n\n## Constraints\n\n- Deterministic. No `Math.random()`, no `Date.now()`.\n- Sync to transcript timestamps.\n- One group visible at a time.\n- Every group must have a hard `tl.set` kill at `group.end`.\n- Fonts: the compiler auto-embeds only its **built-in mapped set** (Inter, Roboto, Montserrat, …) — for those, just declare `font-family` in CSS. Any **other** font (a brand/custom font like `TT Norms Pro`, or a non-Latin CJK/Devanagari family) is **not** auto-supplied: it needs an `@font-face` pointing at a real `.woff2` shipped with the project, or the text silently falls back to a generic font in the render. Don't assume a `font-family` you can see locally will render — the render machine is a clean headless Chrome with no installed fonts.\n\nFile v1.0.75:audio/references/captions/motion.md\n\n# Dynamic Caption Techniques\n\nYou are here because SKILL.md told you to read this file before writing animation code. Pick your technique combination from the table below based on the energy level you detected from the transcript, then implement using standard GSAP patterns.\n\n## Technique Selection by Energy\n\n| Energy level | Highlight                             | Exit                | Cycle pattern                             |\n| ------------ | ------------------------------------- | ------------------- | ----------------------------------------- |\n| High         | Karaoke with accent glow + scale pop  | Scatter or drop     | Alternate highlight styles every 2 groups |\n| Medium-high  | Karaoke with color pop                | Scatter or collapse | Alternate every 3 groups                  |\n| Medium       | Karaoke (subtle, white only)          | Fade + slide        | Alternate every 3 groups                  |\n| Medium-low   | Karaoke (minimal scale change)        | Fade                | Single style, vary ease per group         |\n| Low          | Karaoke (warm tones, slow transition) | Collapse            | Alternate every 4 groups                  |\n\n**All energy levels use karaoke highlight as the baseline.** The difference is intensity — high energy gets accent color + glow + 15% scale pop on active words, low energy gets a gentle white shift with 3% scale.\n\n**Emphasis words always break the pattern.** When a word is flagged as emphasis (emotional keyword, ALL CAPS, brand name), give it a stronger animation than surrounding words (larger scale, accent color, overshoot ease). This creates contrast.\n\n**Marker highlight modes add a visual layer on top of karaoke.** For emphasis words that need more than color/scale, add a marker-style effect: highlight sweep, circle, burst, scribble, or sketchout. See `hyperframes-animation/rules/css-marker-patterns.md` for implementation details. Match mode to energy: burst for hype, circle for key terms, highlight for standard, scribble for subtle.\n\n## Audio-Reactive Captions (Mandatory for Music)\n\n**If the source audio is music (vocals over instrumentation, beats, any musical content), you MUST extract audio data and add audio-reactive animations.** This is not optional — music without audio reactivity looks disconnected. Even low-energy ballads get subtle bass pulse and treble glow.\n\nNo special wiring is needed. The group loop already iterates over every caption group to build entrance, karaoke, and exit tweens. At that point, read the audio data for each group's time range and use it to modulate the group's animation intensity with regular GSAP tweens.\n\n```js\n// Load audio data inline (same pattern as TRANSCRIPT)\nvar AUDIO = JSON.parse(audioDataJson); // { fps, totalFrames, frames: [{ bands: [...] }] }\n\nGROUPS.forEach(function (group, gi) {\n  var groupEl = document.getElementById(\"cg-\" + gi);\n  if (!groupEl) return;\n\n  // Read peak energy for this group's time range\n  var startFrame = Math.floor(group.start * AUDIO.fps);\n  var endFrame = Math.min(Math.floor(group.end * AUDIO.fps), AUDIO.totalFrames - 1);\n  var peakBass = 0;\n  var peakTreble = 0;\n  for (var f = startFrame; f <= endFrame; f++) {\n    var frame = AUDIO.frames[f];\n    if (!frame) continue;\n    peakBass = Math.max(peakBass, frame.bands[0] || 0, frame.bands[1] || 0);\n    peakTreble = Math.max(peakTreble, frame.bands[6] || 0, frame.bands[7] || 0);\n  }\n\n  // Modulate entrance — louder groups enter bigger and glowier\n  tl.to(\n    groupEl,\n    {\n      scale: 1 + peakBass * 0.06,\n      textShadow:\n        \"0 0 \" + Math.round(peakTreble * 12) + \"px rgba(255,255,255,\" + peakTreble * 0.4 + \")\",\n      duration: 0.3,\n      ease: \"power2.out\",\n    },\n    group.start,\n  );\n\n  // Reset at exit so audio-driven values don't persist\n  tl.set(groupEl, { scale: 1, textShadow: \"none\" }, group.end - 0.15);\n});\n```\n\nThis shapes the animation at build time, not playback time — no per-frame callbacks, no `tl.call()` loops, no async fetch timing issues. Loud groups come in with more weight and glow; quiet groups come in soft. The audio data modulates _how much_, the content determines _what_.\n\nKeep audio reactivity subtle — 3-6% scale variation and soft glow. Heavy pulsing makes text unreadable.\n\nTo generate the audio data file:\n\n```bash\npython3 skills/hyperframes-creative/scripts/extract-audio-data.py audio.mp3 --fps 30 --bands 8 -o audio-data.json\n```\n\n## Combining Techniques\n\nDon't use the same highlight animation on every group — cycle through styles using the group index. Don't combine multiple competing animations on the same word at the same timestamp. Vary techniques across groups to match the content's pace changes.\n\n**Marker highlight effects** layer well with karaoke — use karaoke for the word-by-word reveal, then add a marker effect on emphasis words only. For example: karaoke highlights each word in white, but brand names get a yellow highlight sweep and stats get a red circle. Cycle marker modes across groups for visual variety.\n\n## Runtime Tools\n\nCaption motion uses standard HyperFrames runtime APIs. Use the canonical sources:\n\n- **GSAP timeline + tween syntax** — `hyperframes-animation/adapters/gsap.md` (eases, position parameter, performance)\n- **`window.__hyperframes.fitTextFontSize` / `pretext`** — `hyperframes-core/references/determinism-rules.md` → Layout Contract (overflow prevention, per-frame text measurement)\n- **Audio data extraction** — generate via `python3 skills/hyperframes-creative/scripts/extract-audio-data.py audio.mp3 --fps 30 --bands 8 -o audio-data.json`, then load inline as shown in \"Audio-Reactive Captions\" above\n\nFile v1.0.75:audio/references/captions/transcript-handling.md\n\n# Transcript Guide\n\nFor the `transcribe` CLI invocation, the `.en`-translates-non-English rule, and whisper model selection, see [`../transcribe.md`](../transcribe.md). This file covers what to do with the resulting transcript when authoring captions: input formats, mandatory quality checks, cleaning code, external-API fallbacks.\n\n## Supported Input Formats\n\nThe CLI auto-detects and normalizes these formats:\n\n| Format                | Extension | Source                                                                      | Word-level?       |\n| --------------------- | --------- | --------------------------------------------------------------------------- | ----------------- |\n| whisper.cpp JSON      | `.json`   | `hyperframes init --video`, `hyperframes transcribe`                        | Yes               |\n| OpenAI Whisper API    | `.json`   | `openai.audio.transcriptions.create({ timestamp_granularities: [\"word\"] })` | Yes               |\n| SRT subtitles         | `.srt`    | Video editors, subtitle tools, YouTube                                      | No (phrase-level) |\n| VTT subtitles         | `.vtt`    | Web players, YouTube, transcription services                                | No (phrase-level) |\n| Normalized word array | `.json`   | Pre-processed by any tool                                                   | Yes               |\n\n**Word-level timestamps produce better captions.** SRT/VTT give phrase-level timing, which works but can't do per-word animation effects.\n\n## Transcript Quality Check (Mandatory)\n\nAfter every transcription, **read the transcript and check for quality issues before proceeding.** Bad transcripts produce nonsensical captions. Never skip this step.\n\n### What to look for\n\n| Signal                       | Example                                | Cause                                                                        |\n| ---------------------------- | -------------------------------------- | ---------------------------------------------------------------------------- |\n| Music note tokens (`♪`, `�`) | `{ \"text\": \"♪\" }` or `{ \"text\": \"�\" }` | Whisper detected music, not speech                                           |\n| Garbled / nonsense words     | \"Do a chin\", \"Get so gay\", \"huh\"       | Model misheard lyrics or background noise                                    |\n| Long gaps with no words      | 20+ seconds of only `♪` tokens         | Instrumental section — expected, but high ratio means speech is being missed |\n| Repeated filler              | Many \"huh\", \"uh\", \"oh\" entries         | Model is hallucinating on music                                              |\n| Very short word spans        | Words with `end - start < 0.05`        | Unreliable timestamp alignment                                               |\n\n### Automatic retry rules\n\n**If more than 20% of entries are `♪`/`�` tokens, or the transcript contains obvious nonsense words, the transcription failed.** Do not proceed with the bad transcript. Instead:\n\n1. **Retry with `medium.en`** if the original used `small.en` or smaller:\n   ```bash\n   npx hyperframes transcribe audio.mp3 --model medium.en\n   ```\n2. **If `medium.en` also fails** (still >20% music tokens or garbled), tell the user the audio is too noisy for local transcription and suggest:\n   - Providing lyrics manually as an SRT/VTT file\n   - Using an external API (OpenAI or Groq Whisper — see below)\n3. **Always clean the transcript** before building captions — filter out `♪`/`�` tokens and entries where `text` is a single non-word character. Only real words should reach the caption composition.\n\n### Cleaning a transcript\n\nAfter transcription (even with a good model), strip non-word entries:\n\n```js\nvar raw = JSON.parse(transcriptJson);\nvar words = raw.filter(function (w) {\n  if (!w.text || w.text.trim().length === 0) return false;\n  if (/^[♪�\\u266a\\u266b\\u266c\\u266d\\u266e\\u266f]+$/.test(w.text)) return false;\n  if (/^(huh|uh|um|ah|oh)$/i.test(w.text) && w.end - w.start < 0.1) return false;\n  return true;\n});\n```\n\nFor model-selection guidance by content type, see [`../transcribe.md`](../transcribe.md) → \"Picking a model by content type\".\n\n## Using External Transcription APIs\n\nFor the best accuracy, use an external API and import the result:\n\n**OpenAI Whisper API** (recommended for quality):\n\n```bash\n# Generate with word timestamps, then import\ncurl https://api.openai.com/v1/audio/transcriptions \\\n  -H \"Authorization: Bearer $OPENAI_API_KEY\" \\\n  -F file=@audio.mp3 -F model=whisper-1 \\\n  -F response_format=verbose_json \\\n  -F \"timestamp_granularities[]=word\" \\\n  -o transcript-openai.json\n\nnpx hyperframes transcribe transcript-openai.json\n```\n\n**Groq Whisper API** (fast, free tier available):\n\n```bash\ncurl https://api.groq.com/openai/v1/audio/transcriptions \\\n  -H \"Authorization: Bearer $GROQ_API_KEY\" \\\n  -F file=@audio.mp3 -F model=whisper-large-v3 \\\n  -F response_format=verbose_json \\\n  -F \"timestamp_granularities[]=word\" \\\n  -o transcript-groq.json\n\nnpx hyperframes transcribe transcript-groq.json\n```\n\n## If No Transcript Exists\n\n1. Check the project root for `transcript.json`, `.srt`, or `.vtt` files.\n2. If none found, run [`../transcribe.md`](../transcribe.md) — pick the starting model from \"Picking a model by content type\" there.\n3. Run the quality check above. If it fails, retry with a larger model or fall back to manual lyrics / external API.\n\nFile v1.0.75:audio/references/remove-background.md\n\n# Background Removal\n\nMake a transparent overlay (typical: a talking head over an arbitrary scene). Uses `u2net_human_seg` (Apache-2.0).\n\n```bash\nnpx hyperframes remove-background subject.mp4 -o transparent.webm          # default: VP9 + alpha\nnpx hyperframes remove-background subject.mp4 -o transparent.mov           # ProRes 4444 (editing)\nnpx hyperframes remove-background portrait.jpg -o cutout.png               # single-image cutout\nnpx hyperframes remove-background subject.mp4 -o subject.webm \\\n  --background-output plate.webm                                           # both layers, one pass\nnpx hyperframes remove-background subject.mp4 -o transparent.webm --device cpu\nnpx hyperframes remove-background --info                                   # detected providers\n```\n\n## Output Format\n\n- **`.webm` (VP9 alpha)** — default. Plug straight into `<video>` for Chrome-native transparent playback (~1 MB / 4s @ 1080p).\n- **`.mov` (ProRes 4444)** — round-trip in editors (Premiere / Resolve / DaVinci). ~50 MB / 4s.\n- **`.png`** — single-image cutout.\n\n## Quality (`--quality`)\n\nControls VP9 encoder CRF only — segmentation quality is fixed. Higher quality keeps the cutout's RGB closer to the source MP4 (important when overlaying the cutout on its own source).\n\n| Preset     | CRF | When                                          |\n| ---------- | --- | --------------------------------------------- |\n| `fast`     | 30  | Iterating, smaller files, looser color match  |\n| `balanced` | 18  | **Default**; visually identical for most uses |\n| `best`     | 12  | Master / final delivery, tightest color match |\n\n## Device (`--device`)\n\n`auto` (default) picks CoreML on Apple Silicon, CUDA when available, otherwise CPU. Force with `--device cpu | coreml | cuda`. CUDA requires `HYPERFRAMES_CUDA=1` plus a GPU-enabled `onnxruntime-node` build. Use `--info` to inspect detected providers without rendering.\n\n## Compositing patterns — pick the right one\n\nThe cutout WebM is a **re-encoded copy** of the source MP4's RGB. What sits behind it matters.\n\n| Pattern                                                  | Behind the cutout                       | Result                                                                                                          |\n| -------------------------------------------------------- | --------------------------------------- | --------------------------------------------------------------------------------------------------------------- |\n| **Cutout over a different scene** (most common)          | Static image, gradient, unrelated video | Looks great. Single RGB source for the subject.                                                                 |\n| **Cutout over its own source mp4** (text-behind-subject) | Same mp4 the cutout came from           | At `balanced` doubling is barely visible; at `fast` you'll see color shift / edge halo. Use `best` for masters. |\n| **Cutout over a different take of the same person**      | Footage of the same subject             | **Two overlapping people. Don't do this.**                                                                      |\n\n## Text-behind-subject pattern (two non-obvious rules)\n\nPutting a headline behind a presenter cutout:\n\n```html\n<video\n  src=\"presenter.mp4\"\n  id=\"bg\"\n  data-start=\"0\"\n  data-duration=\"6\"\n  data-track-index=\"0\"\n  muted\n  playsinline\n></video>\n\n<h1 id=\"headline\" style=\"z-index:2; ...\">MAKE IT IN HYPERFRAMES</h1>\n\n<div class=\"cutout-wrap\" style=\"position:absolute; inset:0; z-index:3; opacity:0\">\n  <video\n    src=\"presenter.webm\"\n    data-start=\"0\"\n    data-duration=\"6\"\n    data-track-index=\"1\"\n    muted\n    playsinline\n  ></video>\n</div>\n```\n\n```js\n// Flip the wrapper's opacity at the cut, NOT the video's\ntl.set(\".cutout-wrap\", { opacity: 1 }, 3.3);\n```\n\nTwo rules that are easy to miss:\n\n1. **Wrap the cutout `<video>` in a non-timed `<div>` and animate the wrapper's opacity, not the video element's.** The framework forces `opacity: 1` on active clips (any element with `data-start` / `data-duration`), so animating the video's opacity directly is silently overridden. The wrapper has no `data-*` attributes, so it's owned by your CSS / GSAP.\n2. **Both videos use `data-start=\"0\"` and `data-media-start=\"0\"`** so the framework decodes them in sync from t=0. Late-mounting the cutout (`data-start=3.3`) introduces a seek + warm-up that lands a frame off the base mp4 — visible as one frame of misalignment at the cut.\n\n## Layer separation (`--background-output`)\n\nEmits a **second** transparent video alongside the cutout: same source RGB, alpha is `255 - mask` instead of `mask`. The cutout has the subject opaque; the plate has the surroundings opaque (with a transparent hole where the subject was). Use it when text / graphics need to live **between** the two layers.\n\n| File                             | Alpha is…                                               | Use it for                                                       |\n| -------------------------------- | ------------------------------------------------------- | ---------------------------------------------------------------- |\n| `-o subject.webm`                | mask — subject opaque, background transparent           | Foreground layer (top)                                           |\n| `--background-output plate.webm` | inverse mask — surroundings opaque, subject transparent | Bottom layer; place text / graphics between this and the subject |\n\nBoth share the same `--quality` and run from a single inference pass — only encode cost roughly doubles. Only valid for video inputs with `.webm` / `.mov` outputs.\n\n**Hole-cut, not inpainted.** The subject region in `plate.webm` is fully transparent — composite something opaque under it to fill the hole.\n\n**Single test for whether `--background-output` is the right tool:** _will anything ever be visible through the subject's silhouette where the subject used to be?_ If no, you don't need the plate — `subject.webm` alone over a different background is enough.\n\n### Use case → right tool\n\n| Use case                                                                            | Right tool                                                                         |\n| ----------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------- |\n| Text/graphics between the cutout and the plate (this command's reason for existing) | **Hole-cut** (`--background-output`)                                               |\n| Subject onto an unrelated scene                                                     | Just `subject.webm`; ignore the plate                                              |\n| Show the room _without_ the person, alone over no other content                     | **Clean plate** — needs an inpainter (LaMa, ProPainter, E2FGVI). Not this command. |\n| Replace the subject with a different subject                                        | **Clean plate** — same as above                                                    |\n\n### Canonical 3-layer template (plate + content + cutout)\n\nShip just the two transparent layers and let arbitrary content live between them — no original mp4 needed:\n\n```html\n<!-- z=1 plate: surroundings opaque, subject silhouette transparent -->\n<video\n  src=\"plate.webm\"\n  data-start=\"0\"\n  data-duration=\"6\"\n  data-track-index=\"0\"\n  muted\n  playsinline\n></video>\n\n<!-- z=2 your content lives between the layers -->\n<h1 id=\"headline\" style=\"z-index:2; ...\">MAKE IT IN HYPERFRAMES</h1>\n\n<!-- z=3 cutout floats the subject back on top -->\n<div class=\"cutout-wrap\" style=\"position:absolute; inset:0; z-index:3\">\n  <video\n    src=\"subject.webm\"\n    data-start=\"0\"\n    data-duration=\"6\"\n    data-track-index=\"1\"\n    muted\n    playsinline\n  ></video>\n</div>\n```\n\nFunctionally equivalent to the text-behind-subject pattern above, but doesn't require shipping the original mp4 — the plate replaces it. Use this when delivering just the two transparent layers as a reusable asset.\n\n## When `remove-background` is NOT the right tool\n\nIf a user asks for \"the room **without** the person, displayed standalone\" (no subject anywhere, no compositing on top), `--background-output` is wrong — its plate has a transparent hole, not a filled-in clean plate. They need an **inpainter**: LaMa, ProPainter, or E2FGVI. Tell them this command can't do it.\n\nFile v1.0.75:audio/references/requirements.md\n\n# Requirements & Caches\n\n## Credential & key priority\n\nRun `npx hyperframes auth status` to see what's configured and which engines a workflow will use (see the skill's **Preflight** section). Keys resolve in this order — **first match wins**:\n\n| Provider                             | Resolution order (first non-empty wins)                                                                                                                                    | Local deps when used                             |\n| ------------------------------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------ |\n| **HeyGen** (TTS + BGM/SFX retrieval) | `$HEYGEN_API_KEY` → `$HYPERFRAMES_API_KEY` → `~/.heygen/credentials` (shared with heygen-cli; `$HEYGEN_CONFIG_DIR` overrides the dir; written by `hyperframes auth login`) | none (REST)                                      |\n| **ElevenLabs** (TTS fallback)        | `$ELEVENLABS_API_KEY`                                                                                                                                                      | `pip install elevenlabs`                         |\n| **Lyria** (BGM fallback)             | `$GEMINI_API_KEY` → `$GOOGLE_API_KEY`                                                                                                                                      | `pip install google-genai`                       |\n| **Kokoro** (TTS, no key)             | always — final voice fallback                                                                                                                                              | `pip install kokoro-onnx soundfile`              |\n| **MusicGen** (BGM, no key)           | always — final music fallback                                                                                                                                              | `pip install transformers torch soundfile numpy` |\n\n`hyperframes auth login` (browser OAuth) is the recommended setup: one sign-in, every project, no per-repo `.env`. An OAuth login is sent as `Authorization: Bearer`; an API key as `X-Api-Key`; both are tagged with `X-HeyGen-Source: cli`. OAuth CLI users can consume the web-plan free allowance for HeyGen TTS (10 min/month); API keys follow the normal API billing path. With no HeyGen credential, voice/BGM run fully locally (Kokoro / MusicGen) — `hyperframes auth status` and `hyperframes doctor` both report whether those local deps are installed.\n\n## Model caches & system dependencies\n\nEach command downloads its own model on first run and caches it under `~/.cache/hyperframes/`:\n\n- **TTS (HeyGen)** — no local deps; needs a HeyGen credential + `ffmpeg` on PATH (to transcode the mp3 response to `.wav`). Credential resolves like the CLI: `$HEYGEN_API_KEY` → `$HYPERFRAMES_API_KEY` → `~/.heygen/credentials` (shared with heygen-cli; run `npx hyperframes auth login`). An OAuth login is sent as `Authorization: Bearer`; an API key as `X-Api-Key`; both include `X-HeyGen-Source: cli` so the backend can apply CLI OAuth free usage.\n- **TTS (ElevenLabs)** — same as HeyGen: API key + `ffmpeg`.\n- **TTS (Kokoro)** — Kokoro-82M (~311 MB) + voices (~27 MB) in `tts/`. Requires Python 3.8+ with `kokoro-onnx` and `soundfile` (`pip install kokoro-onnx soundfile`). Non-English text also needs `espeak-ng` system-wide.\n- **BGM (Lyria)** — needs `$GEMINI_API_KEY` or `$GOOGLE_API_KEY` + `pip install google-genai`. No local model cache.\n- **BGM (MusicGen)** — `pip install transformers torch soundfile`. `facebook/musicgen-small` (~300 MB) cached under `~/.cache/huggingface/` on first run.\n- **Transcribe** — Whisper model size depending on choice (75 MB – 3.1 GB) in `whisper/`, downloaded from HuggingFace on first use. `whisper.cpp` itself is NOT bundled: the CLI resolves it from PATH, installs via Homebrew (macOS), or builds it from source with git+cmake on first use (`$HYPERFRAMES_WHISPER_PATH` overrides).\n- **Remove-background** — `u2net_human_seg` (~168 MB ONNX) in `background-removal/models/`. Peak inference RAM ~1.5 GB.\n\nRun `npx hyperframes doctor` if a command fails because of a missing dependency.\n\nFile v1.0.75:audio/references/sfx.md\n\n# Sound effects (SFX)\n\nNamed sound effects, produced by the shared audio engine (`scripts/audio.mjs` → `scripts/lib/sfx.mjs`). **Provider-gated** by the engine's one switch — whether a HeyGen credential is present, decided once (not per cue):\n\n- **HeyGen credential present → retrieve every cue** from HeyGen's audio library (`/v3/audio/sounds`, `type=sound_effects`, `min_score=0.4`). Search-and-download, **not** generation. The bundled library is NOT consulted.\n- **No credential → the bundled 21-file library** (`assets/sfx/` + `manifest.json`): match each cue name, copy the matched file into the project. Offline, deterministic, free.\n\nThere is no `npx hyperframes sfx` command. SFX is never generated — it is retrieved (online) or taken from the bundled library (offline).\n\n## Cues — request → meta\n\nEach line names the effects it wants: `lines[].sfx: [\"whoosh\", \"ui click\"]`. The engine flattens these into cues, resolves them per the switch, dedupes identical `(id, name)` pairs (the same effect named twice downloads/copies once), and writes `audio_meta.sfx[]`:\n\n```jsonc\n{\n  \"id\": \"3\",                       // joins the cue to the caller's model (frame / scene / segment)\n  \"name\": \"whoosh\",\n  \"file\": \"assets/sfx/whoosh.mp3\", // downloaded or copied, relative to project root\n  \"source\": \"heygen\" | \"local\",    // which route resolved it\n  \"offset_s\": 0,                   // delay from the line's start\n  \"duration_s\": 0.57,\n  \"volume\": 0.35                   // SFX sit UNDER voice + BGM\n}\n```\n\nA cue that matches nothing is **skipped** (recorded as an anomaly); SFX never blocks a render. Neither route replaces a file of yours already at the output name: the cue gets the next free name (`whoosh-2.mp3`), reported as an anomaly, and `file` carries the real path.\n\n## HeyGen retrieval (credentialed)\n\n`searchSounds(name, \"sound_effects\", { limit: 3, minScore: 0.4 })` → top hit → `assets/sfx/<slug>.mp3`. Results are ranked by `score` (each carries a presigned `audio_url`, `duration`, `description`). The floor is **0.4** because good SFX hits score ~0.5–0.67 — below the API's default `0.7`, which would silently drop most named cues (only whoosh/swoosh-family clears 0.7). `duration_s` comes from the result (else 1.0). Name effects concretely (`glass shatter`, not `dramatic sound`); a vague query returns a poor match.\n\n## Bundled library (no credential)\n\n21 curated files in `assets/sfx/`, indexed by `manifest.json` — `{ file, duration, description }` per key (e.g. `whoosh`, `pop`, `click`, `chime`, `riser`, `impact-bass-1`, `glitch-1`, `typing`, …). A cue name resolves by **manifest key, file basename, or slug**, so `whoosh`, `whoosh.mp3`, or `\"ui click\"` (→ slug) all match. Matched files are copied into the project's `assets/sfx/`; `duration_s` comes from the manifest, so timing is known **offline** — e.g. `riser` is 10.03s, so trigger it at `climax − 10.03s`. The manifest's `description` field carries placement hints per effect; read `assets/sfx/manifest.json` for the full set and usage.\n\n## Rules\n\n- **Volume ~0.35.** SFX must sit under narration and BGM, not fight them.\n- **No match → skip, don't fail.** A missing effect logs an anomaly and moves on; never a render blocker.\n- **Retrieval (credentialed) or bundled library (offline) — never generation.** You search HeyGen by text, or match a name against the 21-file manifest.\n- **One asset per distinct name.** Reuse across lines is deduped to a single download/copy, many cues.\n- **The switch is global, not per cue.** With a credential, retrieval handles even the long tail (effects not in the 21); without one, only the 21 bundled names resolve.\n\nFile v1.0.75:audio/references/transcribe.md\n\n# Transcription\n\nCreate normalized word-level timestamps. **Always specify `--model` explicitly** — the CLI default is `small.en`, which silently translates non-English audio into English.\n\n```bash\nnpx hyperframes transcribe audio.mp3  --model small.en             # known English\nnpx hyperframes transcribe video.mp4  --model small --language es  # known Spanish\nnpx hyperframes transcribe audio.mp3  --model small                # unknown language (auto-detect)\nnpx hyperframes transcribe subtitles.srt                           # import existing\nnpx hyperframes transcribe subtitles.vtt\nnpx hyperframes transcribe openai-response.json\n```\n\n## Language Rule (Non-Negotiable)\n\n`.en` models (`tiny.en` / `base.en` / `small.en` / `medium.en`) **translate** non-English audio into English. This silently destroys the original language.\n\n1. **Known English** → `--model small.en` (or `medium.en` for music / noisy audio)\n2. **Known non-English** → `--model small --language <iso-code>` (no `.en` suffix)\n3. **Unknown language** → `--model small` (whisper auto-detects)\n\n**CLI default is `small.en`** — do not rely on it; always pass `--model` to make the choice explicit. `--language` also filters out non-target-language segments from mixed-language audio.\n\n## Model Sizes\n\n| Model      | Size   | Speed    | When                                  |\n| ---------- | ------ | -------- | ------------------------------------- |\n| `tiny`     | 75 MB  | Fastest  | Quick previews, smoke tests           |\n| `base`     | 142 MB | Fast     | Short clips, clear audio              |\n| `small`    | 466 MB | Moderate | Default for most multilingual content |\n| `medium`   | 1.5 GB | Slow     | Music with vocals, noisy audio        |\n| `large-v3` | 3.1 GB | Slowest  | Production quality                    |\n\n### Picking a model by content type\n\n1. Speech over silence / light background → `small.en`\n2. Speech over music, or music with vocals → start with `medium.en`\n3. Produced music track (vocals + full instrumentation) → start with `medium.en`; expect to need manual lyrics or an external API ([`captions/transcript-handling.md`](captions/transcript-handling.md) → \"Using External Transcription APIs\")\n4. Multilingual → `medium` or `large-v3` (no `.en` suffix), pair with `--language`\n\n## Output Shape\n\nCompositions consume a flat array of word objects. The `id` (`w0`, `w1`, …) is added during normalization for stable references in caption overrides; optional for backwards compatibility.\n\n```json\n[\n  { \"id\": \"w0\", \"text\": \"Hello\", \"start\": 0.0, \"end\": 0.5 },\n  { \"id\": \"w1\", \"text\": \"world.\", \"start\": 0.6, \"end\": 1.2 }\n]\n```\n\nFor mandatory caption-quality checks, retry rules, and the OpenAI/Groq Whisper API import path, see `captions/transcript-handling.md`.\n\nFile v1.0.75:audio/references/tts-to-captions.md\n\n# TTS → Captions\n\nWhen no recorded voiceover exists, generate one and obtain word-level caption timing. Two paths depending on which TTS provider is in use:\n\n## Path A — HeyGen (single call, no Whisper)\n\nHeyGen returns word timestamps in the same response as the audio. Use the\nbundled REST helper (the `hyperframes tts` command is Kokoro-only):\n\n```bash\nnode skills/media-use/audio/scripts/heygen-tts.mjs \\\n  script.txt --output narration.wav --words narration.words.json\n```\n\n`narration.words.json` is already in the `[{ id, text, start, end }]` shape the captions pipeline consumes — no separate transcribe pass.\n\n## Path B — Gemini / ElevenLabs / Kokoro (TTS → transcription)\n\nThese adapters supply audio without word data. The shared audio engine runs\ntranscription automatically when timings are absent. For Gemini, use the\nrequest in [Text to speech](tts.md#gemini-narration), then consume\n`audio_meta.json` → `voices[].words`.\n\nFor a standalone local Kokoro generation, generate the audio, then transcribe:\n\n```bash\nnpx hyperframes tts script.txt --voice af_heart --output narration.wav\nnpx hyperframes transcribe narration.wav --model small.en   # voice af_heart is American English\n```\n\nWhisper extracts precise word boundaries from the generated audio, so caption timing matches delivery without hand-tuning. Match `--model` to the voice's language (use `small.en` for `a`/`b` prefixes, `small --language <code>` otherwise). Then consume `transcript.json` via the caption references in `captions/`.\n\nFor Gemini, verify that transcription preserved the script, especially names,\nnumbers, and delivery pauses. If `words` is empty, resolve the transcription\nfailure before captioning. Generate and align again after changing the read.\n\nArchive v1.0.74: 91 files, 211147 bytes\n\nFiles: audio/assets/sfx/CREDITS.md (1183b), audio/assets/sfx/manifest.json (3721b), audio/references/bgm.md (8141b), audio/references/captions/authoring.md (9545b), audio/references/captions/motion.md (5656b), audio/references/captions/transcript-handling.md (5439b), audio/references/remove-background.md (8472b), audio/references/requirements.md (4347b), audio/references/sfx.md (3678b), audio/references/transcribe.md (2783b), audio/references/tts-to-captions.md (1755b), audio/references/tts.md (14011b), audio/scripts/audio.mjs (13474b), audio/scripts/audio.test.mjs (5930b), audio/scripts/gemini-pipeline.test.mjs (4932b), audio/scripts/heygen-tts.mjs (4388b), audio/scripts/heygen-tts.test.mjs (1408b), audio/scripts/heygen-voice.mjs (3868b), audio/scripts/heygen-voice.test.mjs (8536b), audio/scripts/lib/audio-meta.mjs (1166b), audio/scripts/lib/audio-meta.test.mjs (4586b), audio/scripts/lib/bgm-volume.mjs (236b), audio/scripts/lib/bgm.mjs (11558b), audio/scripts/lib/bgm.test.mjs (2639b), audio/scripts/lib/concurrency.mjs (527b), audio/scripts/lib/concurrency.test.mjs (1654b), audio/scripts/lib/gemini-auth_test.py (4430b), audio/scripts/lib/gemini-auth.mjs (1610b), audio/scripts/lib/gemini-auth.py (1981b), audio/scripts/lib/gemini-auth.test.mjs (3097b), audio/scripts/lib/gemini-tts.mjs (4711b), audio/scripts/lib/gemini-tts.test.mjs (9018b), audio/scripts/lib/heygen.mjs (10925b), audio/scripts/lib/heygen.test.mjs (10191b), audio/scripts/lib/host-audio.mjs (1470b), audio/scripts/lib/host-audio.test.mjs (4036b), audio/scripts/lib/media-record.mjs (4028b), audio/scripts/lib/media-record.test.mjs (7057b), audio/scripts/lib/python.mjs (3290b), audio/scripts/lib/python.test.mjs (5282b), audio/scripts/lib/sfx.mjs (6577b), audio/scripts/lib/sfx.test.mjs (4698b), audio/scripts/lib/tts.mjs (17162b), audio/scripts/lib/tts.spawn.test.mjs (5249b), audio/scripts/lib/tts.test.mjs (8282b), audio/scripts/lyria-recipe.py (5023b), audio/scripts/wait-bgm.mjs (5813b), audio/scripts/wait-bgm.test.mjs (3336b), luts/index.json (1997b), luts/README.md (1087b), references/audio.md (2234b), references/grading.md (6892b), references/media-treatment-recipes.md (35868b), references/media-treatments.md (13920b), references/memory.md (3498b), references/meta.md (6780b), references/operations.md (14823b), references/resolve.md (9605b), references/setup-providers.md (8542b), references/telemetry-dashboard.md (4200b), scripts/audio-duck.mjs (4069b), scripts/compatibility.test.mjs (3804b), scripts/dither.mjs (9969b), scripts/dither.test.mjs (4723b), scripts/eval.mjs (14891b), scripts/lib/config-lock.mjs (1520b), scripts/lib/cutlist.mjs (6095b), scripts/lib/duck.mjs (3266b), scripts/lib/error-diffusion.mjs (6631b), scripts/lib/index-gen.mjs (1924b), scripts/lib/manifest.mjs (9313b), scripts/lib/media-fetch.mjs (2838b), scripts/lib/media-home.mjs (730b), scripts/lib/npx-sync.mjs (2328b), scripts/lib/parakeet-words.mjs (1149b), scripts/lib/prefs-store.mjs (6617b), scripts/lib/recipe-store.mjs (13033b), scripts/lib/telemetry.mjs (7107b), scripts/lib/transcriptCutFade.mjs (911b), scripts/lib/words.mjs (764b)\n\nFile v1.0.74:SKILL.md\n\n---\nname: media-use\ndescription: Agent Media OS for a HyperFrames project. Resolve BGM, SFX, image, icon, brand logo, voice, color grade, or LUT into a frozen local file or paste-ready block + ledger record (one verb, `resolve`); generate via TTS / music / image models when the catalog misses; produce voiceover, transcription, captions, and background removal through one shared audio engine; operate on media (cut / reframe / transform); and reuse assets across projects. Also use for vague feedback that real footage looks dark, flat, boring, should feel retro/camcorder/print/ASCII, needs privacy, or needs a media reveal. When the host app provides its own music or sound-effect tools, use those for music and sound effects; `resolve --type bgm|sfx` needs the heygen CLI. When `HEYGEN_API_BASE` is set, HeyGen calls go through that host with no CLI sign-in.\n---\n\n**Plugin installs:** Before setup or freshness commands, follow [plugin execution rules](../hyperframes/references/plugin-installation.md) when this skill is inside a HyperFrames plugin. Standalone installs keep the update instructions below.\n\n# media-use\n\nThe media OS for HyperFrames: resolve · generate · operate · remember — every media type, one skill, zero context noise.\n\nOnly when `HEYGEN_API_BASE` is set in your environment (a host app set it and pays for HeyGen with its own key): HeyGen media is already paid for. Do not ask the person to install or sign in to the `heygen` CLI and do not offer its OAuth allowance; catalog search, TTS and avatar calls here go through the host. When that same host also gives you its own HeyGen tools, use those first. When a call through the host is refused, tell the person the host's message as written (it names the fix, such as adding or replacing the key in the app's Settings) and stop; do not switch to another provider unless they ask.\n\nFirst run otherwise (no `HEYGEN_API_BASE`), when you will use HeyGen media (catalog search, TTS, avatar video): install and sign in to the `heygen` CLI (the free-usage path), then verify with `npx hyperframes media-use resolve --doctor`. Setup and providers: `references/setup-providers.md`.\n\nMusic and sound effects inside a host app: when the app you run in gives you its own music or sound-effect tools, use those. Without `HEYGEN_API_BASE`, `resolve --type bgm` and `--type sfx` search the HeyGen catalog through the `heygen` CLI; without it they fail and say so (`sfx` still answers from its bundled library).\n\nWithout `HEYGEN_API_BASE`, before generating a voiceover or an avatar video, tell the person: signing in to the heygen CLI with OAuth (`heygen auth login --oauth`) gives a free allowance for TTS voiceover and avatar videos, while an API key bills API credits.\n\n## Resolve — the one verb\n\n```bash\nnpx hyperframes media-use resolve --type <type> --intent \"<description>\" --project <dir>\n```\n\nReturns one line: `resolved <id> → <path> (<type>, <metadata>)`. All search noise stays on disk.\n\n| Type    | One-line intent                                                                  |\n| ------- | -------------------------------------------------------------------------------- |\n| `bgm`   | background music (HeyGen catalog via the `heygen` CLI, 10k+ tracks)              |\n| `sfx`   | sound effects (bundled 19-file library + catalog via the `heygen` CLI)           |\n| `image` | photos, backgrounds (HeyGen asset search, 75k+ vectors)                          |\n| `icon`  | icons, symbols (transparent)                                                     |\n| `logo`  | official brand marks (theSVG → GitHub avatar → favicon; never redrawn)           |\n| `voice` | TTS voiceover (HeyGen free-usage path; optional local Kokoro)                    |\n| `grade` | measured correction candidate; broad polish/stylization follows Media Treatments |\n| `lut`   | user-provided or explicitly chosen reusable validated `.cube` file               |\n\nBefore resolving fresh, list reusable candidates with `--candidates` and judge fit yourself — reuse rules, all flags, ingest (`--from`), and adopt are in `references/resolve.md`.\n\n## Treat broad visual feedback as media intent\n\nWhen a user explicitly asks to fix, polish, stylize, obscure, emphasize, or\nreveal photographic media, read `references/media-treatments.md` even if they\ndo not name color grading or an effect. Inspect the real `<img>`/`<video>`,\nchoose one primary intent, then use deterministic persistence and verification.\nUse a matching recipe as an optional tested seed, or inspect\n`hyperframes media-treatment --capabilities --json`, then request one relevant\nfamily/effect with `--capability <id>` and assemble a custom treatment from\ncanonical controls. Never load `--all` for ordinary authoring. A treatment may\ncompose correction, a preset, finishing, compatible shader effects, supported\nkeyframes, and optional Registry overlays. Add only source-justified bounded\ntuning and compatible parts, never effects merely to make the result look more\nsophisticated. Persist the final combined payload with\n`hyperframes media-treatment`.\n\nUse one progressively escalating workflow. For video, inspect one labeled\nearly/middle/late contact sheet rather than reading frames separately. Apply one\ncandidate and inspect one after-sheet for ordinary correction or polish.\nEscalate to individual frames or moving draft evidence only when the result is\nambiguous, temporal, stylized, LUT-based, HDR/LOG-sensitive, private, or\nbrand-critical.\n\nFor ordinary correction or polish, persist the final treatment's\npreset/adjustment JSON.\nDo not generate a `.cube` LUT merely to encode exposure, shadows, contrast, or\nwarmth. Use a LUT only when the user supplies one or the selected treatment\nexplicitly owns one. `resolve --type grade --for ... --analyze` is measurement\nevidence, not permission to replace the chosen treatment with a generated LUT.\nDo not recreate supported vignette, grain, blur, pixelate, color, or treatment\neffects with CSS/SVG overlays; that bypasses Studio controls and the canonical\npreview/render shader path.\n\n## Be proactive — run a media opportunity pass\n\nThe human usually can't tell which media would lift the piece. You can. When you build or review a composition, do **one** grounded scan and then **ask once** — don't silently add, and don't nag per asset.\n\nSurface an opportunity only when a concrete signal is present:\n\n| Signal detected                                          | Offer                                                                                                  |\n| -------------------------------------------------------- | ------------------------------------------------------------------------------------------------------ |\n| On-screen text / a script with no voiceover              | TTS voiceover (audio engine)                                                                           |\n| Emoji or a `<div>` styled as an icon                     | resolve real `icon`s                                                                                   |\n| Image that is a placeholder, tiny, or upscaled-looking   | a better `image` (and/or upscale — see `references/operations.md`)                                     |\n| Hard scene cuts / transitions with no sound              | transition `sfx`                                                                                       |\n| A piece over ~10s with no music bed                      | `bgm`                                                                                                  |\n| Footage that reads under/over-exposed or color-cast      | a corrective grade (inspect it with `hyperframes media-treatment --selector '#hero' --analyze --json`) |\n| Photographic media that feels visually flat or off-topic | one specific source-appropriate preset or custom treatment, with the intended target named             |\n| A meaningful media entrance/reveal that feels static     | one supported seek-safe treatment animation; preserve color unless the request also justifies a preset |\n\nRules that keep this a help, not nagware: **grounded, not generic** (no signal → no suggestion); **opinionated + concrete** (propose the specific fix with defaults chosen — the human approves **all / some / none**); **once per project** (one consolidated ask; respect \"leave it\"); **surface, never silently mutate** (color grades especially: propose and preview — a gray-world \"correction\" ruins an intentional sunset or neon look).\n\n## Where to look — read only the file your task needs\n\n| Task                                                                      | Read                             |\n| ------------------------------------------------------------------------- | -------------------------------- |\n| resolve / reuse / adopt / ingest, flags, cascade, inventory               | `references/resolve.md`          |\n| color grading, LUTs, smart grade (`--for`), grade-compare                 | `references/grading.md`          |\n| voiceover / TTS, music, SFX, captions, transcription (audio engine)       | `references/audio.md`            |\n| cut / reframe / transform existing media, exact error diffusion, HEVC     | `references/operations.md`       |\n| source-aware creative treatments, realtime effects, overlays, reveals     | `references/media-treatments.md` |\n| install + auth, provider table, RAM ladders, `--local-only`, `--provider` | `references/setup-providers.md`  |\n| remembered preferences + frozen recipes (user memory)                     | `references/memory.md`           |\n| ownership matrix, usage stats, telemetry, privacy (maintainer-facing)     | `references/meta.md`             |\n\nFile v1.0.74:luts/README.md\n\n# LUT library (authoring)\n\n`index.json` is the agent-consumed catalog of color-grade looks. Each entry resolves\non demand — no `.cube` bodies are committed to the repo.\n\nEach look has:\n\n- `id`, `description`, `tags`, `intensity` — matching + application metadata.\n- `url` (optional) — a hosted `.cube` downloaded, validated, and frozen at resolve\n  time, exactly like bgm/image assets.\n- `params` (optional) — a deterministic `buildCube` spec used offline (`--local-only`)\n  or as a fallback if the `url` download/validation fails.\n\nAn entry needs at least one of `url` or `params`; prefer both (CDN url with a params\nfallback) so resolution is never blocked on the network.\n\n## Hosting a new look (operators)\n\n1. Generate the `.cube` (e.g. `resolve -t lut --params '{...}'` or a graded export).\n2. Upload it to the public CDN origin bucket:\n\n   ```\n   aws s3 cp <id>.cube s3://heygen-public/luts/<id>.cube\n   ```\n\n   It is then served at `https://static.heygen.ai/luts/<id>.cube` (CloudFront).\n\n3. Add an entry to `index.json` with that `url` (and ideally a `params` fallback).\n\nFile v1.0.74:_meta.json\n\n{\n  \"ownerId\": \"kn77d06grj6xqp3dqwkk4bavhn89pegt\",\n  \"slug\": \"media-use\",\n  \"version\": \"1.0.74\",\n  \"publishedAt\": 1791474081947\n}\n\nFile v1.0.74:audio/references/bgm.md\n\n# Background music (BGM)\n\nOne music bed per composition, produced by the shared audio engine (`scripts/audio.mjs` → `scripts/lib/bgm.mjs`). Two routes, chosen by the engine's one switch — whether a HeyGen credential is present:\n\n- **HeyGen retrieval — the default when credentialed.** Search HeyGen's music catalog by mood, download the top track. No generation; same `~/.heygen` / `$HEYGEN_API_KEY` credential as TTS.\n- **Local generation (Lyria → MusicGen) — the fallback when there is no credential** (or when asked for explicitly). Generate a WAV from a mood prompt. There is **no `npx hyperframes bgm` command**; the engine spawns `scripts/lyria-recipe.py` or an inline MusicGen script directly.\n\n> **Run the Preflight first — no credential is not a green light to silently generate locally.** Before generating, complete the sign-in **Preflight** (see `../../SKILL.md` → Preflight): run `npx hyperframes auth status`, recommend signing in, and **STOP for the user's choice** (sign in for HeyGen's music library, or continue offline with local generation). This applies to a one-off \"generate a BGM\" request just as much as inside a full workflow.\n\n## Driving it from the request\n\n`audio_request.json` → `bgm: { mode?, query?, prompt? }`:\n\n- **`mode`** — `retrieve | generate | none`. Omit for **auto** (retrieve when credentialed, else generate). An **explicit** `retrieve` is strict: no credential ⇒ skip, never a detached generate (so a caller with no `wait-bgm` step, e.g. product-launch, can't get a pending job it won't await).\n- **`query`** — the mood, used for retrieval and as a fallback prompt seed (e.g. a storyboard's `music:` field, falling back to `message` → `arc` → `\"calm cinematic underscore\"`).\n- **`prompt`** — an explicit full prompt for generation; omit and the engine infers one (see Mood inference). Optional `blob` / `archetype` / `arc` feed that inference.\n\nBoth routes keep a file of yours already at the output name: the engine writes the next free name (`track-2.mp3`), reports it as an anomaly, and `bgm.path` carries the real path.\n\n## HeyGen retrieval (default)\n\n`searchSounds(query, \"music\", { limit: 5 })` → `GET /audio/sounds?query=<mood>&type=music&limit=5`. Take the top result (ranked by `score`), download its presigned `audio_url` → `assets/bgm/track.mp3`. Synchronous. No match → skip (BGM is optional; never fail the render over it). Cue written to `audio_meta.json`:\n\n```jsonc\n{\n  \"path\": \"assets/bgm/track.mp3\",\n  \"volume\": 0.12,\n  \"mode\": \"retrieve\",\n  \"query\": \"calm cinematic underscore\",\n  \"duration_s\": 42.0,\n}\n```\n\n`volume` comes from the engine's `bgmDefaultVolume()`: `BGM_BED_VOLUME` (currently `0.12` ≈ -18 dB — a bed under the voice) under narration, `BGM_SILENT_VOLUME` (currently `0.9`) for a silent film (no voice). Tune those constants in `scripts/lib/bgm.mjs`, not call sites. An explicit `volume` in `audio_meta.json` always overrides this default. `bgm_pending` is `false` — the file is on disk when the engine returns.\n\nFor short launch videos, do not assume the beginning of the retrieved file is the best edit point. Check the opening against later five-second sections. If the track starts with a quiet build but a later section has a stronger, clean musical entrance, trim from that section and apply a short fade-in and longer fade-out. Repeat this check whenever the composition duration changes; the final music file must cover the full cut without a silent tail.\n\n## Local generation (fallback) — Lyria → MusicGen\n\nSpawned **detached** so voice work isn't blocked; `audio_meta.bgm_pending: true` and `bgm_pid` / `bgm_log` are set until it finishes. **Run `scripts/wait-bgm.mjs` before assembling** — it polls the output file / process / log, detects crashes, and writes `bgm_status.json` (`status: ready | failed | timeout | disabled`). A failed/absent track is simply omitted; it never blocks voice/SFX.\n\n| Order | Provider                             | Env / deps                                                                            | Speed                                   | Quality                     |\n| ----- | ------------------------------------ | ------------------------------------------------------------------------------------- | --------------------------------------- | --------------------------- |\n| 1     | Google Lyria RealTime                | `$GEMINI_API_KEY` or `$GOOGLE_API_KEY` + `google-genai` (auto-installed on demand)    | Real-time stream (≈ requested duration) | Production-grade            |\n| 2     | MusicGen (`facebook/musicgen-small`) | Python `transformers + torch + soundfile + numpy` (~300 MB first run; auto-installed) | Slow on CPU; fast on Apple MPS / CUDA   | Decent; prompt-only control |\n\nOutput → `assets/bgm/track.wav`, target = total voice duration. MusicGen generates **one** seed clip (≤28–30s, under the decoder's positional limit) then crossfade-loops it up to the target (or trims down if shorter), avoiding per-segment seams. Backend selection is by what can actually **run**: Lyria only when `import google.genai` succeeds, else MusicGen; if neither can be made to run, BGM is skipped (voice + SFX still render).\n\n## Mood inference (the generate prompt)\n\n`inferBgmPrompt()` in `scripts/lib/bgm.mjs`: an explicit `prompt` wins; otherwise industry-keyword **base** → narrative-**archetype** shape → emotional-**arc** tiebreaker.\n\n| Match in `blob` / `query`                              | Base prompt                                                                 | BPM |\n| ------------------------------------------------------ | --------------------------------------------------------------------------- | --- |\n| `crypto / nft / web3 / defi / token / blockchain`      | atmospheric electronic, deep bass, futuristic synths, restrained percussion | 100 |\n| `finance / fintech / bank / payment / invest / wealth` | calm cinematic, soft strings, subtle piano, restrained percussion           | 92  |\n| `creative / agency / design / studio / art / brand`    | playful electronic, warm pads, light percussion                             | 115 |\n| _(default: SaaS / tech / platform)_                    | uplifting corporate tech, bright modern piano with synth pads               | 108 |\n\nArchetype then reshapes the arc — PAS → \"MINOR to MAJOR\" build; BAB / future-pacing → aspirational rising; feature-cascade → +10 BPM driving; demo-loop → −8 BPM minimal. The emotional arc breaks remaining ties (tension→relief, excitement, trust/reassurance).\n\n## Lyria knobs (direct recipe use)\n\nThe engine bakes BPM / scale into the **prompt text** (via the inference above) and passes only `--output` / `--duration` / `--prompt` to the recipe. If you invoke `scripts/lyria-recipe.py` directly you can also set: `--bpm` (90–110 calm, 110–130 energetic), `--brightness` (0–1, ≥0.7 promotional), `--density` (0–1, higher = fuller), `--scale` (`MAJOR` / `MINOR` / `PENTATONIC` / …), `--negative-prompt` (styles to exclude). MusicGen ignores all of these — put the mood in the prompt.\n\n## Failure modes\n\n| Failure                                       | Behavior                                                                                 |\n| --------------------------------------------- | ---------------------------------------------------------------------------------------- |\n| No music match (retrieve)                     | `bgm: null`, anomaly logged. Render proceeds without BGM.                                |\n| Explicit `retrieve`, no credential            | Skipped (no silent generate fallback). Use `mode: generate` or omit `mode` for auto.     |\n| Neither Lyria nor MusicGen can run (generate) | `bgm` disabled with a `pip install …` hint. Voice + SFX still render.                    |\n| Generate still rendering at assemble time     | `bgm_pending: true`; `wait-bgm.mjs` waits/checks and writes `bgm_status.json` first.     |\n| Generate crashed                              | `wait-bgm.mjs` → `bgm_status.json { status: \"failed\" }`; the `<audio>` track is omitted. |\n\nBGM failure never blocks a render.\n\nFile v1.0.74:audio/references/captions/authoring.md\n\n# Captions\n\n<!-- registry-items: allow=max-width,data-composition-src,hyperframes-registry,blend-mode,caption-style,font-family -->\n\n**The live search is the source of truth for what the registry has.** The table(s) below are a hand-maintained sample and under-cover by design: run `npx hyperframes catalog --query \"<what you want>\" --json` — it needs nothing installed — before concluding the registry lacks something. Item names here are checked against `registry/registry.json` by `bun run lint:skills`.\n\nBefore authoring: confirm the transcript came from the right Whisper model. CLI default `small.en` silently translates non-English audio — see [`../transcribe.md`](../transcribe.md) → \"Language Rule\" and [`transcript-handling.md`](transcript-handling.md) for the mandatory quality check.\n\nAnalyze spoken content to determine caption style. If user specifies a style, use that. Otherwise, detect tone from the transcript.\n\n## Transcript Source\n\n```json\n[\n  { \"id\": \"w0\", \"text\": \"Hello\", \"start\": 0.0, \"end\": 0.5 },\n  { \"id\": \"w1\", \"text\": \"world.\", \"start\": 0.6, \"end\": 1.2 }\n]\n```\n\n`id` (`w0`, `w1`, …) is the stable reference for per-word overrides and is added by `hyperframes transcribe`. It's optional for backwards compatibility with hand-authored transcripts. See [`../transcribe.md`](../transcribe.md) → \"Output Shape\" for how this is produced, and [`transcript-handling.md`](transcript-handling.md) for cleanup before consumption.\n\n## Style Detection (When No Style Specified)\n\nRead the full transcript before choosing. Four dimensions:\n\n**1. Visual feel** — corporate→clean; energetic→bold; storytelling→elegant; technical→precise; social→playful.\n\n**2. Color palette** — dark+bright for energy; muted for professional; high contrast for clarity; one accent color.\n\n**3. Font mood** — heavy/condensed for impact; clean sans for modern; rounded for friendly; serif for elegance.\n\n**4. Animation character** — scale-pop for punchy; gentle fade for calm; word-by-word for emphasis; typewriter for technical.\n\n## Per-Word Styling\n\nScan for words deserving distinct treatment:\n\n- **Brand/product names** — larger size, unique color\n- **ALL CAPS** — scale boost, flash, accent color\n- **Numbers/statistics** — bold weight, accent color\n- **Emotional keywords** — exaggerated animation (overshoot, bounce)\n- **Call-to-action** — highlight, underline, color pop\n- **Marker highlight** — for beyond-color emphasis (highlight sweep, circle, burst, scribble, sketchout), see `hyperframes-animation/rules/css-marker-patterns.md`.\n\n## Script-to-Style Mapping\n\n| Tone         | Font mood                | Animation                          | Color                       | Size    |\n| ------------ | ------------------------ | ---------------------------------- | --------------------------- | ------- |\n| Hype/launch  | Heavy condensed, 800-900 | Scale-pop, back.out(1.7), 0.1-0.2s | Bright on dark              | 72-96px |\n| Corporate    | Clean sans, 600-700      | Fade+slide, power3.out, 0.3s       | White/neutral, muted accent | 56-72px |\n| Tutorial     | Mono/clean sans, 500-600 | Typewriter/fade, 0.4-0.5s          | High contrast, minimal      | 48-64px |\n| Storytelling | Serif/elegant, 400-500   | Slow fade, power2.out, 0.5-0.6s    | Warm muted tones            | 44-56px |\n| Social       | Rounded sans, 700-800    | Bounce, elastic.out, word-by-word  | Playful, colored pills      | 56-80px |\n\n## Word Grouping\n\n- **High energy:** 2-3 words. Quick turnover.\n- **Conversational:** 3-5 words. Natural phrases.\n- **Measured/calm:** 4-6 words. Longer groups.\n\nBreak on sentence boundaries, 150ms+ pauses, or max word count.\n\n## Positioning\n\n- **Landscape (1920x1080):** Bottom 80-120px, centered\n- **Portrait (1080x1920):** Lower middle ~600-700px from bottom, centered\n- Never cover the subject's face\n- `position: absolute` — never relative\n- One caption group visible at a time\n\n## Text Overflow Prevention\n\nUse `window.__hyperframes.fitTextFontSize()`:\n\n```js\nvar result = window.__hyperframes.fitTextFontSize(group.text.toUpperCase(), {\n  fontFamily: \"Outfit\",\n  fontWeight: 900,\n  maxWidth: 1600,\n});\nel.style.fontSize = result.fontSize + \"px\";\n```\n\nOptions: `maxWidth` (1600 landscape, 900 portrait), `baseFontSize` (78), `minFontSize` (42), `fontWeight`, `fontFamily`, `step` (2).\n\nCSS safety nets: `max-width` on container, `overflow: visible` (**not** `hidden` — hidden clips scaled emphasis words and glow effects), `position: absolute`, explicit `height`. When per-word styling uses `scale > 1.0`, compute `maxWidth = safeWidth / maxScale` to leave headroom.\n\n**Container pattern:** Full-width absolute container, centered. Do **not** use `left: 50%; transform: translateX(-50%)` — causes clipping at composition edges.\n\n## Caption Exit Guarantee\n\nEvery group **must** have a hard kill after exit animation:\n\n```js\ntl.to(groupEl, { opacity: 0, scale: 0.95, duration: 0.12, ease: \"power2.in\" }, group.end - 0.12);\n// `tl.set` is an instant flip, not a tween — safe to set `visibility` here (core's \"no animating\n// visibility\" rule applies to tweens, which can't smoothly interpolate non-numeric values anyway).\ntl.set(groupEl, { opacity: 0, visibility: \"hidden\" }, group.end);\n```\n\nSelf-lint after building timeline — place **before** `window.__timelines[id] = tl` so it runs at composition init:\n\n```js\nGROUPS.forEach(function (group, gi) {\n  var el = document.getElementById(\"cg-\" + gi);\n  if (!el) return;\n  tl.seek(group.end + 0.01);\n  var computed = window.getComputedStyle(el);\n  if (computed.opacity !== \"0\" && computed.visibility !== \"hidden\") {\n    console.warn(\n      \"[caption-lint] group \" + gi + \" still visible at t=\" + (group.end + 0.01).toFixed(2) + \"s\",\n    );\n  }\n});\ntl.seek(0);\n```\n\n## Pre-Built Caption Components\n\nBefore building caption styles from scratch, check the registry — 15 ready-to-use caption components cover the most common styles. Install with `npx hyperframes add <name>` and wire as a sub-composition via `data-composition-src` (see `hyperframes-registry`).\n\n```bash\nnpx hyperframes catalog --tag caption-style   # list all caption components\nnpx hyperframes add caption-highlight         # install a specific one\n```\n\n| Style                     | Component                    | Best for                     |\n| ------------------------- | ---------------------------- | ---------------------------- |\n| TikTok-style highlight    | `caption-highlight`          | Social, high-energy          |\n| Karaoke pill              | `caption-pill-karaoke`       | Music, lyric videos          |\n| Cinematic editorial       | `caption-editorial-emphasis` | Documentary, storytelling    |\n| Glitch / cyber            | `caption-glitch-rgb`         | Tech, gaming                 |\n| Full-screen slam          | `caption-kinetic-slam`       | Hype, announcements          |\n| Neon glow                 | `caption-neon-glow`          | Night, club, neon aesthetics |\n| Neon accent (multi-color) | `caption-neon-accent`        | Colorful, playful            |\n| Wipe reveal               | `caption-clip-wipe`          | Clean, modern                |\n| Gradient fill             | `caption-gradient-fill`      | Vibrant, eye-catching        |\n| Matrix decode             | `caption-matrix-decode`      | Sci-fi, tech reveals         |\n| Emoji pop                 | `caption-emoji-pop`          | Social, casual               |\n| Parallax layers           | `caption-parallax-layers`    | Depth, cinematic             |\n| Particle burst            | `caption-particle-burst`     | Celebration, impact keywords |\n| Lava texture              | `caption-texture`            | Bold, dramatic               |\n| Weight shift              | `caption-weight-shift`       | Elegant, typographic         |\n\nRelated: `caption-blend-difference` (tagged `text` / `blend-mode`, not `caption-style`, so it won't appear under the filter above) auto-inverts text against any background via `mix-blend-mode: difference` — useful when the background is busy or unpredictable.\n\nBrowse all with previews: [hyperframes.heygen.com/catalog](https://hyperframes.heygen.com/catalog)\n\nCaption components ship with transparent backgrounds — they're pure overlays. If the underlying video is bright or busy, add a contrast layer (e.g. a semi-transparent dark div) in the host composition beneath the caption sub-composition, not inside the component itself.\n\n## Further References\n\n- [`motion.md`](motion.md) — karaoke, marker effects, audio-reactive modulation, scatter exits.\n- [`transcript-handling.md`](transcript-handling.md) — input formats, quality checks, cleaning, external API fallback.\n- `hyperframes-animation/rules/css-marker-patterns.md` — marker highlighting (deterministic, fully seekable).\n\n## Constraints\n\n- Deterministic. No `Math.random()`, no `Date.now()`.\n- Sync to transcript timestamps.\n- One group visible at a time.\n- Every group must have a hard `tl.set` kill at `group.end`.\n- Fonts: the compiler auto-embeds only its **built-in mapped set** (Inter, Roboto, Montserrat, …) — for those, just declare `font-family` in CSS. Any **other** font (a brand/custom font like `TT Norms Pro`, or a non-Latin CJK/Devanagari family) is **not** auto-supplied: it needs an `@font-face` pointing at a real `.woff2` shipped with the project, or the text silently falls back to a generic font in the render. Don't assume a `font-family` you can see locally will render — the render machine is a clean headless Chrome with no installed fonts.\n\nFile v1.0.74:audio/references/captions/motion.md\n\n# Dynamic Caption Techniques\n\nYou are here because SKILL.md told you to read this file before writing animation code. Pick your technique combination from the table below based on the energy level you detected from the transcript, then implement using standard GSAP patterns.\n\n## Technique Selection by Energy\n\n| Energy level | Highlight                             | Exit                | Cycle pattern                             |\n| ------------ | ------------------------------------- | ------------------- | ----------------------------------------- |\n| High         | Karaoke with accent glow + scale pop  | Scatter or drop     | Alternate highlight styles every 2 groups |\n| Medium-high  | Karaoke with color pop                | Scatter or collapse | Alternate every 3 groups                  |\n| Medium       | Karaoke (subtle, white only)          | Fade + slide        | Alternate every 3 groups                  |\n| Medium-low   | Karaoke (minimal scale change)        | Fade                | Single style, vary ease per group         |\n| Low          | Karaoke (warm tones, slow transition) | Collapse            | Alternate every 4 groups                  |\n\n**All energy levels use karaoke highlight as the baseline.** The difference is intensity — high energy gets accent color + glow + 15% scale pop on active words, low energy gets a gentle white shift with 3% scale.\n\n**Emphasis words always break the pattern.** When a word is flagged as emphasis (emotional keyword, ALL CAPS, brand name), give it a stronger animation than surrounding words (larger scale, accent color, overshoot ease). This creates contrast.\n\n**Marker highlight modes add a visual layer on top of karaoke.** For emphasis words that need more than color/scale, add a marker-style effect: highlight sweep, circle, burst, scribble, or sketchout. See `hyperframes-animation/rules/css-marker-patterns.md` for implementation details. Match mode to energy: burst for hype, circle for key terms, highlight for standard, scribble for subtle.\n\n## Audio-Reactive Captions (Mandatory for Music)\n\n**If the source audio is music (vocals over instrumentation, beats, any musical content), you MUST extract audio data and add audio-reactive animations.** This is not optional — music without audio reactivity looks disconnected. Even low-energy ballads get subtle bass pulse and treble glow.\n\nNo special wiring is needed. The group loop already iterates over every caption group to build entrance, karaoke, and exit tweens. At that point, read the audio data for each group's time range and use it to modulate the group's animation intensity with regular GSAP tweens.\n\n```js\n// Load audio data inline (same pattern as TRANSCRIPT)\nvar AUDIO = JSON.parse(audioDataJson); // { fps, totalFrames, frames: [{ bands: [...] }] }\n\nGROUPS.forEach(function (group, gi) {\n  var groupEl = document.getElementById(\"cg-\" + gi);\n  if (!groupEl) return;\n\n  // Read peak energy for this group's time range\n  var startFrame = Math.floor(group.start * AUDIO.fps);\n  var endFrame = Math.min(Math.floor(group.end * AUDIO.fps), AUDIO.totalFrames - 1);\n  var peakBass = 0;\n  var peakTreble = 0;\n  for (var f = startFrame; f <= endFrame; f++) {\n    var frame = AUDIO.frames[f];\n    if (!frame) continue;\n    peakBass = Math.max(peakBass, frame.bands[0] || 0, frame.bands[1] || 0);\n    peakTreble = Math.max(peakTreble, frame.bands[6] || 0, frame.bands[7] || 0);\n  }\n\n  // Modulate entrance — louder groups enter bigger and glowier\n  tl.to(\n    groupEl,\n    {\n      scale: 1 + peakBass * 0.06,\n      textShadow:\n        \"0 0 \" + Math.round(peakTreble * 12) + \"px rgba(255,255,255,\" + peakTreble * 0.4 + \")\",\n      duration: 0.3,\n      ease: \"power2.out\",\n    },\n    group.start,\n  );\n\n  // Reset at exit so audio-driven values don't persist\n  tl.set(groupEl, { scale: 1, textShadow: \"none\" }, group.end - 0.15);\n});\n```\n\nThis shapes the animation at build time, not playback time — no per-frame callbacks, no `tl.call()` loops, no async fetch timing issues. Loud groups come in with more weight and glow; quiet groups come in soft. The audio data modulates _how much_, the content determines _what_.\n\nKeep audio reactivity subtle — 3-6% scale variation and soft glow. Heavy pulsing makes text unreadable.\n\nTo generate the audio data file:\n\n```bash\npython3 skills/hyperframes-creative/scripts/extract-audio-data.py audio.mp3 --fps 30 --bands 8 -o audio-data.json\n```\n\n## Combining Techniques\n\nDon't use the same highlight animation on every group — cycle through styles using the group index. Don't combine multiple competing animations on the same word at the same timestamp. Vary techniques across groups to match the content's pace changes.\n\n**Marker highlight effects** layer well with karaoke — use karaoke for the word-by-word reveal, then add a marker effect on emphasis words only. For example: karaoke highlights each word in white, but brand names get a yellow highlight sweep and stats get a red circle. Cycle marker modes across groups for visual variety.\n\n## Runtime Tools\n\nCaption motion uses standard HyperFrames runtime APIs. Use the canonical sources:\n\n- **GSAP timeline + tween syntax** — `hyperframes-animation/adapters/gsap.md` (eases, position parameter, performance)\n- **`window.__hyperframes.fitTextFontSize` / `pretext`** — `hyperframes-core/references/determinism-rules.md` → Layout Contract (overflow prevention, per-frame text measurement)\n- **Audio data extraction** — generate via `python3 skills/hyperframes-creative/scripts/extract-audio-data.py audio.mp3 --fps 30 --bands 8 -o audio-data.json`, then load inline as shown in \"Audio-Reactive Captions\" above\n\nFile v1.0.74:audio/references/captions/transcript-handling.md\n\n# Transcript Guide\n\nFor the `transcribe` CLI invocation, the `.en`-translates-non-English rule, and whisper model selection, see [`../transcribe.md`](../transcribe.md). This file covers what to do with the resulting transcript when authoring captions: input formats, mandatory quality checks, cleaning code, external-API fallbacks.\n\n## Supported Input Formats\n\nThe CLI auto-detects and normalizes these formats:\n\n| Format                | Extension | Source                                                                      | Word-level?       |\n| --------------------- | --------- | --------------------------------------------------------------------------- | ----------------- |\n| whisper.cpp JSON      | `.json`   | `hyperframes init --video`, `hyperframes transcribe`                        | Yes               |\n| OpenAI Whisper API    | `.json`   | `openai.audio.transcriptions.create({ timestamp_granularities: [\"word\"] })` | Yes               |\n| SRT subtitles         | `.srt`    | Video editors, subtitle tools, YouTube                                      | No (phrase-level) |\n| VTT subtitles         | `.vtt`    | Web players, YouTube, transcription services                                | No (phrase-level) |\n| Normalized word array | `.json`   | Pre-processed by any tool                                                   | Yes               |\n\n**Word-level timestamps produce better captions.** SRT/VTT give phrase-level timing, which works but can't do per-word animation effects.\n\n## Transcript Quality Check (Mandatory)\n\nAfter every transcription, **read the transcript and check for quality issues before proceeding.** Bad transcripts produce nonsensical captions. Never skip this step.\n\n### What to look for\n\n| Signal                       | Example                                | Cause                                                                        |\n| ---------------------------- | -------------------------------------- | ---------------------------------------------------------------------------- |\n| Music note tokens (`♪`, `�`) | `{ \"text\": \"♪\" }` or `{ \"text\": \"�\" }` | Whisper detected music, not speech                                           |\n| Garbled / nonsense words     | \"Do a chin\", \"Get so gay\", \"huh\"       | Model misheard lyrics or background noise                                    |\n| Long gaps with no words      | 20+ seconds of only `♪` tokens         | Instrumental section — expected, but high ratio means speech is being missed |\n| Repeated filler              | Many \"huh\", \"uh\", \"oh\" entries         | Model is hallucinating on music                                              |\n| Very short word spans        | Words with `end - start < 0.05`        | Unreliable timestamp alignment                                               |\n\n### Automatic retry rules\n\n**If more than 20% of entries are `♪`/`�` tokens, or the transcript contains obvious nonsense words, the transcription failed.** Do not proceed with the bad transcript. Instead:\n\n1. **Retry with `medium.en`** if the original used `small.en` or smaller:\n   ```bash\n   npx hyperframes transcribe audio.mp3 --model medium.en\n   ```\n2. **If `medium.en` also fails** (still >20% music tokens or garbled), tell the user the audio is too noisy for local transcription and suggest:\n   - Providing lyrics manually as an SRT/VTT file\n   - Using an external API (OpenAI or Groq Whisper — see below)\n3. **Always clean the transcript** before building captions — filter out `♪`/`�` tokens and entries where `text` is a single non-word character. Only real words should reach the caption composition.\n\n### Cleaning a transcript\n\nAfter transcription (even with a good model), strip non-word entries:\n\n```js\nvar raw = JSON.parse(transcriptJson);\nvar words = raw.filter(function (w) {\n  if (!w.text || w.text.trim().length === 0) return false;\n  if (/^[♪�\\u266a\\u266b\\u266c\\u266d\\u266e\\u266f]+$/.test(w.text)) return false;\n  if (/^(huh|uh|um|ah|oh)$/i.test(w.text) && w.end - w.start < 0.1) return false;\n  return true;\n});\n```\n\nFor model-selection guidance by content type, see [`../transcribe.md`](../transcribe.md) → \"Picking a model by content type\".\n\n## Using External Transcription APIs\n\nFor the best accuracy, use an external API and import the result:\n\n**OpenAI Whisper API** (recommended for quality):\n\n```bash\n# Generate with word timestamps, then import\ncurl https://api.openai.com/v1/audio/transcriptions \\\n  -H \"Authorization: Bearer $OPENAI_API_KEY\" \\\n  -F file=@audio.mp3 -F model=whisper-1 \\\n  -F response_format=verbose_json \\\n  -F \"timestamp_granularities[]=word\" \\\n  -o transcript-openai.json\n\nnpx hyperframes transcribe transcript-openai.json\n```\n\n**Groq Whisper API** (fast, free tier available):\n\n```bash\ncurl https://api.groq.com/openai/v1/audio/transcriptions \\\n  -H \"Authorization: Bearer $GROQ_API_KEY\" \\\n  -F file=@audio.mp3 -F model=whisper-large-v3 \\\n  -F response_format=verbose_json \\\n  -F \"timestamp_granularities[]=word\" \\\n  -o transcript-groq.json\n\nnpx hyperframes transcribe transcript-groq.json\n```\n\n## If No Transcript Exists\n\n1. Check the project root for `transcript.json`, `.srt`, or `.vtt` files.\n2. If none found, run [`../transcribe.md`](../transcribe.md) — pick the starting model from \"Picking a model by content type\" there.\n3. Run the quality check above. If it fails, retry with a larger model or fall back to manual lyrics / external API.\n\nFile v1.0.74:audio/references/remove-background.md\n\n# Background Removal\n\nMake a transparent overlay (typical: a talking head over an arbitrary scene). Uses `u2net_human_seg` (Apache-2.0).\n\n```bash\nnpx hyperframes remove-background subject.mp4 -o transparent.webm          # default: VP9 + alpha\nnpx hyperframes remove-background subject.mp4 -o transparent.mov           # ProRes 4444 (editing)\nnpx hyperframes remove-background portrait.jpg -o cutout.png               # single-image cutout\nnpx hyperframes remove-background subject.mp4 -o subject.webm \\\n  --background-output plate.webm                                           # both layers, one pass\nnpx hyperframes remove-background subject.mp4 -o transparent.webm --device cpu\nnpx hyperframes remove-background --info                                   # detected providers\n```\n\n## Output Format\n\n- **`.webm` (VP9 alpha)** — default. Plug straight into `<video>` for Chrome-native transparent playback (~1 MB / 4s @ 1080p).\n- **`.mov` (ProRes 4444)** — round-trip in editors (Premiere / Resolve / DaVinci). ~50 MB / 4s.\n- **`.png`** — single-image cutout.\n\n## Quality (`--quality`)\n\nControls VP9 encoder CRF only — segmentation quality is fixed. Higher quality keeps the cutout's RGB closer to the source MP4 (important when overlaying the cutout on its own source).\n\n| Preset     | CRF | When                                          |\n| ---------- | --- | --------------------------------------------- |\n| `fast`     | 30  | Iterating, smaller files, looser color match  |\n| `balanced` | 18  | **Default**; visually identical for most uses |\n| `best`     | 12  | Master / final delivery, tightest color match |\n\n## Device (`--device`)\n\n`auto` (default) picks CoreML on Apple Silicon, CUDA when available, otherwise CPU. Force with `--device cpu | coreml | cuda`. CUDA requires `HYPERFRAMES_CUDA=1` plus a GPU-enabled `onnxruntime-node` build. Use `--info` to inspect detected providers without rendering.\n\n## Compositing patterns — pick the right one\n\nThe cutout WebM is a **re-encoded copy** of the source MP4's RGB. What sits behind it matters.\n\n| Pattern                                                  | Behind the cutout                       | Result                                                                                                          |\n| -------------------------------------------------------- | --------------------------------------- | --------------------------------------------------------------------------------------------------------------- |\n| **Cutout over a different scene** (most common)          | Static image, gradient, unrelated video | Looks great. Single RGB source for the subject.                                                                 |\n| **Cutout over its own source mp4** (text-behind-subject) | Same mp4 the cutout came from           | At `balanced` doubling is barely visible; at `fast` you'll see color shift / edge halo. Use `best` for masters. |\n| **Cutout over a different take of the same person**      | Footage of the same subject             | **Two overlapping people. Don't do this.**                                                                      |\n\n## Text-behind-subject pattern (two non-obvious rules)\n\nPutting a headline behind a presenter cutout:\n\n```html\n<video\n  src=\"presenter.mp4\"\n  id=\"bg\"\n  data-start=\"0\"\n  data-duration=\"6\"\n  data-track-index=\"0\"\n  muted\n  playsinline\n></video>\n\n<h1 id=\"headline\" style=\"z-index:2; ...\">MAKE IT IN HYPERFRAMES</h1>\n\n<div class=\"cutout-wrap\" style=\"position:absolute; inset:0; z-index:3; opacity:0\">\n  <video\n    src=\"presenter.webm\"\n    data-start=\"0\"\n    data-duration=\"6\"\n    data-track-index=\"1\"\n    muted\n    playsinline\n  ></video>\n</div>\n```\n\n```js\n// Flip the wrapper's opacity at the cut, NOT the video's\ntl.set(\".cutout-wrap\", { opacity: 1 }, 3.3);\n```\n\nTwo rules that are easy to miss:\n\n1. **Wrap the cutout `<video>` in a non-timed `<div>` and animate the wrapper's opacity, not the video element's.** The framework forces `opacity: 1` on active clips (any element with `data-start` / `data-duration`), so animating the video's opacity directly is silently overridden. The wrapper has no `data-*` attributes, so it's owned by your CSS / GSAP.\n2. **Both videos use `data-start=\"0\"` and `data-media-start=\"0\"`** so the framework decodes them in sync from t=0. Late-mounting the cutout (`data-start=3.3`) introduces a seek + warm-up that lands a frame off the base mp4 — visible as one frame of misalignment at the cut.\n\n## Layer separation (`--background-output`)\n\nEmits a **second** transparent video alongside the cutout: same source RGB, alpha is `255 - mask` instead of `mask`. The cutout has the subject opaque; the plate has the surroundings opaque (with a transparent hole where the subject was). Use it when text / graphics need to live **between** the two layers.\n\n| File                             | Alpha is…                                               | Use it for                                                       |\n| -------------------------------- | ------------------------------------------------------- | ---------------------------------------------------------------- |\n| `-o subject.webm`                | mask — subject opaque, background transparent           | Foreground layer (top)                                           |\n| `--background-output plate.webm` | inverse mask — surroundings opaque, subject transparent | Bottom layer; place text / graphics between this and the subject |\n\nBoth share the same `--quality` and run from a single inference pass — only encode cost roughly doubles. Only valid for video inputs with `.webm` / `.mov` outputs.\n\n**Hole-cut, not inpainted.** The subject region in `plate.webm` is fully transparent — composite something opaque under it to fill the hole.\n\n**Single test for whether `--background-output` is the right tool:** _will anything ever be visible through the subject's silhouette where the subject used to be?_ If no, you don't need the plate — `subject.webm` alone over a different background is enough.\n\n### Use case → right tool\n\n| Use case                                                                            | Right tool                                                                         |\n| ----------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------- |\n| Text/graphics between the cutout and the plate (this command's reason for existing) | **Hole-cut** (`--background-output`)                                               |\n| Subject onto an unrelated scene                                                     | Just `subject.webm`; ignore the plate                                              |\n| Show the room _without_ the person, alone over no other content                     | **Clean plate** — needs an inpainter (LaMa, ProPainter, E2FGVI). Not this command. |\n| Replace the subject with a different subject                                        | **Clean plate** — same as above                                                    |\n\n### Canonical 3-layer template (plate + content + cutout)\n\nShip just the two transparent layers and let arbitrary content live between them — no original mp4 needed:\n\n```html\n<!-- z=1 plate: surroundings opaque, subject silhouette transparent -->\n<video\n  src=\"plate.webm\"\n  data-start=\"0\"\n  data-duration=\"6\"\n  data-track-index=\"0\"\n  muted\n  playsinline\n></video>\n\n<!-- z=2 your content lives between the layers -->\n<h1 id=\"headline\" style=\"z-index:2; ...\">MAKE IT IN HYPERFRAMES</h1>\n\n<!-- z=3 cutout floats the subject back on top -->\n<div class=\"cutout-wrap\" style=\"position:absolute; inset:0; z-index:3\">\n  <video\n    src=\"subject.webm\"\n    data-start=\"0\"\n    data-duration=\"6\"\n    data-track-index=\"1\"\n    muted\n    playsinline\n  ></video>\n</div>\n```\n\nFunctionally equivalent to the text-behind-subject pattern above, but doesn't require shipping the original mp4 — the plate replaces it. Use this when delivering just the two transparent layers as a reusable asset.\n\n## When `remove-background` is NOT the right tool\n\nIf a user asks for \"the room **without** the person, displayed standalone\" (no subject anywhere, no compositing on top), `--background-output` is wrong — its plate has a transparent hole, not a filled-in clean plate. They need an **inpainter**: LaMa, ProPainter, or E2FGVI. Tell them this command can't do it.\n\nFile v1.0.74:audio/references/requirements.md\n\n# Requirements & Caches\n\n## Credential & key priority\n\nRun `npx hyperframes auth status` to see what's configured and which engines a workflow will use (see the skill's **Preflight** section). Keys resolve in this order — **first match wins**:\n\n| Provider                             | Resolution order (first non-empty wins)                                                                                                                                    | Local deps when used                             |\n| ------------------------------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------ |\n| **HeyGen** (TTS + BGM/SFX retrieval) | `$HEYGEN_API_KEY` → `$HYPERFRAMES_API_KEY` → `~/.heygen/credentials` (shared with heygen-cli; `$HEYGEN_CONFIG_DIR` overrides the dir; written by `hyperframes auth login`) | none (REST)                                      |\n| **ElevenLabs** (TTS fallback)        | `$ELEVENLABS_API_KEY`                                                                                                                                                      | `pip install elevenlabs`                         |\n| **Lyria** (BGM fallback)             | `$GEMINI_API_KEY` → `$GOOGLE_API_KEY`                                                                                                                                      | `pip install google-genai`                       |\n| **Kokoro** (TTS, no key)             | always — final voice fallback                                                                                                                                              | `pip install kokoro-onnx soundfile`              |\n| **MusicGen** (BGM, no key)           | always — final music fallback                                                                                                                                              | `pip install transformers torch soundfile numpy` |\n\n`hyperframes auth login` (browser OAuth) is the recommended setup: one sign-in, every project, no per-repo `.env`. An OAuth login is sent as `Authorization: Bearer`; an API key as `X-Api-Key`; both are tagged with `X-HeyGen-Source: cli`. OAuth CLI users can consume the web-plan free allowance for HeyGen TTS (10 min/month); API keys follow the normal API billing path. With no HeyGen credential, voice/BGM run fully locally (Kokoro / MusicGen) — `hyperframes auth status` and `hyperframes doctor` both report whether those local deps are installed.\n\n## Model caches & system dependencies\n\nEach command downloads its own model on first run and caches it under `~/.cache/hyperframes/`:\n\n- **TTS (HeyGen)** — no local deps; needs a HeyGen credential + `ffmpeg` on PATH (to transcode the mp3 response to `.wav`). Credential resolves like the CLI: `$HEYGEN_API_KEY` → `$HYPERFRAMES_API_KEY` → `~/.heygen/credentials` (shared with heygen-cli; run `npx hyperframes auth login`). An OAuth login is sent as `Authorization: Bearer`; an API key as `X-Api-Key`; both include `X-HeyGen-Source: cli` so the backend can apply CLI OAuth free usage.\n- **TTS (ElevenLabs)** — same as HeyGen: API key + `ffmpeg`.\n- **TTS (Kokoro)** — Kokoro-82M (~311 MB) + voices (~27 MB) in `tts/`. Requires Python 3.8+ with `kokoro-onnx` and `soundfile` (`pip install kokoro-onnx soundfile`). Non-English text also needs `espeak-ng` system-wide.\n- **BGM (Lyria)** — needs `$GEMINI_API_KEY` or `$GOOGLE_API_KEY` + `pip install google-genai`. No local model cache.\n- **BGM (MusicGen)** — `pip install transformers torch soundfile`. `facebook/musicgen-small` (~300 MB) cached under `~/.cache/huggingface/` on first run.\n- **Transcribe** — Whisper model size depending on choice (75 MB – 3.1 GB) in `whisper/`, downloaded from HuggingFace on first use. `whisper.cpp` itself is NOT bundled: the CLI resolves it from PATH, installs via Homebrew (macOS), or builds it from source with git+cmake on first use (`$HYPERFRAMES_WHISPER_PATH` overrides).\n- **Remove-background** — `u2net_human_seg` (~168 MB ONNX) in `background-removal/models/`. Peak inference RAM ~1.5 GB.\n\nRun `npx hyperframes doctor` if a command fails because of a missing dependency.\n\nFile v1.0.74:audio/references/sfx.md\n\n# Sound effects (SFX)\n\nNamed sound effects, produced by the shared audio engine (`scripts/audio.mjs` → `scripts/lib/sfx.mjs`). **Provider-gated** by the engine's one switch — whether a HeyGen credential is present, decided once (not per cue):\n\n- **HeyGen credential present → retrieve every cue** from HeyGen's audio library (`/v3/audio/sounds`, `type=sound_effects`, `min_score=0.4`). Search-and-download, **not** generation. The bundled library is NOT consulted.\n- **No credential → the bundled 21-file library** (`assets/sfx/` + `manifest.json`): match each cue name, copy the matched file into the project. Offline, deterministic, free.\n\nThere is no `npx hyperframes sfx` command. SFX is never generated — it is retrieved (online) or taken from the bundled library (offline).\n\n## Cues — request → meta\n\nEach line names the effects it wants: `lines[].sfx: [\"whoosh\", \"ui click\"]`. The engine flattens these into cues, resolves them per the switch, dedupes identical `(id, name)` pairs (the same effect named twice downloads/copies once), and writes `audio_meta.sfx[]`:\n\n```jsonc\n{\n  \"id\": \"3\",                       // joins the cue to the caller's model (frame / scene / segment)\n  \"name\": \"whoosh\",\n  \"file\": \"assets/sfx/whoosh.mp3\", // downloaded or copied, relative to project root\n  \"source\": \"heygen\" | \"local\",    // which route resolved it\n  \"offset_s\": 0,                   // delay from the line's start\n  \"duration_s\": 0.57,\n  \"volume\": 0.35                   // SFX sit UNDER voice + BGM\n}\n```\n\nA cue that matches nothing is **skipped** (recorded as an anomaly); SFX never blocks a render. Neither route replaces a file of yours already at the output name: the cue gets the next free name (`whoosh-2.mp3`), reported as an anomaly, and `file` carries the real path.\n\n## HeyGen retrieval (credentialed)\n\n`searchSounds(name, \"sound_effects\", { limit: 3, minScore: 0.4 })` → top hit → `assets/sfx/<slug>.mp3`. Results are ranked by `score` (each carries a presigned `audio_url`, `duration`, `description`). The floor is **0.4** because good SFX hits score ~0.5–0.67 — below the API's default `0.7`, which would silently drop most named cues (only whoosh/swoosh-family clears 0.7). `duration_s` comes from the result (else 1.0). Name effects concretely (`glass shatter`, not `dramatic sound`); a vague query returns a poor match.\n\n## Bundled library (no credential)\n\n21 curated files in `assets/sfx/`, indexed by `manifest.json` — `{ file, duration, description }` per key (e.g. `whoosh`, `pop`, `click`, `chime`, `riser`, `impact-bass-1`, `glitch-1`, `typing`, …). A cue name resolves by **manifest key, file basename, or slug**, so `whoosh`, `whoosh.mp3`, or `\"ui click\"` (→ slug) all match. Matched files are copied into the project's `assets/sfx/`; `duration_s` comes from the manifest, so timing is known **offline** — e.g. `riser` is 10.03s, so trigger it at `climax − 10.03s`. The manifest's `description` field carries placement hints per effect; read `assets/sfx/manifest.json` for the full set and usage.\n\n## Rules\n\n- **Volume ~0.35.** SFX must sit under narration and BGM, not fight them.\n- **No match → skip, don't fail.** A missing effect logs an anomaly and moves on; never a render blocker.\n- **Retrieval (credentialed) or bundled library (offline) — never generation.** You search HeyGen by text, or match a name against the 21-file manifest.\n- **One asset per distinct name.** Reuse across lines is deduped to a single download/copy, many cues.\n- **The switch is global, not per cue.** With a credential, retrieval handles even the long tail (effects not in the 21); without one, only the 21 bundled names resolve.\n\nFile v1.0.74:audio/references/transcribe.md\n\n# Transcription\n\nCreate normalized word-level timestamps. **Always specify `--model` explicitly** — the CLI default is `small.en`, which silently translates non-English audio into English.\n\n```bash\nnpx hyperframes transcribe audio.mp3  --model small.en             # known English\nnpx hyperframes transcribe video.mp4  --model small --language es  # known Spanish\nnpx hyperframes transcribe audio.mp3  --model small                # unknown language (auto-detect)\nnpx hyperframes transcribe subtitles.srt                           # import existing\nnpx hyperframes transcribe subtitles.vtt\nnpx hyperframes transcribe openai-response.json\n```\n\n## Language Rule (Non-Negotiable)\n\n`.en` models (`tiny.en` / `base.en` / `small.en` / `medium.en`) **translate** non-English audio into English. This silently destroys the original language.\n\n1. **Known English** → `--model small.en` (or `medium.en` for music / noisy audio)\n2. **Known non-English** → `--model small --language <iso-code>` (no `.en` suffix)\n3. **Unknown language** → `--model small` (whisper auto-detects)\n\n**CLI default is `small.en`** — do not rely on it; always pass `--model` to make the choice explicit. `--language` also filters out non-target-language segments from mixed-language audio.\n\n## Model Sizes\n\n| Model      | Size   | Speed    | When                                  |\n| ---------- | ------ | -------- | ------------------------------------- |\n| `tiny`     | 75 MB  | Fastest  | Quick previews, smoke tests           |\n| `base`     | 142 MB | Fast     | Short clips, clear audio              |\n| `small`    | 466 MB | Moderate | Default for most multilingual content |\n| `medium`   | 1.5 GB | Slow     | Music with vocals, noisy audio        |\n| `large-v3` | 3.1 GB | Slowest  | Production quality                    |\n\n### Picking a model by content type\n\n1. Speech over silence / light background → `small.en`\n2. Speech over music, or music with vocals → start with `medium.en`\n3. Produced music track (vocals + full instrumentation) → start with `medium.en`; expect to need manual lyrics or an external API ([`captions/transcript-handling.md`](captions/transcript-handling.md) → \"Using External Transcription APIs\")\n4. Multilingual → `medium` or `large-v3` (no `.en` suffix), pair with `--language`\n\n## Output Shape\n\nCompositions consume a flat array of word objects. The `id` (`w0`, `w1`, …) is added during normalization for stable references in caption overrides; optional for backwards compatibility.\n\n```json\n[\n  { \"id\": \"w0\", \"text\": \"Hello\", \"start\": 0.0, \"end\": 0.5 },\n  { \"id\": \"w1\", \"text\": \"world.\", \"start\": 0.6, \"end\": 1.2 }\n]\n```\n\nFor mandatory caption-quality checks, retry rules, and the OpenAI/Groq Whisper API import path, see `captions/transcript-handling.md`.\n\nFile v1.0.74:audio/references/tts-to-captions.md\n\n# TTS → Captions\n\nWhen no recorded voiceover exists, generate one and obtain word-level caption timing. Two paths depending on which TTS provider is in use:\n\n## Path A — HeyGen (single call, no Whisper)\n\nHeyGen returns word timestamps in the same response as the audio. Use the\nbundled REST helper (the `hyperframes tts` command is Kokoro-only):\n\n```bash\nnode skills/media-use/audio/scripts/heygen-tts.mjs \\\n  script.txt --output narration.wav --words narration.words.json\n```\n\n`narration.words.json` is already in the `[{ id, text, start, end }]` shape the captions pipeline consumes — no separate transcribe pass.\n\n## Path B — Gemini / ElevenLabs / Kokoro (TTS → transcription)\n\nThese adapters supply audio without word data. The shared audio engine runs\ntranscription automatically when timings are absent. For Gemini, use the\nrequest in [Text to speech](tts.md#gemini-narration), then consume\n`audio_meta.json` → `voices[].words`.\n\nFor a standalone local Kokoro generation, generate the audio, then transcribe:\n\n```bash\nnpx hyperframes tts script.txt --voice af_heart --output narration.wav\nnpx hyperframes transcribe narration.wav --model small.en   # voice af_heart is American English\n```\n\nWhisper extracts precise word boundaries from the generated audio, so caption timing matches delivery without hand-tuning. Match `--model` to the voice's language (use `small.en` for `a`/`b` prefixes, `small --language <code>` otherwise). Then consume `transcript.json` via the caption references in `captions/`.\n\nFor Gemini, verify that transcription preserved the script, especially names,\nnumbers, and delivery pauses. If `words` is empty, resolve the transcription\nfailure before captioning. Generate and align again after changing the read.\n\nArchive v1.0.73: 91 files, 211035 bytes\n\nFiles: audio/assets/sfx/CREDITS.md (1183b), audio/assets/sfx/manifest.json (3721b), audio/references/bgm.md (8141b), audio/references/captions/authoring.md (9545b), audio/references/captions/motion.md (5656b), audio/references/captions/transcript-handling.md (5439b), audio/references/remove-background.md (8472b), audio/references/requirements.md (4347b), audio/references/sfx.md (3678b), audio/references/transcribe.md (2783b), audio/references/tts-to-captions.md (1755b), audio/references/tts.md (14011b), audio/scripts/audio.mjs (13474b), audio/scripts/audio.test.mjs (5930b), audio/scripts/gemini-pipeline.test.mjs (4932b), audio/scripts/heygen-tts.mjs (4388b), audio/scripts/heygen-tts.test.mjs (1408b), audio/scripts/heygen-voice.mjs (3868b), audio/scripts/heygen-voice.test.mjs (8536b), audio/scripts/lib/audio-meta.mjs (1166b), audio/scripts/lib/audio-meta.test.mjs (4586b), audio/scripts/lib/bgm-volume.mjs (236b), audio/scripts/lib/bgm.mjs (11558b), audio/scripts/lib/bgm.test.mjs (2639b), audio/scripts/lib/concurrency.mjs (527b), audio/scripts/lib/concurrency.test.mjs (1654b), audio/scripts/lib/gemini-auth_test.py (4430b), audio/scripts/lib/gemini-auth.mjs (1610b), audio/scripts/lib/gemini-auth.py (1981b), audio/scripts/lib/gemini-auth.test.mjs (3097b), audio/scripts/lib/gemini-tts.mjs (4711b), audio/scripts/lib/gemini-tts.test.mjs (9018b), audio/scripts/lib/heygen.mjs (10925b), audio/scripts/lib/heygen.test.mjs (10191b), audio/scripts/lib/host-audio.mjs (1470b), audio/scripts/lib/host-audio.test.mjs (4036b), audio/scripts/lib/media-record.mjs (4028b), audio/scripts/lib/media-record.test.mjs (7057b), audio/scripts/lib/python.mjs (3290b), audio/scripts/lib/python.test.mjs (5282b), audio/scripts/lib/sfx.mjs (6577b), audio/scripts/lib/sfx.test.mjs (4698b), audio/scripts/lib/tts.mjs (17162b), audio/scripts/lib/tts.spawn.test.mjs (5249b), audio/scripts/lib/tts.test.mjs (8282b), audio/scripts/lyria-recipe.py (5023b), audio/scripts/wait-bgm.mjs (5813b), audio/scripts/wait-bgm.test.mjs (3336b), luts/index.json (1997b), luts/README.md (1087b), references/audio.md (2234b), references/grading.md (6722b), references/media-treatment-recipes.md (35868b), references/media-treatments.md (13920b), references/memory.md (3498b), references/meta.md (6780b), references/operations.md (14823b), references/resolve.md (9605b), references/setup-providers.md (8542b), references/telemetry-dashboard.md (4200b), scripts/audio-duck.mjs (4069b), scripts/compatibility.test.mjs (3804b), scripts/dither.mjs (9969b), scripts/dither.test.mjs (4723b), scripts/eval.mjs (14891b), scripts/lib/config-lock.mjs (1520b), scripts/lib/cutlist.mjs (6095b), scripts/lib/duck.mjs (3266b), scripts/lib/error-diffusion.mjs (6631b), scripts/lib/index-gen.mjs (1924b), scripts/lib/manifest.mjs (9313b), scripts/lib/media-fetch.mjs (2838b), scripts/lib/media-home.mjs (730b), scripts/lib/npx-sync.mjs (2328b), scripts/lib/parakeet-words.mjs (1149b), scripts/lib/prefs-store.mjs (6617b), scripts/lib/recipe-store.mjs (13033b), scripts/lib/telemetry.mjs (7107b), scripts/lib/transcriptCutFade.mjs (911b), scripts/lib/words.mjs (764b)\n\nFile v1.0.73:SKILL.md\n\n---\nname: media-use\ndescription: Agent Media OS for a HyperFrames project. Resolve BGM, SFX, image, icon, brand logo, voice, color grade, or LUT into a frozen local file or paste-ready block + ledger record (one verb, `resolve`); generate via TTS / music / image models when the catalog misses; produce voiceover, transcription, captions, and background removal through one shared audio engine; operate on media (cut / reframe / transform); and reuse assets across projects. Also use for vague feedback that real footage looks dark, flat, boring, should feel retro/camcorder/print/ASCII, needs privacy, or needs a media reveal. When the host app provides its own music or sound-effect tools, use those for music and sound effects; `resolve --type bgm|sfx` needs the heygen CLI. When `HEYGEN_API_BASE` is set, HeyGen calls go through that host with no CLI sign-in.\n---\n\n**Plugin installs:** Before setup or freshness commands, follow [plugin execution rules](../hyperframes/references/plugin-installation.md) when this skill is inside a HyperFrames plugin. Standalone installs keep the update instructions below.\n\n# media-use\n\nThe media OS for HyperFrames: resolve · generate · operate · remember — every media type, one skill, zero context noise.\n\nOnly when `HEYGEN_API_BASE` is set in your environment (a host app set it and pays for HeyGen with its own key): HeyGen media is already paid for. Do not ask the person to install or sign in to the `heygen` CLI and do not offer its OAuth allowance; catalog search, TTS and avatar calls here go through the host. When that same host also gives you its own HeyGen tools, use those first. When a call through the host is refused, tell the person the host's message as written (it names the fix, such as adding or replacing the key in the app's Settings) and stop; do not switch to another provider unless they ask.\n\nFirst run otherwise (no `HEYGEN_API_BASE`), when you will use HeyGen media (catalog search, TTS, avatar video): install and sign in to the `heygen` CLI (the free-usage path), then verify with `npx hyperframes media-use resolve --doctor`. Setup and providers: `references/setup-providers.md`.\n\nMusic and sound effects inside a host app: when the app you run in gives you its own music or sound-effect tools, use those. Without `HEYGEN_API_BASE`, `resolve --type bgm` and `--type sfx` search the HeyGen catalog through the `heygen` CLI; without it they fail and say so (`sfx` still answers from its bundled library).\n\nWithout `HEYGEN_API_BASE`, before generating a voiceover or an avatar video, tell the person: signing in to the heygen CLI with OAuth (`heygen auth login --oauth`) gives a free allowance for TTS voiceover and avatar videos, while an API key bills API credits.\n\n## Resolve — the one verb\n\n```bash\nnpx hyperframes media-use resolve --type <type> --intent \"<description>\" --project <dir>\n```\n\nReturns one line: `resolved <id> → <path> (<type>, <metadata>)`. All search noise stays on disk.\n\n| Type    | One-line intent                                                                  |\n| ------- | -------------------------------------------------------------------------------- |\n| `bgm`   | background music (HeyGen catalog via the `heygen` CLI, 10k+ tracks)              |\n| `sfx`   | sound effects (bundled 19-file library + catalog via the `heygen` CLI)           |\n| `image` | photos, backgrounds (HeyGen asset search, 75k+ vectors)                          |\n| `icon`  | icons, symbols (transparent)                                                     |\n| `logo`  | official brand marks (theSVG → GitHub avatar → favicon; never redrawn)           |\n| `voice` | TTS voiceover (HeyGen free-usage path; optional local Kokoro)                    |\n| `grade` | measured correction candidate; broad polish/stylization follows Media Treatments |\n| `lut`   | user-provided or explicitly chosen reusable validated `.cube` file               |\n\nBefore resolving fresh, list reusable candidates with `--candidates` and judge fit yourself — reuse rules, all flags, ingest (`--from`), and adopt are in `references/resolve.md`.\n\n## Treat broad visual feedback as media intent\n\nWhen a user explicitly asks to fix, polish, stylize, obscure, emphasize, or\nreveal photographic media, read `references/media-treatments.md` even if they\ndo not name color grading or an effect. Inspect the real `<img>`/`<video>`,\nchoose one primary intent, then use deterministic persistence and verification.\nUse a matching recipe as an optional tested seed, or inspect\n`hyperframes media-treatment --capabilities --json`, then request one relevant\nfamily/effect with `--capability <id>` and assemble a custom treatment from\ncanonical controls. Never load `--all` for ordinary authoring. A treatment may\ncompose correction, a preset, finishing, compatible shader effects, supported\nkeyframes, and optional Registry overlays. Add only source-justified bounded\ntuning and compatible parts, never effects merely to make the result look more\nsophisticated. Persist the final combined payload with\n`hyperframes media-treatment`.\n\nUse one progressively escalating workflow. For video, inspect one labeled\nearly/middle/late contact sheet rather than reading frames separately. Apply one\ncandidate and inspect one after-sheet for ordinary correction or polish.\nEscalate to individual frames or moving draft evidence only when the result is\nambiguous, temporal, stylized, LUT-based, HDR/LOG-sensitive, private, or\nbrand-critical.\n\nFor ordinary correction or polish, persist the final treatment's\npreset/adjustment JSON.\nDo not generate a `.cube` LUT merely to encode exposure, shadows, contrast, or\nwarmth. Use a LUT only when the user supplies one or the selected treatment\nexplicitly owns one. `resolve --type grade --for ... --analyze` is measurement\nevidence, not permission to replace the chosen treatment with a generated LUT.\nDo not recreate supported vignette, grain, blur, pixelate, color, or treatment\neffects with CSS/SVG overlays; that bypasses Studio controls and the canonical\npreview/render shader path.\n\n## Be proactive — run a media opportunity pass\n\nThe human usually can't tell which media would lift the piece. You can. When you build or review a composition, do **one** grounded scan and then **ask once** — don't silently add, and don't nag per asset.\n\nSurface an opportunity only when a concrete signal is present:\n\n| Signal detected                                          | Offer                                                                                                  |\n| -------------------------------------------------------- | ------------------------------------------------------------------------------------------------------ |\n| On-screen text / a script with no voiceover              | TTS voiceover (audio engine)                                                                           |\n| Emoji or a `<div>` styled as an icon                     | resolve real `icon`s                                                                                   |\n| Image that is a placeholder, tiny, or upscaled-looking   | a better `image` (and/or upscale — see `references/operations.md`)                                     |\n| Hard scene cuts / transitions with no sound              | transition `sfx`                                                                                       |\n| A piece over ~10s with no music bed                      | `bgm`                                                                                                  |\n| Footage that reads under/over-exposed or color-cast      | a corrective grade (inspect it with `hyperframes media-treatment --selector '#hero' --analyze --json`) |\n| Photographic media that feels visually flat or off-topic | one specific source-appropriate preset or custom treatment, with the intended target named             |\n| A meaningful media entrance/reveal that feels static     | one supported seek-safe treatment animation; preserve color unless the request also justifies a preset |\n\nRules that keep this a help, not nagware: **grounded, not generic** (no signal → no suggestion); **opinionated + concrete** (propose the specific fix with defaults chosen — the human approves **all / some / none**); **once per project** (one consolidated ask; respect \"leave it\"); **surface, never silently mutate** (color grades especially: propose and preview — a gray-world \"correction\" ruins an intentional sunset or neon look).\n\n## Where to look — read only the file your task needs\n\n| Task                                                                      | Read                             |\n| ------------------------------------------------------------------------- | -------------------------------- |\n| resolve / reuse / adopt / ingest, flags, cascade, inventory               | `references/resolve.md`          |\n| color grading, LUTs, smart grade (`--for`), grade-compare                 | `references/grading.md`          |\n| voiceover / TTS, music, SFX, captions, transcription (audio engine)       | `references/audio.md`            |\n| cut / reframe / transform existing media, exact error diffusion, HEVC     | `references/operations.md`       |\n| source-aware creative treatments, realtime effects, overlays, reveals     | `references/media-treatments.md` |\n| install + auth, provider table, RAM ladders, `--local-only`, `--provider` | `references/setup-providers.md`  |\n| remembered preferences + frozen recipes (user memory)                     | `references/memory.md`           |\n| ownership matrix, usage stats, telemetry, privacy (maintainer-facing)     | `references/meta.md`             |\n\nFile v1.0.73:luts/README.md\n\n# LUT library (authoring)\n\n`index.json` is the agent-consumed catalog of color-grade looks. Each entry resolves\non demand — no `.cube` bodies are committed to the repo.\n\nEach look has:\n\n- `id`, `description`, `tags`, `intensity` — matching + application metadata.\n- `url` (optional) — a hosted `.cube` downloaded, validated, and frozen at resolve\n  time, exactly like bgm/image assets.\n- `params` (optional) — a deterministic `buildCube` spec used offline (`--local-only`)\n  or as a fallback if the `url` download/validation fails.\n\nAn entry needs at least one of `url` or `params`; prefer both (CDN url with a params\nfallback) so resolution is never blocked on the network.\n\n## Hosting a new look (operators)\n\n1. Generate the `.cube` (e.g. `resolve -t lut --params '{...}'` or a graded export).\n2. Upload it to the public CDN origin bucket:\n\n   ```\n   aws s3 cp <id>.cube s3://heygen-public/luts/<id>.cube\n   ```\n\n   It is then served at `https://static.heygen.ai/luts/<id>.cube` (CloudFront).\n\n3. Add an entry to `index.json` with that `url` (and ideally a `params` fallback).\n\nFile v1.0.73:_meta.json\n\n{\n  \"ownerId\": \"kn77d06grj6xqp3dqwkk4bavhn89pegt\",\n  \"slug\": \"media-use\",\n  \"version\": \"1.0.73\",\n  \"publishedAt\": 1791398088007\n}\n\nFile v1.0.73:audio/references/bgm.md\n\n# Background music (BGM)\n\nOne music bed per composition, produced by the shared audio engine (`scripts/audio.mjs` → `scripts/lib/bgm.mjs`). Two routes, chosen by the engine's one switch — whether a HeyGen credential is present:\n\n- **HeyGen retrieval — the default when credentialed.** Search HeyGen's music catalog by mood, download the top track. No generation; same `~/.heygen` / `$HEYGEN_API_KEY` credential as TTS.\n- **Local generation (Lyria → MusicGen) — the fallback when there is no credential** (or when asked for explicitly). Generate a WAV from a mood prompt. There is **no `npx hyperframes bgm` command**; the engine spawns `scripts/lyria-recipe.py` or an inline MusicGen script directly.\n\n> **Run the Preflight first — no credential is not a green light to silently generate locally.** Before generating, complete the sign-in **Preflight** (see `../../SKILL.md` → Preflight): run `npx hyperframes auth status`, recommend signing in, and **STOP for the user's choice** (sign in for HeyGen's music library, or continue offline with local generation). This applies to a one-off \"generate a BGM\" request just as much as inside a full workflow.\n\n## Driving it from the request\n\n`audio_request.json` → `bgm: { mode?, query?, prompt? }`:\n\n- **`mode`** — `retrieve | generate | none`. Omit for **auto** (retrieve when credentialed, else generate). An **explicit** `retrieve` is strict: no credential ⇒ skip, never a detached generate (so a caller with no `wait-bgm` step, e.g. product-launch, can't get a pending job it won't await).\n- **`query`** — the mood, used for retrieval and as a fallback prompt seed (e.g. a storyboard's `music:` field, falling back to `message` → `arc` → `\"calm cinematic underscore\"`).\n- **`prompt`** — an explicit full prompt for generation; omit and the engine infers one (see Mood inference). Optional `blob` / `archetype` / `arc` feed that inference.\n\nBoth routes keep a file of yours already at the output name: the engine writes the next free name (`track-2.mp3`), reports it as an anomaly, and `bgm.path` carries the real path.\n\n## HeyGen retrieval (default)\n\n`searchSounds(query, \"music\", { limit: 5 })` → `GET /audio/sounds?query=<mood>&type=music&limit=5`. Take the top result (ranked by `score`), download its presigned `audio_url` → `assets/bgm/track.mp3`. Synchronous. No match → skip (BGM is optional; never fail the render over it). Cue written to `audio_meta.json`:\n\n```jsonc\n{\n  \"path\": \"assets/bgm/track.mp3\",\n  \"volume\": 0.12,\n  \"mode\": \"retrieve\",\n  \"query\": \"calm cinematic underscore\",\n  \"duration_s\": 42.0,\n}\n```\n\n`volume` comes from the engine's `bgmDefaultVolume()`: `BGM_BED_VOLUME` (currently `0.12` ≈ -18 dB — a bed under the voice) under narration, `BGM_SILENT_VOLUME` (currently `0.9`) for a silent film (no voice). Tune those constants in `scripts/lib/bgm.mjs`, not call sites. An explicit `volume` in `audio_meta.json` always overrides this default. `bgm_pending` is `false` — the file is on disk when the engine returns.\n\nFor short launch videos, do not assume the beginning of the retrieved file is the best edit point. Check the opening against later five-second sections. If the track starts with a quiet build but a later section has a stronger, clean musical entrance, trim from that section and apply a short fade-in and longer fade-out. Repeat this check whenever the composition duration changes; the final music file must cover the full cut without a silent tail.\n\n## Local generation (fallback) — Lyria → MusicGen\n\nSpawned **detached** so voice work isn't blocked; `audio_meta.bgm_pending: true` and `bgm_pid` / `bgm_log` are set until it finishes. **Run `scripts/wait-bgm.mjs` before assembling** — it polls the output file / process / log, detects crashes, and writes `bgm_status.json` (`status: ready | failed | timeout | disabled`). A failed/absent track is simply omitted; it never blocks voice/SFX.\n\n| Order | Provider                             | Env / deps                                                                            | Speed                                   | Quality                     |\n| ----- | ------------------------------------ | ------------------------------------------------------------------------------------- | --------------------------------------- | --------------------------- |\n| 1     | Google Lyria RealTime                | `$GEMINI_API_KEY` or `$GOOGLE_API_KEY` + `google-genai` (auto-installed on demand)    | Real-time stream (≈ requested duration) | Production-grade            |\n| 2     | MusicGen (`facebook/musicgen-small`) | Python `transformers + torch + soundfile + numpy` (~300 MB first run; auto-installed) | Slow on CPU; fast on Apple MPS / CUDA   | Decent; prompt-only control |\n\nOutput → `assets/bgm/track.wav`, target = total voice duration. MusicGen generates **one** seed clip (≤28–30s, under the decoder's positional limit) then crossfade-loops it up to the target (or trims down if shorter), avoiding per-segment seams. Backend selection is by what can actually **run**: Lyria only when `import google.genai` succeeds, else MusicGen; if neither can be made to run, BGM is skipped (voice + SFX still render).\n\n## Mood inference (the generate prompt)\n\n`inferBgmPrompt()` in `scripts/lib/bgm.mjs`: an explicit `prompt` wins; otherwise industry-keyword **base** → narrative-**archetype** shape → emotional-**arc** tiebreaker.\n\n| Match in `blob` / `query`                              | Base prompt                                                                 | BPM |\n| ------------------------------------------------------ | --------------------------------------------------------------------------- | --- |\n| `crypto / nft / web3 / defi / token / blockchain`      | atmospheric electronic, deep bass, futuristic synths, restrained percussion | 100 |\n| `finance / fintech / bank / payment / invest / wealth` | calm cinematic, soft strings, subtle piano, restrained percussion           | 92  |\n| `creative / agency / design / studio / art / brand`    | playful electronic, warm pads, light percussion                             | 115 |\n| _(default: SaaS / tech / platform)_                    | uplifting corporate tech, bright modern piano with synth pads               | 108 |\n\nArchetype then reshapes the arc — PAS → \"MINOR to MAJOR\" build; BAB / future-pacing → aspirational rising; feature-cascade → +10 BPM driving; demo-loop → −8 BPM minimal. The emotional arc breaks remaining ties (tension→relief, excitement, trust/reassurance).\n\n## Lyria knobs (direct recipe use)\n\nThe engine bakes BPM / scale into the **prompt text** (via the inference above) and passes only `--output` / `--duration` / `--prompt` to the recipe. If you invoke `scripts/lyria-recipe.py` directly you can also set: `--bpm` (90–110 calm, 110–130 energetic), `--brightness` (0–1, ≥0.7 promotional), `--density` (0–1, higher = fuller), `--scale` (`MAJOR` / `MINOR` / `PENTATONIC` / …), `--negative-prompt` (styles to exclude). MusicGen ignores all of these — put the mood in the prompt.\n\n## Failure modes\n\n| Failure                                       | Behavior                                                                                 |\n| --------------------------------------------- | ---------------------------------------------------------------------------------------- |\n| No music match (retrieve)                     | `bgm: null`, anomaly logged. Render proceeds without BGM.                                |\n| Explicit `retrieve`, no credential            | Skipped (no silent generate fallback). Use `mode: generate` or omit `mode` for auto.     |\n| Neither Lyria nor MusicGen can run (generate) | `bgm` disabled with a `pip install …` hint. Voice + SFX still render.                    |\n| Generate still rendering at assemble time     | `bgm_pending: true`; `wait-bgm.mjs` waits/checks and writes `bgm_status.json` first.     |\n| Generate crashed                              | `wait-bgm.mjs` → `bgm_status.json { status: \"failed\" }`; the `<audio>` track is omitted. |\n\nBGM failure never blocks a render.\n\nFile v1.0.73:audio/references/captions/authoring.md\n\n# Captions\n\n<!-- registry-items: allow=max-width,data-composition-src,hyperframes-registry,blend-mode,caption-style,font-family -->\n\n**The live search is the source of truth for what the registry has.** The table(s) below are a hand-maintained sample and under-cover by design: run `npx hyperframes catalog --query \"<what you want>\" --json` — it needs nothing installed — before concluding the registry lacks something. Item names here are checked against `regi\n\nArchive v1.0.72: 91 files, 208502 bytes\n\nFiles: audio/assets/sfx/CREDITS.md (1183b), audio/assets/sfx/manifest.json (3721b), audio/references/bgm.md (8141b), audio/references/captions/authoring.md (9545b), audio/references/captions/motion.md (5656b), audio/references/captions/transcript-handling.md (5439b), audio/references/remove-background.md (8472b), audio/references/requirements.md (4347b), audio/references/sfx.md (3678b), audio/references/transcribe.md (2783b), audio/references/tts-to-captions.md (1755b), audio/references/tts.md (14011b), audio/scripts/audio.mjs (13474b), audio/scripts/audio.test.mjs (5930b), audio/scripts/gemini-pipeline.test.mjs (4932b), audio/scripts/heygen-tts.mjs (4388b), audio/scripts/heygen-tts.test.mjs (1408b), audio/scripts/heygen-voice.mjs (3868b), audio/scripts/heygen-voice.test.mjs (8536b), audio/scripts/lib/audio-meta.mjs (1166b), audio/scripts/lib/audio-meta.test.mjs (4586b), audio/scripts/lib/bgm-volume.mjs (236b), audio/scripts/lib/bgm.mjs (11558b), audio/scripts/lib/bgm.test.mjs (2639b), audio/scripts/lib/concurrency.mjs (527b), audio/scripts/lib/concurrency.test.mjs (1654b), audio/scripts/lib/gemini-auth_test.py (4430b), audio/scripts/lib/gemini-auth.mjs (1610b), audio/scripts/lib/gemini-auth.py (1981b), audio/scripts/lib/gemini-auth.test.mjs (3097b), audio/scripts/lib/gemini-tts.mjs (4711b), audio/scripts/lib/gemini-tts.test.mjs (9018b), audio/scripts/lib/heygen.mjs (8874b), audio/scripts/lib/heygen.test.mjs (6414b), audio/scripts/lib/host-audio.mjs (1470b), audio/scripts/lib/host-audio.test.mjs (4036b), audio/scripts/lib/media-record.mjs (4028b), audio/scripts/lib/media-record.test.mjs (7057b), audio/scripts/lib/python.mjs (3290b), audio/scripts/lib/python.test.mjs (5282b), audio/scripts/lib/sfx.mjs (6577b), audio/scripts/lib/sfx.test.mjs (4698b), audio/scripts/lib/tts.mjs (17162b), audio/scripts/lib/tts.spawn.test.mjs (5249b), audio/scripts/lib/tts.test.mjs (8282b), audio/scripts/lyria-recipe.py (5023b), audio/scripts/wait-bgm.mjs (5813b), audio/scripts/wait-bgm.test.mjs (3336b), luts/index.json (1997b), luts/README.md (1087b), references/audio.md (2234b), references/grading.md (6722b), references/media-treatment-recipes.md (35868b), references/media-treatments.md (13920b), references/memory.md (3498b), references/meta.md (6780b), references/operations.md (14823b), references/resolve.md (9605b), references/setup-providers.md (7758b), references/telemetry-dashboard.md (4200b), scripts/audio-duck.mjs (4069b), scripts/compatibility.test.mjs (3804b), scripts/dither.mjs (9969b), scripts/dither.test.mjs (4723b), scripts/eval.mjs (14891b), scripts/lib/config-lock.mjs (1520b), scripts/lib/cutlist.mjs (6095b), scripts/lib/duck.mjs (3266b), scripts/lib/error-diffusion.mjs (6631b), scripts/lib/index-gen.mjs (1924b), scripts/lib/manifest.mjs (9313b), scripts/lib/media-fetch.mjs (2838b), scripts/lib/media-home.mjs (730b), scripts/lib/npx-sync.mjs (2328b), scripts/lib/parakeet-words.mjs (1149b), scripts/lib/prefs-store.mjs (6617b), scripts/lib/recipe-store.mjs (13033b), scripts/lib/telemetry.mjs (7107b), scripts/lib/transcriptCutFade.mjs (911b), scripts/lib/words.mjs (764b)\n\nArchive v1.0.71: 89 files, 206029 bytes\n\nFiles: audio/assets/sfx/CREDITS.md (1183b), audio/assets/sfx/manifest.json (3721b), audio/references/bgm.md (8141b), audio/references/captions/authoring.md (9545b), audio/references/captions/motion.md (5656b), audio/references/captions/transcript-handling.md (5439b), audio/references/remove-background.md (8472b), audio/references/requirements.md (4347b), audio/references/sfx.md (3678b), audio/references/transcribe.md (2783b), audio/references/tts-to-captions.md (1755b), audio/references/tts.md (14011b), audio/scripts/audio.mjs (13474b), audio/scripts/audio.test.mjs (5930b), audio/scripts/gemini-pipeline.test.mjs (4932b), audio/scripts/heygen-tts.mjs (4388b), audio/scripts/heygen-tts.test.mjs (1408b), audio/scripts/heygen-voice.mjs (3868b), audio/scripts/heygen-voice.test.mjs (8536b), audio/scripts/lib/audio-meta.mjs (1166b), audio/scripts/lib/audio-meta.test.mjs (4586b), audio/scripts/lib/bgm-volume.mjs (236b), audio/scripts/lib/bgm.mjs (11558b), audio/scripts/lib/bgm.test.mjs (2639b), audio/scripts/lib/concurrency.mjs (527b), audio/scripts/lib/concurrency.test.mjs (1654b), audio/scripts/lib/gemini-auth_test.py (4430b), audio/scripts/lib/gemini-auth.mjs (1610b), audio/scripts/lib/gemini-auth.py (1981b), audio/scripts/lib/gemini-auth.test.mjs (3097b), audio/scripts/lib/gemini-tts.mjs (4711b), audio/scripts/lib/gemini-tts.test.mjs (9018b), audio/scripts/lib/heygen.mjs (8874b), audio/scripts/lib/heygen.test.mjs (6414b), audio/scripts/lib/media-record.mjs (4028b), audio/scripts/lib/media-record.test.mjs (7057b), audio/scripts/lib/python.mjs (3290b),...","readmeExcerpt":"Skill: media-use Owner: heygen-com Summary: Agent Media OS for a HyperFrames project. Resolve BGM, SFX, image, icon, brand logo, voice, color grade, or LUT into a frozen local file or paste-ready block + ledger record (one verb, resolve); generate via TTS / music / image models when the catalog misses; produce voiceover, transcription, captions, and background removal through one shared audio engine; operate on media","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"npx hyperframes media-use resolve --type <type> --intent \"<description>\" --project <dir>"},{"language":"text","snippet":"aws s3 cp <id>.cube s3://heygen-public/luts/<id>.cube"},{"language":"jsonc","snippet":"{\n  \"path\": \"assets/bgm/track.mp3\",\n  \"volume\": 0.12,\n  \"mode\": \"retrieve\",\n  \"query\": \"calm cinematic underscore\",\n  \"duration_s\": 42.0,\n}"},{"language":"json","snippet":"[\n  { \"id\": \"w0\", \"text\": \"Hello\", \"start\": 0.0, \"end\": 0.5 },\n  { \"id\": \"w1\", \"text\": \"world.\", \"start\": 0.6, \"end\": 1.2 }\n]"},{"language":"js","snippet":"var result = window.__hyperframes.fitTextFontSize(group.text.toUpperCase(), {\n  fontFamily: \"Outfit\",\n  fontWeight: 900,\n  maxWidth: 1600,\n});\nel.style.fontSize = result.fontSize + \"px\";"},{"language":"js","snippet":"tl.to(groupEl, { opacity: 0, scale: 0.95, duration: 0.12, ease: \"power2.in\" }, group.end - 0.12);\n// `tl.set` is an instant flip, not a tween — safe to set `visibility` here (core's \"no animating\n// visibility\" rule applies to tweens, which can't smoothly interpolate non-numeric values anyway).\ntl.set(groupEl, { opacity: 0, visibility: \"hidden\" }, group.end);"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: media-use\ndescription: Agent Media OS for a HyperFrames project. Resolve BGM, SFX, image, icon, brand logo, voice, color grade, or LUT into a frozen local file or paste-ready block + ledger record (one verb, `resolve`); generate via TTS / music / image models when the catalog misses; produce voiceover, transcription, captions, and background removal through one shared audio engine; operate on media (cut / reframe / transform); and reuse assets across projects. Also use for vague feedback that real footage looks dark, flat, boring, should feel retro/camcorder/print/ASCII, needs privacy, or needs a media reveal. When the host app provides its own music or sound-effect tools, use those for music and sound effects; `resolve --type bgm|sfx` needs the heygen CLI. When `HEYGEN_API_BASE` is set, HeyGen calls go through that host with no CLI sign-in.\n---\n\n**Plugin installs:** Before setup or freshness commands, follow [plugin execution rules](../hyperframes/references/plugin-installation.md) when this skill is inside a HyperFrames plugin. Standalone installs keep the update instructions below.\n\n# media-use\n\nThe media OS for HyperFrames: resolve · generate · operate · remember — every media type, one skill, zero context noise.\n\nOnly when `HEYGEN_API_BASE` is set in your environment (a host app set it and pays for HeyGen with its own key): HeyGen media is already paid for. Do not ask the person to install or sign in to the `heygen` CLI and do not offer its OAuth allowance; catalog search, TTS and avatar calls here go through the host. When that same host also gives you its own HeyGen tools, use those first. When a call through the host is refused, tell the person the host's message as written (it names the fix, such as adding or replacing the key in the app's Settings) and stop; do not switch to another provider unless they ask.\n\nFirst run otherwise (no `HEYGEN_API_BASE`), when you will use HeyGen media (catalog search, TTS, avatar video): install and sign in to the `heygen` CLI (the free-usage path), then verify with `npx hyperframes media-use resolve --doctor`. Setup and providers: `references/setup-providers.md`.\n\nMusic and sound effects inside a host app: when the app you run in gives you its own music or sound-effect tools, use those. Without `HEYGEN_API_BASE`, `resolve --type bgm` and `--type sfx` search the HeyGen catalog through the `heygen` CLI; without it they fail and say so (`sfx` still answers from its bundled library).\n\nWithout `HEYGEN_API_BASE`, before generating a voiceover or an avatar video, tell the person: signing in to the heygen CLI with OAuth (`heygen auth login --oauth`) gives a free allowance for TTS voiceover and avatar videos, while an API key bills API credits.\n\n## Resolve — the one verb\n\n```bash\nnpx hyperframes media-use resolve --type <type> --intent \"<description>\" --project <dir>\n```\n\nReturns one line: `resolved <id> → <path> (<type>, <metadata>)`. All search noise stays on disk.\n\n| Type    | One-line intent        "},{"path":"luts/README.md","content":"# LUT library (authoring)\n\n`index.json` is the agent-consumed catalog of color-grade looks. Each entry resolves\non demand — no `.cube` bodies are committed to the repo.\n\nEach look has:\n\n- `id`, `description`, `tags`, `intensity` — matching + application metadata.\n- `url` (optional) — a hosted `.cube` downloaded, validated, and frozen at resolve\n  time, exactly like bgm/image assets.\n- `params` (optional) — a deterministic `buildCube` spec used offline (`--local-only`)\n  or as a fallback if the `url` download/validation fails.\n\nAn entry needs at least one of `url` or `params`; prefer both (CDN url with a params\nfallback) so resolution is never blocked on the network.\n\n## Hosting a new look (operators)\n\n1. Generate the `.cube` (e.g. `resolve -t lut --params '{...}'` or a graded export).\n2. Upload it to the public CDN origin bucket:\n\n   ```\n   aws s3 cp <id>.cube s3://heygen-public/luts/<id>.cube\n   ```\n\n   It is then served at `https://static.heygen.ai/luts/<id>.cube` (CloudFront).\n\n3. Add an entry to `index.json` with that `url` (and ideally a `params` fallback)."},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn77d06grj6xqp3dqwkk4bavhn89pegt\",\n  \"slug\": \"media-use\",\n  \"version\": \"1.0.75\",\n  \"publishedAt\": 1791493414370\n}"},{"path":"audio/references/bgm.md","content":"# Background music (BGM)\n\nOne music bed per composition, produced by the shared audio engine (`scripts/audio.mjs` → `scripts/lib/bgm.mjs`). Two routes, chosen by the engine's one switch — whether a HeyGen credential is present:\n\n- **HeyGen retrieval — the default when credentialed.** Search HeyGen's music catalog by mood, download the top track. No generation; same `~/.heygen` / `$HEYGEN_API_KEY` credential as TTS.\n- **Local generation (Lyria → MusicGen) — the fallback when there is no credential** (or when asked for explicitly). Generate a WAV from a mood prompt. There is **no `npx hyperframes bgm` command**; the engine spawns `scripts/lyria-recipe.py` or an inline MusicGen script directly.\n\n> **Run the Preflight first — no credential is not a green light to silently generate locally.** Before generating, complete the sign-in **Preflight** (see `../../SKILL.md` → Preflight): run `npx hyperframes auth status`, recommend signing in, and **STOP for the user's choice** (sign in for HeyGen's music library, or continue offline with local generation). This applies to a one-off \"generate a BGM\" request just as much as inside a full workflow.\n\n## Driving it from the request\n\n`audio_request.json` → `bgm: { mode?, query?, prompt? }`:\n\n- **`mode`** — `retrieve | generate | none`. Omit for **auto** (retrieve when credentialed, else generate). An **explicit** `retrieve` is strict: no credential ⇒ skip, never a detached generate (so a caller with no `wait-bgm` step, e.g. product-launch, can't get a pending job it won't await).\n- **`query`** — the mood, used for retrieval and as a fallback prompt seed (e.g. a storyboard's `music:` field, falling back to `message` → `arc` → `\"calm cinematic underscore\"`).\n- **`prompt`** — an explicit full prompt for generation; omit and the engine infers one (see Mood inference). Optional `blob` / `archetype` / `arc` feed that inference.\n\nBoth routes keep a file of yours already at the output name: the engine writes the next free name (`track-2.mp3`), reports it as an anomaly, and `bgm.path` carries the real path.\n\n## HeyGen retrieval (default)\n\n`searchSounds(query, \"music\", { limit: 5 })` → `GET /audio/sounds?query=<mood>&type=music&limit=5`. Take the top result (ranked by `score`), download its presigned `audio_url` → `assets/bgm/track.mp3`. Synchronous. No match → skip (BGM is optional; never fail the render over it). Cue written to `audio_meta.json`:\n\n```jsonc\n{\n  \"path\": \"assets/bgm/track.mp3\",\n  \"volume\": 0.12,\n  \"mode\": \"retrieve\",\n  \"query\": \"calm cinematic underscore\",\n  \"duration_s\": 42.0,\n}\n```\n\n`volume` comes from the engine's `bgmDefaultVolume()`: `BGM_BED_VOLUME` (currently `0.12` ≈ -18 dB — a bed under the voice) under narration, `BGM_SILENT_VOLUME` (currently `0.9`) for a silent film (no voice). Tune those constants in `scripts/lib/bgm.mjs`, not call sites. An explicit `volume` in `audio_meta.json` always overrides this default. `bgm_pending` is `false` — the file is on disk when the engine returns.\n\nFor short la"},{"path":"audio/references/captions/authoring.md","content":"# Captions\n\n<!-- registry-items: allow=max-width,data-composition-src,hyperframes-registry,blend-mode,caption-style,font-family -->\n\n**The live search is the source of truth for what the registry has.** The table(s) below are a hand-maintained sample and under-cover by design: run `npx hyperframes catalog --query \"<what you want>\" --json` — it needs nothing installed — before concluding the registry lacks something. Item names here are checked against `registry/registry.json` by `bun run lint:skills`.\n\nBefore authoring: confirm the transcript came from the right Whisper model. CLI default `small.en` silently translates non-English audio — see [`../transcribe.md`](../transcribe.md) → \"Language Rule\" and [`transcript-handling.md`](transcript-handling.md) for the mandatory quality check.\n\nAnalyze spoken content to determine caption style. If user specifies a style, use that. Otherwise, detect tone from the transcript.\n\n## Transcript Source\n\n```json\n[\n  { \"id\": \"w0\", \"text\": \"Hello\", \"start\": 0.0, \"end\": 0.5 },\n  { \"id\": \"w1\", \"text\": \"world.\", \"start\": 0.6, \"end\": 1.2 }\n]\n```\n\n`id` (`w0`, `w1`, …) is the stable reference for per-word overrides and is added by `hyperframes transcribe`. It's optional for backwards compatibility with hand-authored transcripts. See [`../transcribe.md`](../transcribe.md) → \"Output Shape\" for how this is produced, and [`transcript-handling.md`](transcript-handling.md) for cleanup before consumption.\n\n## Style Detection (When No Style Specified)\n\nRead the full transcript before choosing. Four dimensions:\n\n**1. Visual feel** — corporate→clean; energetic→bold; storytelling→elegant; technical→precise; social→playful.\n\n**2. Color palette** — dark+bright for energy; muted for professional; high contrast for clarity; one accent color.\n\n**3. Font mood** — heavy/condensed for impact; clean sans for modern; rounded for friendly; serif for elegance.\n\n**4. Animation character** — scale-pop for punchy; gentle fade for calm; word-by-word for emphasis; typewriter for technical.\n\n## Per-Word Styling\n\nScan for words deserving distinct treatment:\n\n- **Brand/product names** — larger size, unique color\n- **ALL CAPS** — scale boost, flash, accent color\n- **Numbers/statistics** — bold weight, accent color\n- **Emotional keywords** — exaggerated animation (overshoot, bounce)\n- **Call-to-action** — highlight, underline, color pop\n- **Marker highlight** — for beyond-color emphasis (highlight sweep, circle, burst, scribble, sketchout), see `hyperframes-animation/rules/css-marker-patterns.md`.\n\n## Script-to-Style Mapping\n\n| Tone         | Font mood                | Animation                          | Color                       | Size    |\n| ------------ | ------------------------ | ---------------------------------- | --------------------------- | ------- |\n| Hype/launch  | Heavy condensed, 800-900 | Scale-pop, back.out(1.7), 0.1-0.2s | Bright on dark              | 72-96px |\n| Corporate    | Clean sans, 600-700      | Fade+slide, power3.out, 0.3s"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":1919,"uniquenessScore":44,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T05:10:42.588Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T05:10:42.588Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T16:13:17.275Z","emptyReason":null},"items":[{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-10T18:48:31.762Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}