{"id":"62082117-ea1b-4ce7-a74c-fc224300d428","entityType":"agent","slug":"clawhub-heygen-com-embedded-captions","name":"embedded-captions","canonicalUrl":"https://www.xpersona.co/agent/clawhub-heygen-com-embedded-captions","canonicalPath":"/agent/clawhub-heygen-com-embedded-captions","generatedAt":"2026-10-10T06:43:38.102Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-09T16:46:35.121Z","emptyReason":null},"description":"Add captions or subtitles to an existing single-subject talking-head video without editing the footage. Use for plain verbatim captions, cinematic captions embedded behind the subject, VFX captions, “炸/特效/酷炫字幕,” or a named identity from the 35-style catalog. Route by visual identity, not by backend engine. The quiet `anchor` rail is the default; embed every word only when the user explicitly wants a fully cinematic treatment. The workflow runs locally end to end, including transcription and subject matting; split multi-shot footage before applying it. Skill: embedded-captions Owner: heygen-com Summary: Add captions or subtitles to an existing single-subject talking-head video without editing the footage. Use for plain verbatim captions, cinematic captions embedded behind the subject, VFX captions, “炸/特效/酷炫字幕,” or a named identity from the 35-style catalog. Route by visual identity, not by backend engine. The quiet anchor rail is the default; embed every word only","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 2.3K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s17fpgb0p797dzkbtbrxw5x1hh89qs64:embedded-captions","sourceUrl":"https://clawhub.ai/heygen-com/embedded-captions","homepage":"https://clawhub.ai/heygen-com/skills/embedded-captions","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/heygen-com/embedded-captions","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/heygen-com/skills/embedded-captions","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":67,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Add captions or subtitles to an existing single-subject talking-head video without editing the footage. Use for plain verbatim captions, cinematic captions embe"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T16:46:35.121Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T16:46:35.121Z","emptyReason":null},"stars":null,"forks":null,"downloads":2278,"packageName":null,"latestVersion":"1.0.25","tractionLabel":"2.3K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T16:46:35.088Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T16:46:35.121Z","lastCrawledAt":"2026-10-09T16:46:35.088Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T16:46:35.088Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.25","createdAt":"2026-10-04T19:28:55.378Z","changelog":"Synced from 0c3e244 (main)","fileCount":100,"zipByteSize":1397551},{"version":"1.0.24","createdAt":"2026-10-04T19:15:49.412Z","changelog":"Synced from 173103d (main)","fileCount":100,"zipByteSize":1397513},{"version":"1.0.23","createdAt":"2026-10-02T14:30:24.269Z","changelog":"Synced from 9465048 (main)","fileCount":100,"zipByteSize":1397515},{"version":"1.0.22","createdAt":"2026-10-02T13:00:20.160Z","changelog":"Synced from 6f799aa (main)","fileCount":100,"zipByteSize":1397518},{"version":"1.0.21","createdAt":"2026-10-02T00:10:59.261Z","changelog":"Synced from 37f30b1 (main)","fileCount":100,"zipByteSize":1397548},{"version":"1.0.20","createdAt":"2026-09-27T23:31:55.062Z","changelog":"Synced from 471e90c (main)","fileCount":100,"zipByteSize":1397407},{"version":"1.0.19","createdAt":"2026-09-27T22:09:58.521Z","changelog":"Synced from 0a5e3c7 (main)","fileCount":99,"zipByteSize":1396377},{"version":"1.0.18","createdAt":"2026-09-27T21:17:31.413Z","changelog":"Synced from ff6e210 (main)","fileCount":98,"zipByteSize":1395971}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17fpgb0p797dzkbtbrxw5x1hh89qs64:embedded-captions","setupComplexity":"low","setupSteps":["Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-heygen-com-embedded-captions/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-heygen-com-embedded-captions/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-heygen-com-embedded-captions/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-heygen-com-embedded-captions/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-heygen-com-embedded-captions/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-heygen-com-embedded-captions/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T06:43:38.100Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-heygen-com-embedded-captions/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-heygen-com-embedded-captions/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-heygen-com-embedded-captions/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-heygen-com-embedded-captions/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-09T16:46:35.121Z","emptyReason":null},"readme":"Skill: embedded-captions\n\nOwner: heygen-com\n\nSummary: Add captions or subtitles to an existing single-subject talking-head video without editing the footage. Use for plain verbatim captions, cinematic captions embedded behind the subject, VFX captions, “炸/特效/酷炫字幕,” or a named identity from the 35-style catalog. Route by visual identity, not by backend engine. The quiet `anchor` rail is the default; embed every word only when the user explicitly wants a fully cinematic treatment. The workflow runs locally end to end, including transcription and subject matting; split multi-shot footage before applying it.\n\nTags: latest:1.0.25\n\nVersion history:\n\nv1.0.25 | 2026-10-04T19:28:55.378Z | user\n\nSynced from 0c3e244 (main)\n\nv1.0.24 | 2026-10-04T19:15:49.412Z | user\n\nSynced from 173103d (main)\n\nv1.0.23 | 2026-10-02T14:30:24.269Z | user\n\nSynced from 9465048 (main)\n\nv1.0.22 | 2026-10-02T13:00:20.160Z | user\n\nSynced from 6f799aa (main)\n\nv1.0.21 | 2026-10-02T00:10:59.261Z | user\n\nSynced from 37f30b1 (main)\n\nv1.0.20 | 2026-09-27T23:31:55.062Z | user\n\nSynced from 471e90c (main)\n\nv1.0.19 | 2026-09-27T22:09:58.521Z | user\n\nSynced from 0a5e3c7 (main)\n\nv1.0.18 | 2026-09-27T21:17:31.413Z | user\n\nSynced from ff6e210 (main)\n\nv1.0.17 | 2026-09-19T03:20:31.936Z | user\n\nSynced from 2db126d (main)\n\nv1.0.16 | 2026-09-14T01:22:00.643Z | user\n\nSynced from 95bea16 (main)\n\nv1.0.15 | 2026-09-10T03:22:39.845Z | user\n\nSynced from 0f8eb89 (main)\n\nv1.0.14 | 2026-09-04T21:39:45.461Z | user\n\nSynced from 924b98a (main)\n\nv1.0.13 | 2026-09-04T18:28:46.244Z | user\n\nSynced from 62d3304 (main)\n\nv1.0.12 | 2026-09-04T18:05:27.484Z | user\n\nSynced from 0154b40 (main)\n\nv1.0.11 | 2026-08-21T03:09:05.685Z | user\n\nSynced from efc2e19 (main)\n\nv1.0.10 | 2026-08-18T01:45:34.146Z | user\n\nSynced from f8a1e2d (main)\n\nv1.0.9 | 2026-08-11T18:58:49.990Z | user\n\nSynced from d3a4e90 (main)\n\nv1.0.8 | 2026-08-04T17:44:19.488Z | user\n\nSynced from f9ec934 (main)\n\nv1.0.7 | 2026-07-28T11:28:25.024Z | user\n\nSynced from d287e52 (main)\n\nv1.0.6 | 2026-07-21T16:42:48.437Z | user\n\nSynced from 696cbdb (main)\n\nv1.0.5 | 2026-07-15T13:20:32.129Z | user\n\nSynced from b9be0b2 (main)\n\nv1.0.4 | 2026-07-08T17:59:56.607Z | user\n\nSynced from 17b8527 (main)\n\nv1.0.3 | 2026-07-08T17:31:58.173Z | user\n\nSynced from 81884a7 (main)\n\nv1.0.2 | 2026-07-07T20:26:36.308Z | user\n\nSynced from 7286b00 (main)\n\nv1.0.1 | 2026-07-07T19:00:04.886Z | user\n\nSynced from 5fe9573 (main)\n\nv1.0.0 | 2026-07-01T08:42:00.837Z | user\n\nOfficial HyperFrames skills from heygen-com/hyperframes\n\nArchive index:\n\nArchive v1.0.25: 100 files, 1397551 bytes\n\nFiles: assets/fonts/char-widths.json (39965b), assets/strokefonts/HersheyScript1.svg (60060b), assets/strokefonts/HersheyScriptMed.svg (71046b), CATALOG.md (54574b), dna/chrome.json (1769b), dna/cream.json (1537b), dna/documentary.json (1441b), dna/editorial.json (1674b), dna/glitch.json (1805b), dna/ink.json (1529b), dna/keynote.json (1400b), dna/loud.json (1912b), dna/neon.json (1567b), dna/README.md (12207b), dna/velocity.json (1968b), modes/cinematic/_archive/champion/spec.md (4718b), modes/cinematic/_archive/champion/template.html (5784b), modes/cinematic/_archive/memory-wall/spec.md (5878b), modes/cinematic/_archive/memory-wall/template.html (5624b), modes/cinematic/_archive/portrait-header/spec.md (3621b), modes/cinematic/_archive/portrait-header/template.html (5312b), modes/cinematic/cinematic-cream/spec.md (1139b), modes/cinematic/cinematic-cream/template.html (10266b), modes/cinematic/engine.html (17093b), modes/cinematic/README.md (2677b), modes/standard/_anatomy.md (11200b), modes/standard/_motion.md (18518b), modes/standard/fonts/build-fonts-css.cjs (3597b), modes/standard/fonts/fonts.css (1203094b), references/aesthetic-principles.md (7779b), references/anti-patterns.md (12023b), references/bespoke-vs-presets.md (9286b), references/caption-grouping.md (4283b), references/composition-craft.md (55818b), references/direction-catalog.md (7855b), references/example-renders/champion.html (11122b), references/example-renders/memory-wall.html (10707b), references/failure-modes.md (11303b), references/layout-heuristics.md (13202b), references/motion-vocabulary.md (5609b), references/rail.md (4576b), references/reference-bar.md (2689b), references/scene-types.md (6215b), references/typographic-moves.md (8655b), references/typography-presets.md (4714b), scripts/audio-envelope.cjs (3093b), scripts/check-occlusion.cjs (10490b), scripts/check-overflow.cjs (6817b), scripts/check-rail-climax.cjs (8577b), scripts/check-timing.cjs (6354b), scripts/fill-timings.cjs (4587b), scripts/fit-fonts.cjs (6326b), scripts/fixtures/heroless/theme.json (159b), scripts/gen-stroke-path.py (2503b), scripts/hf-cli.cjs (1258b), scripts/inject-fonts.cjs (5901b), scripts/lib-dna.cjs (7410b), scripts/make-cinematic.cjs (56867b), scripts/make-composition.cjs (16750b), scripts/make-theme.cjs (460163b), scripts/make-theme.test.mjs (4623b), scripts/matte.cjs (9125b), scripts/measure-layout.cjs (10839b), scripts/prepare.sh (1881b), scripts/preview-frames.cjs (11515b), scripts/preview-frames.test.mjs (1160b), scripts/render-and-composite.sh (23051b), scripts/render-theme.sh (1786b), scripts/safe-zones.cjs (30395b), scripts/transcribe.cjs (12471b), scripts/transcribe.test.mjs (2847b), skill-card.md (2116b), SKILL.md (34642b), themes/anchor.json (1090b), themes/arcade.json (2031b), themes/aurora.json (1967b), themes/biolume.json (1924b), themes/brush.json (2668b), themes/chalkboard.json (2040b), themes/dossier.json (1905b)\n\nFile v1.0.25:SKILL.md\n\n---\nname: embedded-captions\ndescription: >\n  Add captions or subtitles to an existing single-subject talking-head video without editing the\n  footage. Use for plain verbatim captions, cinematic captions embedded behind the subject, VFX\n  captions, “炸/特效/酷炫字幕,” or a named identity from the 35-style catalog. Route by visual\n  identity, not by backend engine. The quiet `anchor` rail is the default; embed every word only\n  when the user explicitly wants a fully cinematic treatment. The workflow runs locally end to end,\n  including transcription and subject matting; split multi-shot footage before applying it.\n---\n\n**Plugin installs:** Before setup or freshness commands, follow [plugin execution rules](../hyperframes/references/plugin-installation.md) when this skill is inside a HyperFrames plugin. Standalone installs keep the update instructions below.\n\n> **First, keep this skill fresh — confirm with the user before running:** `npx hyperframes skills update embedded-captions`. A fast no-op when everything is current; otherwise it refreshes this skill plus the core domain skills it depends on before you rely on them.\n\n# Embedded Captions\n\n**One catalog, picked up front** ([CATALOG.md](CATALOG.md) — 35 identities; the engines behind it are backend detail). **Standard** (default) builds a clean verbatim **rail** (lower-third subtitle carrying most text) + an **embed** climax composited _into_ the scene behind the subject at the peak. **Cinematic** is pure embed — no rail, every caption composited behind the subject (hero typography, accumulation, occlusion as the effect). **Theme** is a complete themed constitution — body paradigm × hero setpiece × front fx × plate reaction, composed from registries ([themes/README.md](themes/README.md)): `ordnance` `terminal` `neonsign` `stardust` `stomp`. Most explainer / voiceover is **Standard**; **embed is the scarce, earned peak** — embedding every word is the common mistake; Theme is for VFX-grade asks (\"炸\", \"特效\", \"像 AE 做的\").\n\n---\n\n## Runtime prerequisites\n\nPlugin installs use the bundled, manifest-pinned CLI for matting, transcription,\nand rendering; no source checkout is required. The local preview and caption\nmeasurement helpers also need Sharp, Puppeteer (with its Chromium browser), and\nGSAP. Install these in the **caption project**, not inside the read-only plugin:\n\n```bash\nnpm install --prefix <project> --save-dev --save-exact sharp@0.35.3 puppeteer@25.8.0 gsap@3.15.0\n```\n\nKeep the project's lockfile. If these dependencies already exist, use its locked\nversions instead of overwriting them. Bash and FFmpeg/ffprobe must be on PATH.\nMatting and transcription may download their own models on first use.\n\nRendering waits for the CLI to exit successfully before compositing. The old\n`HF_TIMEOUT_S` shell watchdog is no longer used: a large partial file is not proof\nthat rendering finished. An explicit built-checkout argument or `HYPERFRAMES_ROOT`\nselects the contributor CLI instead of the plugin pin. Cancel a stalled render normally through the CLI/terminal;\nthe caption helper does not force-kill or recover a render from a process snapshot.\n\n## Operational flow (TL;DR)\n\nRouted through `/hyperframes`, the intent layer confirms only the input (which clip) and **announces** the identity pick as a deferred ask — the shortlist needs the probed clip, so it stays at step 1 below; the layer's run-shape questions don't apply (the footage is untouched, there is no storyboard to review). A `BRIEF.md`, when present, carries the confirmed input and any user notes — read it first.\n\nThe craft prose below is long; the **pipeline itself is short** — and everything deterministic is computed or compiled, never hand-written:\n\n1. **Decision gate** (refuse bad clips) → **pick ONE identity from [CATALOG.md](CATALOG.md)** (35 identities; engine/compiler derived by lookup — never surface a mode/category question)\n2. `hyperframes init` (skip it if the project dir already exists with the video inside — `matte.cjs`/`transcribe.cjs` adopt any video in the dir as source.mp4) → **`bash scripts/prepare.sh <project>`** (matte ∥ transcribe ∥ audio-envelope in parallel, then safe-zones v2 with scene palette/optics/lighting — one command, nothing forgotten)\n3. **author a small JSON of creative choices** (read `safe-zones.json` first): Cinematic → `cinematic.json` → `make-cinematic.cjs` (derives `plan.json` and compiles it); Theme → `theme.json` → `make-theme.cjs` (rail/panel/poem/takeover paradigms; `anchor` is the quiet rail default)\n4. **Visual QA**: `node scripts/preview-frames.cjs <project>` → faithful composite previews in ~2s/frame (no render). Check § Visual QA before paying for a render.\n5. `render-and-composite.sh` → gates (timing / occlusion+hero / overflow / hand-off) → `final.mp4`\n\nLoad-bearing rules people miss:\n\n- **rail (default) + embed (promotion).** `drop` (filler, not shown) / `rail` (verbatim lower-third subtitle, in front, carries most text) / `embed` (a peak word composited behind the subject). **Standard mode does both**, embedding only the peak(s). See **§ Caption model**.\n- **The video is delivered UNTOUCHED (Standard/Cinematic; **Theme mode's PLATE budget is the one sanctioned exception** — register-gated reaction beats (charge-dim, punch, shake, grain) defined per theme DNA and applied AFTER the matte composite so subject+text+plate move as one frame)** — captions are the only thing added; the matte just lets the subject occlude the embed track. Never grade/recolor/scanline the footage.\n- Two rulebooks: **rail → [references/rail.md](references/rail.md)** (thin), **embed craft → [references/composition-craft.md](references/composition-craft.md)** (rich, embed-only). Skim by need.\n\n---\n\n## Caption model — rail + embed\n\nEvery spoken phrase is one of three things:\n\n|           | What                                             | How it's shown                                                                                                                                                    |\n| --------- | ------------------------------------------------ | ----------------------------------------------------------------------------------------------------------------------------------------------------------------- |\n| **drop**  | filler — um/uh, stutters, self-corrections       | not shown                                                                                                                                                         |\n| **rail**  | the default — ordinary spoken content (verbatim) | clean lower-third subtitle, **in front**, readable. A punch word can get an inline `emphasis` highlight (accent colour / active-word pop) — it stays on the rail. |\n| **embed** | a promoted peak — the headline beat              | one big word composited **behind the subject** (matte occlusion), designed entrance + exit                                                                        |\n\n**The rail carries most of the text; embed is the scarce, earned peak.** Scarcity is **per beat/block, not per clip**: ≤1 hero per block (thought), never two co-visible, ≥ a beat of air between hero windows (the compiler warns under 0.6s). A short clip → usually 1–2; a long explainer → ~one per section. Among multiple heroes, the **largest authored one is the APEX** (it alone gets the full lockup embed + width-fit raise); smaller ones are **MINOR peaks** that ride their column as oversized emphasis lines (fg, damped motion) — not every beat needs the matte showcase, which is exactly what keeps the apex an event. Embedding every word is still the common mistake.\n\nRail-surface identities build exactly this (rail = `rail.html`, embed = the climax in `index.html`). Column-flow identities drop the rail and make everything embed-style — recommend them only for mood-over-verbatim asks, never for explainer / voiceover where the words must read (CATALOG.md encodes this per identity).\n\n---\n\n## Step 0 — pick ONE identity from the CATALOG\n\n**One front-end, three engines behind.** The user picks an IDENTITY from [CATALOG.md](CATALOG.md) (35 entries: 10 classic + 25 themed); the engine, compiler and authoring file are derived by lookup from the catalog row. **Never surface \"Standard vs Cinematic vs Theme\" as a question** — those are backend names (a product has one UX even with several engines). The catalog encodes everything routing needs: reading surface, voice, recommend-for, scene needs, adjacency notes for the genuinely-close pairs (loud↔ordnance, neon↔neonsign, cream↔stardust).\n\nThe identity pick is a **preference gate** (`../hyperframes/references/brief-contract.md` § 1): in autonomous mode (\"surprise me\" / \"decide for me\"), pick from your shortlist yourself and state the one-line why instead of asking.\n\nProcedure: probe the clip → shortlist 2–3 identities from the catalog → recommend ONE with a one-line why → **the user picks** (autonomous mode: you pick, stating the why) → author that identity's file. Identities are engine-locked (no cross combos; opening one is a validation event — see dna/README.md).\n\n**Always present your recommendation and let the user pick before you author.** Don't silently default.\n\n(The full identity table lives in [CATALOG.md](CATALOG.md) — single source of truth for routing. The engine docs below describe each backend's authoring contract.)\n\n**CATALOG.md is the whole answer space here: this workflow does not search the HyperFrames component registry.** The composition workflows run `npx hyperframes catalog` before authoring a named look; this one must not. Its engines are locked compilers that consume `cinematic.json` / `theme.json` and emit the composition themselves, so a registry item — the `caption-*` blocks included — has nothing to mount into. A registry block styles text on a designed canvas; this skill burns captions into somebody's footage through a matte. When no identity fits the ask, say so and pick the nearest, rather than reaching outside the catalog.\n\n**Recommendation heuristic**: use the \"Shortlisting heuristics\" in [CATALOG.md](CATALOG.md) — they are identity-level (e.g. \"炸\" shortlists ordnance/stomp/terminal/loud and picks by WHAT should explode), never category-level. Unsure → `anchor`.\n\n- **Cinematic** → write `cinematic.json` for a locked template, compiled by `make-cinematic.cjs`.\n- **Theme** → read [themes/README.md](themes/README.md), author `theme.json`, run `scripts/render-theme.sh` (compiles + renders + plate reaction → **final_fx.mp4**).\n\n---\n\n## Decision gate — RUN FIRST\n\nProbe the video and classify the scene before either mode.\n\n```bash\nffprobe <video.mp4>                    # specs\nffmpeg -ss <t> -i <video.mp4> -vframes 1 sample.png   # at 20/50/80%\n```\n\nRead the samples. Refuse if:\n\n- Multiple speakers / hard cuts (split & render each shot, or refuse)\n- No human subject (this skill is for talking-head)\n- Under 3 seconds, **no speech**, or face never clearly visible — `transcribe.cjs` warns when audio is near-silent (Whisper hallucinates words like \"Thank you.\" over silence); **heed it and refuse** rather than caption fabricated words\n- **Source already has burned-in captions / subtitles / heavy text graphics** — adding a second caption system conflicts and the footage ships untouched (no covering/inpainting). Burned text often appears only mid-clip: sample a **1fps contact sheet** (`ffmpeg -i in.mp4 -vf \"fps=1,scale=160:-1,tile=10x5\" sheet.png`), don't trust 3 spot frames.\n- **Transcript is garbage** — non-native/heavy-accent speech can transcribe into confident gibberish. Sanity-read `transcript.json` before authoring; if it doesn't parse as language, try `WHISPER_MODEL=medium` once, else refuse (a verbatim rail of fabricated words is worse than no captions).\n- Busy handheld with fast motion (matte flickers)\n\n### Pre-flight probes (cost nothing, prevent the worst failures)\n\n1. **Shot-cut probe.** Sample frames at 20%, 50%, 80%. If a different subject/scene appears, **trim the clip** before the cut.\n2. **Letterbox / pillarbox probe.** Black bars on the first frame? Compute safe content rect and constrain caption placement inside it.\n3. **Luminance probe.** Sample the caption region's average luminance — `under 60` → light text reads as-is, `60-180` → add the glyph scrim, `180+` → opaque text + scrim (never bare light text). **Cinematic templates are cream+`screen` and LOCKED** — use this probe to _pick a fitting identity_ (bright scenes → `ink`, or the opaque-rail `anchor` theme), never to recolour one.\n4. **Identity recommendation by tone (you recommend; the user picks — see Step 0 + CATALOG.md).** explainer / interview / must-read words → rail/panel-surface identities; poetic / social / \"cinematic\" → column-flow identities by register; \"炸 / 特效 / VFX\" / named worlds → themed identities. When unsure → `anchor` (words read, scene safe) — but present a shortlist and let the user choose.\n\n---\n\n## Pipeline — 5 steps\n\n```\n1. hyperframes init <project> --non-interactive --video <video.mp4> --skill=embedded-captions\n2. bash scripts/prepare.sh <project>       # matte ∥ transcribe (parallel) → safe-zones. One command.\n                                           #   → frames_fg/ transcript.json safe-zones.json\n3. [AGENT STEP — the only creative step] author a small JSON; see below by mode\n   Cinematic: author cinematic.json → node scripts/make-cinematic.cjs <project>\n   Theme:     author theme.json → bash scripts/render-theme.sh <project>   (compiles + renders + plate fx)\n4. node scripts/preview-frames.cjs <project>   # ~2s/frame composite previews → § Visual QA (BEFORE the render)\n5. bash scripts/render-and-composite.sh <project>  # gates → final.mp4 + history/ snapshot\n   (Theme mode: SKIP steps 3b/5 — render-theme.sh already runs compile + render-and-composite\n    + _postfx.sh; the deliverable is final_fx.mp4, final.mp4 is pre-plate-reaction)\n```\n\nStep 1's `init` checks the installed skills against the latest on GitHub and updates the global set if any are out of date.\n\nStep 3 differs by mode:\n\n### Step 3 — Cinematic mode (pure embed)\n\n1. **Read `safe-zones.json` first.** Narration planes go in **`zones.hugLeft`/`hugRight`** — clean strips ABUTTING the silhouette (text far from the body reads as floating, not embedded; far corners are the fallback, not the default). The hero defaults to `heroAnchor`/`heroBands.best` (centered ON the subject, ~30–55% occluded). `recommendation:\"fg\"` moves NARRATION in front for legibility; **the hero stays embedded whenever `heroBands.feasible`** — hero-fg is the last resort.\n2. **The DNA is the identity you picked in Step 0** (CATALOG.md) — do not re-open the choice here. Sanity-check it against the scene (bright hero band luma > 150 wants `ink`; full pick guidance lives in the catalog, covering all ten incl. neon / glitch / chrome / velocity). State your pick + why; the user decides. The DNA locks type/palette/blend/motion + hero three-act; safe-zones v2 (`palette`/`optics`/`lighting`) parameterizes it to THIS scene automatically.\n3. **Author `<project>/cinematic.json`** — `\"dna\": \"<name>\"` + thought-BLOCKS, not raw groups: each block = lines of words (grouped 2–5 at clause boundaries) + the plane it stacks in + per-line `css` (size/weight/style only — no positions) + at most ONE line marked `\"hero\": true` (the promoted word; `\"text\"` for display form). Schema: `scripts/make-cinematic.cjs` header.\n4. **Compile**: `node scripts/make-cinematic.cjs <project>` — lowers blocks → plan.json → index.html. Generated for you: transcript-sequenced timings, accumulate-within-block, page-flip-between-blocks, **the hero LOCKUP** (a hero block's pre-context, HERO and post-context stack as ONE bonded composition centered on the subject — reading order top→bottom = spoken order by construction; context floats in FRONT while the hero embeds BEHIND = the depth sandwich; a mass rule keeps the hero dominating its context), apex/minor hero split, **reading order by construction**, fg fallback per safe-zones. Then the gates run as usual. _(Hand-authoring plan.json directly remains possible for designs blocks can't express — then run `fill-timings.cjs` + `fit-fonts.cjs` + `make-composition.cjs` yourself.)_\n\n### Step 3 — Theme mode (themed constitution)\n\n**Read [themes/README.md](themes/README.md) FIRST** — paradigm/setpiece registries, linkages, hard rules, and the exact `theme.json` schema.\n\n1. **Pick a theme DNA** by content register (each `themes/<name>.json` has `voice` + `when`). State your pick + why; the user decides.\n2. **Author `<project>/theme.json`** — `dna`, `lines` (verbatim, transcript order; 1–5 words each — for `takeover` each line is one CARD), `minors` (emphasis words), `hero:{match}` (the climax word/phrase; leave it OUT of `lines` for embed setpieces, keep it IN for inline setpieces and panel+redact).\n3. **Render**: `bash scripts/render-theme.sh <project>` — compiles (verbatim-completeness gate at compile time), renders both layers, composites, applies the plate reaction → `final_fx.mp4`. Use `preview-frames.cjs` between compile and render for Visual QA.\n\n---\n\n## Visual QA — preview BEFORE you render\n\n`node scripts/preview-frames.cjs <project> [t…]` composites **faithful preview frames in ~2s each** (caption layers screenshotted at seek-time + real video frame + matte occlusion + rail overlay = what the final composite will look like at that moment). Default samples = each group/climax window. A full render costs minutes — never use it to _discover_ layout problems.\n\nCheck the previews (`<project>/preview/sheet.png`) against this list — these are the failures the geometric gates **cannot** catch:\n\n1. **Washout** — light text over a bright region (window/sign/sky): unreadable → move the plane or change DNA/mode (bright scene → `ink`).\n2. **Text-on-text** — captions over the scene's own text/graphics, or two caption groups colliding.\n3. **Reading order** — on-screen vertical order must match spoken order; the hero must not sit below later words.\n4. **Hero presence** — the climax should be BIG and visibly behind the subject (~30–55% occluded), not a floating label in a margin.\n5. **Balance** — one coherent column/band, not scattered fragments; margins breathing; nothing clipped.\n\nThen the **5 positive checks** in [references/reference-bar.md](references/reference-bar.md) (poster test · timid test · one-glance hierarchy · scene handshake · dead-air audit) — the failure list keeps a render from being broken; the positive list is what makes it _designed_. Ship when both pass.\n\n**Fresh-eyes review (recommended for anything user-facing):** you have confirmation bias about your own layout. If you can spawn a subagent, give it ONLY the preview sheet + this checklist and ask for PASS/FIX verdicts per frame (\"review these caption previews against the 5-point checklist; answer PASS or the specific fix per frame\"). Apply fixes in cinematic.json / theme.json, recompile, re-preview — each loop costs seconds. Render once, when the previews pass.\n\n---\n\n## The DNA registry — ten visual languages (replaces the template catalog)\n\nBoth modes draw from **[dna/](dna/README.md)** — ten art-directed visual languages that **parameterize per scene** (accent sampled from the footage, contact shadow along the measured light direction, depth-match blur, RMS-coupled hero amplitude):\n\n| DNA             | Register       | Scene fit                                       | Voice                                                                                              |\n| --------------- | -------------- | ----------------------------------------------- | -------------------------------------------------------------------------------------------------- |\n| **cream**       | premium-warm   | dark/mid warm scenes                            | Inter + warm cream + screen; glowing emergence hero (successor of cinematic-cream)                 |\n| **ink**         | premium        | **bright scenes (luma > 150)**                  | near-black multiply — type printed ON the wall; the bright-scene answer                            |\n| **editorial**   | editorial-luxe | introspective / fashion / poetic                | Bodoni Moda, lowercase-italic hero — magazine elegance                                             |\n| **keynote**     | tech-premium   | product / launch                                | opaque white Inter 800, dead-center stillness                                                      |\n| **documentary** | formal         | interview / serious                             | burn-in reveals, no hero — gravitas IS the style                                                   |\n| **loud**        | loud           | hype / sport / social                           | Anton + scene-sampled accent, single-unit slam + ripple; body ANNOUNCES in front (`bodyLayer: fg`) |\n| **neon**        | loud-neon      | neon-noir / nightlife / tech-noir (dark scenes) | electric-cyan signage, ignition flicker, the hero powers ON like a sign                            |\n| **glitch**      | loud-neon      | digital / hacker / AI                           | RGB-split echoes snap together on landing; machine-percussive timing                               |\n| **chrome**      | loud-luxe      | Y2K / fashion-tech / music                      | liquid-metal gradient hero + one sheen sweep during the hold                                       |\n| **velocity**    | loud-sport     | sport / auto / fitness                          | every word arrives along its motion vector (streak+skew), hero passes with speed trails            |\n\nPick by `safe-zones.json` (`heroAnchor.bandLuma`, `palette.temperature`) × content register — [dna/README.md](dna/README.md) has the decision rule. Authoring: `cinematic.json` takes `\"dna\": \"<name>\"`.\n\nThe engine generates the **hero three-act** from the DNA (no authoring needed): co-visible captions dim (setup) → per-letter entrance with amplitude ∝ spoken loudness (impact) → breathe + glow until exit (afterglow).\n\n(Legacy: `plan.template:\"cinematic-cream\"` maps to `dna:\"cream\"` automatically. The retired 54-template library is archived outside this repo and is not distributed with the skill; `_motion.md` remains in-skill as the motion-verb reference catalog.)\n\n---\n\n## Aesthetic decision — tone × shot × platform (input to the catalog shortlist, NOT a second router)\n\nClassify the clip on 3 axes and feed the result into CATALOG.md's shortlisting — this section never picks a mode/engine by itself:\n\n**Tone** (what feel does the content have?)\n\n- documentary | conversational | energetic | poetic | keynote | investigative | music-video\n\n**Shot** (what's the framing?)\n\n- close-up (head + shoulders) | mid-shot (torso+) | wide (full body+) | cut-montage (mixed shots)\n\n**Platform** (where will it play?)\n\n- 9:16 portrait (TikTok/IG/Shorts) | 16:9 landscape (YouTube/web) | 1:1 square | broadcast export\n\nCross-reference in [references/direction-catalog.md § Classification matrix](references/direction-catalog.md) for direction language — then return to [CATALOG.md](CATALOG.md) to shortlist identities (this matrix informs the shortlist; the catalog is the only routing surface).\n\n## Composition craft (embed track) — read before embedding\n\nThe full **embed-track** playbook lives in **[references/composition-craft.md](references/composition-craft.md)**: transcript role-annotation, phrase grouping, planes & clean-zone anchoring, zone coherence, climax pop & readability, edge-breathing, the occlusion 3-step judgement, and accumulation/persistence. It governs how a _promoted_ phrase sits INTO the scene — read it before authoring any embed (Cinematic `cinematic.json` or Standard `index.html`). The default **rail** track has its own, much simpler spec → **[references/rail.md](references/rail.md)**.\n\n---\n\n## Shared knowledge\n\n| Doc                                                                      | What                                                                                                                               |\n| ------------------------------------------------------------------------ | ---------------------------------------------------------------------------------------------------------------------------------- |\n| [references/rail.md](references/rail.md)                                 | **The rail track** — standard lower-third subtitle spec (the default; carries most text).                                          |\n| [references/composition-craft.md](references/composition-craft.md)       | **The embed-track playbook** — grouping, planes, climax pop, occlusion judgement, accumulation/persistence. Read before embedding. |\n| [dna/README.md](dna/README.md)                                           | **The DNA registry** — ten scene-parameterized visual languages; how to pick.                                                      |\n| [references/reference-bar.md](references/reference-bar.md)               | **The taste bar** — per-register world-class references + the 5 positive checks.                                                   |\n| [references/aesthetic-principles.md](references/aesthetic-principles.md) | **The 18 rules.** Beat Veed AI on taste. Read first.                                                                               |\n| [references/motion-vocabulary.md](references/motion-vocabulary.md)       | 10 named motion primitives + tone→timing lookup                                                                                    |\n| [references/direction-catalog.md](references/direction-catalog.md)       | 10 ship-ready aesthetics + tone×shot×platform matrix                                                                               |\n| [references/anti-patterns.md](references/anti-patterns.md)               | Bugs already locked out (CoreML, letter-spacing reflow, etc.)                                                                      |\n| [references/scene-types.md](references/scene-types.md)                   | When a wall surface is usable (4 conditions)                                                                                       |\n| [references/layout-heuristics.md](references/layout-heuristics.md)       | Plane positioning, clean-zone selection, crown 3 conditions, pillarbox math                                                        |\n| [references/typography-presets.md](references/typography-presets.md)     | Font-size × column-width matrix (starting points)                                                                                  |\n| [references/caption-grouping.md](references/caption-grouping.md)         | Word → group rules (pauses, sentence boundaries)                                                                                   |\n| [references/failure-modes.md](references/failure-modes.md)               | Long tail of dev gotchas                                                                                                           |\n| [references/bespoke-vs-presets.md](references/bespoke-vs-presets.md)     | Why presets fail sometimes; clone-and-tweak pattern                                                                                |\n\n**Read the aesthetic principles and direction catalog FIRST.** Everything else is implementation detail.\n\n---\n\n## Non-negotiables\n\n- **Face must never be 100%-covered continuously** — every 0.3s window, face bbox ≥30% uncovered.\n- **WCAG contrast** — final render lints; fix palette if it fails.\n- **Deterministic** — no `Math.random()`, no `Date.now()`, no `repeat:-1`.\n- **Never grade/recolor the video.** The footage ships untouched — captions are the only addition. No full-frame scanlines / duotone / darken / vignette over the a-roll. neon-noir/CRT texture belongs _inside_ a caption element, not over the whole frame.\n- **Rail-first for talking-head / explainer.** Don't embed the whole transcript — most text is the rail; embed only peaks. Embedding everything is the default mistake.\n- **Embed is scarce + spaced.** ≤1 embed per sentence/beat, never two adjacent or co-visible, ≥ a beat apart, at most one `apex`. climax = per-beat peak, **not** \"the single payoff of the entire clip.\"\n- **Matte = the PERSON (hyperframes `remove-background`, u2net_human_seg, Apache-2.0).** Human segmentation by intent, but not surgically: thin offset furniture (mic boom arms) is usually excluded — captions render over it, behind the person — while large salient objects NEAR the subject (a telescope, a desk rig) can still leak into the matte and occlude captions. Objects HELD by the subject (products, phones) may drop out intermittently, letting captions pass in front. NEVER assume: sample `frames_fg/` at 2-3 timestamps before placing the hero, and prefer hero positions clear of any leaked furniture (`heroAnchor` can be skewed by leaks — cross-check against frames_bg).\n- **safe-zones is PROP-BLIND — eyeball every band you use.** Zones/heroBands score _subject_ occlusion + luma only: a mic, telescope, or screen sitting inside a \"clean\" zone is invisible to them (and a prop leaking INTO the matte skews `heroAnchor.centerXPct` off the person). Before authoring, extract ONE frame of each band you intend to use; if a prop lives there, measure its bbox and move/shrink the plane. Two real cases shipped clean only because the agent did exactly this. (Auto prop-saliency is a known gap; zones' `peakLuma` only catches _moving_ bright objects.)\n- **Captions stay on-frame.** Cinematic mode hard-gates frame-overflow; Standard mode runs `check-overflow.cjs` as a WARNING (intentional bleed is the only exception — read the warning).\n- **Each caption ≥ 0.5s on screen** — shorter = unreadable.\n- **Word timings must match transcript.json within 80ms** — a caption firing 500ms off-beat destroys the scene illusion. Cinematic runs `check-timing.cjs --strict` before rendering (via render-and-composite.sh); THEME mode enforces the same timings at compile time instead (make-theme's sequential transcript matcher + verbatim completeness gate — drift is a compile error). Never pack multiple transcript words into one entry (e.g. `\"FUTURE OF\"` or an `IT` + line-break + `ALL` stack with one start/end) — the second word inherits the first's timestamp and fires early. Split them into separate word entries with their own timings, even if you want them on the same visual line (use CSS `white-space` / natural wrap instead of `<br>`). Creative substitutions where caption text ≠ transcript (e.g. `\"15%\"` replacing `\"fifteen percent\"`) are supported — register them in `CREATIVE_SUBS` inside `check-timing.cjs`.\n- **Group windows must envelop their words** — `group.in ≤ min(word.start)` and `group.out ≥ max(word.end)` for every group. If `group.in` is later than a word's start, the word is silently delayed until the container mounts (we've shipped 800ms lag bugs from this). The validator enforces this.\n- **No two caption groups may overlap in both time AND screen region** — overlapping-in-time captions create text-on-text pileups. Options: (a) **spatial separation** — place each group in a non-overlapping vertical band so they can coexist (memory-wall cascade style); (b) **handoff** — set the earlier group's `out` ≤ the next group's `in` so only one is on screen; (c) **deliberate layered typography** — add `\"allow_overlap\": true` on one of the groups to silence the validator. The validator estimates each group's vertical bbox from its CSS and flags collisions. Pick (a) by default — it's what makes cinematic-cream feel like a poem accumulating, not a subtitle track replacing itself.\n- **Screen-blend fails on bright backgrounds (>180 luminance).** **Cinematic** templates are cream + `screen` and that DNA is **locked** (the plan can't recolour them) → on a bright backdrop they wash out, so pick `ink` (letterpress built FOR bright surfaces) or the `anchor` theme (opaque rail surface) rather than overriding a look.\n- **Don't animate `letter-spacing` or `filter:blur` on word entrance** — inline-block reflow causes line-jumps.\n- **CoreML banned for matting** — the onnxruntime CoreML EP's mixed-precision partitioning corrupted face alpha (observed with the previous RVM engine; don't re-try it). Matting is CPU-only (~2 fps @1080p ≈ 2-3 min per 10s clip; budget for it on long clips).\n\n---\n\n## Dependencies\n\n- **HyperFrames CLI:** plugin installs use the bundled manifest-pinned launcher. Source contributors can use a built checkout (`packages/cli/dist/cli.js`) via `HYPERFRAMES_ROOT`, the skill’s source tree, or `~/Downloads/hyperframes`.\n- **Node-first; two Python touchpoints via `uvx` (no manual installs):** transcription runs WhisperX through `uvx` (word-level timings; falls back to an existing word-level `transcript.json`), and Theme's `drawon` setpiece shells `python3 scripts/gen-stroke-path.py` at compile time. Everything else runs on the toolchain hyperframes already ships: matting via the hyperframes CLI's **`remove-background`** (u2net_human_seg; weights auto-download once, ~168 MB, to `~/.cache/hyperframes/`), image/alpha math via **`sharp`**, layout/occlusion/overflow via **`puppeteer`**, plus **`ffmpeg`**. Install Sharp, Puppeteer, and GSAP in the caption project as described in **Runtime prerequisites** above. The helpers check that project first and retain checkout dependency lookup for source contributors.\n- **Transcription = WhisperX via `uvx`** (word-level timings + alignment; no manual install — `transcribe.cjs` drives `uvx whisperx`). Falls back to an existing word-level `transcript.json` if present.\n- **Source video** — `matte.cjs` / `transcribe.cjs` auto-resolve `source.mp4` (or glob the clip / read `hyperframes.json`), so `hyperframes init --video X.mp4` needs no manual rename.\n- **fps** — `matte.cjs` extracts at the source's native rate and records `matte.fps`; `render-and-composite.sh` uses that so the matte stays frame-aligned.\n- Matting weights are NOT bundled: `matte.cjs` shells the hyperframes CLI's `remove-background`, which downloads u2net_human_seg (~168 MB, Apache-2.0) once to `~/.cache/hyperframes/background-removal/models/`. First prepare on a fresh machine needs network for that one download.\n\nIf a hard dependency is missing, STOP and ask the user — don't silently skip steps.\n\nFile v1.0.25:dna/README.md\n\n# DNA registry — pick a visual language, not a preset\n\nA **DNA** is a complete, art-directed visual language: typeface, palette logic, motion\ngrammar, and hero orchestration. It **parameterizes per scene** instead of shipping a\nfixed look: the accent color is sampled from THIS scene, the contact shadow falls along\nTHIS scene's light, embed text blur matches THIS scene's depth-of-field, and the hero's\nentrance amplitude follows how hard the word was actually spoken (RMS).\n\nThis replaces the template grab-bag. Six deep languages × scene adaptation beats 54\nshallow presets — every render is already fitted to its footage.\n\n## Category lock (deliveries field, enforced by the compilers)\n\nEvery classic DNA's **home is Cinematic (column)** — that is where all ten were\nbuilt and validated. (Standard/rail mode was retired 2026-06-12; the verbatim-rail\nneed is served by the `anchor` theme. The old rail combos are archived outside\nthis repo and are not distributed with the skill.)\n\n## The ten\n\n| DNA             | Register       | Scene fit                                       | Voice                                                                                                                                                                       |\n| --------------- | -------------- | ----------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |\n| **cream**       | premium-warm   | dark / mid warm scenes (band luma < 150)        | Inter, warm cream, screen blend, glowing emergence hero. The poetic default.                                                                                                |\n| **ink**         | premium        | **bright scenes (band luma > 150)**             | Inter, near-black, multiply blend — type reads as _printed on_ the wall. Fixes the bright-scene hole.                                                                       |\n| **editorial**   | editorial-luxe | introspective / fashion / poetic, mid-dark      | Bodoni Moda, bone, _lowercase italic hero_ — magazine elegance over shout.                                                                                                  |\n| **keynote**     | tech-premium   | product / launch / founder updates              | Inter 800, opaque white, line-wipe reveals, hero wipes UP. Stillness = confidence.                                                                                          |\n| **documentary** | formal         | interviews, serious subject matter              | Inter, bone, **burn-in reveals**, no hero. Gravitas IS the style.                                                                                                           |\n| **loud**        | loud           | hype / sport / music / social                   | Anton, scene-sampled accent hero, single-unit slam + caption-layer ripple; **body announces in front** (`bodyLayer: \"fg\"` — k-pop depth, deliberate).                       |\n| **neon**        | loud-neon      | neon-noir / nightlife / tech-noir (dark scenes) | Orbitron + electric cyan; words flicker-ignite like tubes; the hero **powers ON** with a strobe and hums (glow 0.5).                                                        |\n| **glitch**      | loud-neon      | digital / hacker / AI / dystopia                | Space Grotesk; **RGB-split echo layers converge** as each beat lands; 0.10–0.12s machine percussion; landing bumps co-visible captions 2px.                                 |\n| **chrome**      | loud-luxe      | Y2K / fashion-tech / music                      | Audiowide cast in a **liquid-metal gradient** (background-clip:text on the word spans); one sheen sweep crosses the hero during the hold.                                   |\n| **velocity**    | loud-sport     | sport / automotive / fitness                    | Teko italic; every word arrives **along its motion vector** (streak + skew settling upright); the hero passes through with blurred speed-trail echoes and looms while held. |\n\n### fx fields (hero block) the engine understands\n\n`entrance: emergence | settle | slam | rise | wipe-up | flicker-on | streak` ·\n`glow` (0–0.5) · `echoes: [{dx,dy,color,opacity,blur}]` (converge-on-land duplicates) ·\n`sheen: true` (gradient sweep — pair with `wordCss` background-clip) · `wordCss`\n(per-word treatment; required for background-clip:text, which can't clip through\ncomposited children) · `ripple` (px — landing bump, captions only) · `loom` / `breathe`\n(hold-life) · `letterBlur` · `bodyLayer` (top-level: default layer for narration lines).\n\n## How to pick (agent)\n\n1. Read `safe-zones.json` → `heroAnchor.bandLuma` + `palette.temperature`.\n2. Bright band (>150) → **ink** (never fight a bright scene with cream/screen).\n3. Else pick by content register: poetic/warm → **cream** · introspective/luxe →\n   **editorial** · product/tech → **keynote** · serious/interview → **documentary** ·\n   hype/social → **loud**.\n4. State the pick + why in one line; the user decides (SKILL.md Step 0 still applies).\n\n## Authoring\n\n- Cinematic mode: `cinematic.json` → `\"dna\": \"<name>\"` (drop the `template` field).\n- Locked per DNA: family, palette scheme, blend, motion curves, hero orchestration.\n- Open per group (unchanged): size / weight / style / case / spacing + planes.\n- `var(--accent)` is available in per-group CSS — it resolves to the scene-sampled accent.\n\n## What the engine generates from the DNA (no authoring needed)\n\n- **Hero three-act**: co-visible captions dim 0.35s before the hero lands (setup) →\n  entrance with per-letter stagger, amplitude ∝ spoken loudness (impact) → slow breathe\n  - glow until exit (afterglow).\n- **Scene optics**: depth-match blur on embed captions, light-direction contact shadow.\n- All deterministic — measured from files, no randomness.\n\n## Climax selection — what earns the promotion\n\nPick the **payoff, not the topic**: the words you would quote when paraphrasing the\nbeat. \"The stars were only limited by the pixel size\" — the payoff is **pixel size**\n(the surprising claim), not \"stars\" (the subject). Test: would the word alone, big on\nscreen, make the listener nod? A topic word makes them wait.\n\n- **Phrases are legal** (`match: \"pixel size\"` / `\"skin food\"`): a peak is a semantic\n  unit, not necessarily one token. 2–4 words max — beyond that it's a sentence, not a\n  peak.\n- **Prefer clause-final** — nothing left dangling to place after the lift.\n- **Never strand a determiner**: lifting \"stars\" out of \"The stars…\" leaves an orphan\n  \"The\" that vanishes mid-air. The compiler auto-absorbs a line-leading article into\n  the promoted phrase; mid-line, match the phrase including its article.\n- Sizes are TARGETED by the compiler, not just floored — author sizes are hints:\n  - **apex = frame event**: under 88% of usable width → raised toward a 93% fill.\n    Std height cap = `sizeRange[1] × 1.25` (≤46cqh; formal register ×1.0) — a std apex\n    bursts a rail, not a body composition. Cinematic cap = `sizeRange[1]`.\n  - **short words fill with TRACKING**: when the height cap binds before the width\n    target (SHINE — 5 glyphs), the compiler letterspaces up to +0.32em until the word\n    owns ~88% of the width (film-title craft: HER / DUNE). Long words fill by scale,\n    short words by air — impact is WIDTH-led either way.\n  - **minor = damped beat in the apex family**: `max(3× rail, 0.55× apex-final)`,\n    width-capped. A peak that reads as a label is a bug.\n  - **lockup context is RATIO-LOCKED**: kicker/tail = `0.26× hero` (clamped\n    0.05–0.085·h), recomputed after any width-fit raise. Don't hand-tune context\n    sizes — size the hero; the context follows like a poster system.\n\n## Climax placement — the lockup is the default, NOT the only composition\n\nA hero line takes `\"placement\": \"subject\" | \"column\"`:\n\n- **subject** (default) — the **ORBIT lockup** centered ON the subject. NOT a centered\n  sandwich: context anchors to the hero's EDGES like a poster kicker/tagline —\n  pre-context flush-left off the hero's top-left corner, post-context flush-right off\n  its bottom-right — so the small lines land BESIDE the head/shoulder (on the scene),\n  never on the face, and therefore EMBED (bg) like everything else. The eye reads a\n  diagonal: kicker ↘ HERO ↘ tail. Only the hero crosses the silhouette (the\n  matte-occlusion showcase). The context's job is reading continuity — it gets the\n  margins; the hero's job is the cinematic event — it gets the subject. Right when the\n  peak is THE dramatic moment: cream / loud / glitch / neon / velocity lean here.\n- **column** — the apex stays IN its column's reading flow and BURSTS it (oversized,\n  full hero motion + dim/glow privileges, but composed within the sentence's home\n  zone; spills toward the subject naturally when oversized). Right when the content\n  is a continuous read and the column is the narrative's home: keynotes, tutorials,\n  explainers, documentary. Registers: keynote / documentary / editorial lean here.\n\nApplying ONE composition to every peak is monoculture — the same failure the motion\nreview caught. Pick per beat: a 15s explainer might run column peaks throughout and\nsave the subject lockup for the single apex. (A third pattern — the editorial split,\nbg context upper + fg hero lower-third — lives in references/composition-craft.md and\nis the next placement candidate.)\n\n## The motion language (five layers, each DNA answers all five)\n\nA DNA's animation is not a parameter tweak on a shared fade — it's a distinct physics:\n\n| Layer                  | What it governs                                                                                                    | Where it lives                                        |\n| ---------------------- | ------------------------------------------------------------------------------------------------------------------ | ----------------------------------------------------- |\n| **word entrance**      | how each word materializes on its beat ({y,x,blur,scale,rot,dur,ease} or burn)                                     | `motion.soft/present/impact`                          |\n| **line exit**          | direction + speed of leaving, optional top-to-bottom cascade                                                       | `motion.exit` ({dur,y,x,scale,ease,stagger})          |\n| **hold-life**          | what text does while on screen — stillness is a deliberate choice                                                  | `hero.breathe` / `hero.loom` (0 = print-still)        |\n| **hero orchestration** | the three-act entrance signature (emergence / settle / slam / rise / **wipe-up**) + glow / letterBlur / **ripple** | `hero.*`                                              |\n| **timing physics**     | the duration & easing register everything obeys (percussive 0.15s ↔ atmospheric 0.6s)                              | implied by the values above — keep one family per DNA |\n\nShipped signatures: cream = _light condenses_ (blur-settle, drift-up exits) · ink =\n_letterpress_ (scale-press, no float, dead-still) · editorial = _the pen glides_\n(x-glide with the italic, cascade-out page turns) · keynote = _surgical reveal_\n(burn words + line wipe + hero wipe-up, exits never move) · documentary = _burn and\nhold_ (motion's absence IS the language) · loud = _percussion_ (back-eased punches,\nalternating tilt, hero ripple that bumps every caption — never the footage).\n\nRules inherited from hyperframes motion-principles: exits faster than entrances; vary\nease families BETWEEN DNAs (not within one); `.out` enters, `.in` leaves; never two\ntransform tweens on one element in the same window (wipe-up is clip-path so the\nafterglow scale channel stays free; loom subsumes breathe).\n\n## Adding a DNA\n\nCopy an existing `dna/<name>.json`, change the voice, keep the schema. The engine\n(`modes/cinematic/engine.html`) and both compilers consume it as-is. A DNA earns its\nplace by being a _distinct voice with a reason to exist_ — not a recolor.\n\nFile v1.0.25:modes/cinematic/README.md\n\n# Cinematic mode (pure embed) — one engine, six DNAs\n\n> Cinematic mode compiles **[../../dna/](../../dna/README.md)** through\n> **[engine.html](engine.html)** (`make-composition.cjs`). The old per-template HTML\n> shells are retired — `cinematic-cream` maps to `dna: \"cream\"` automatically; the other\n> archived templates (memory-wall / champion / portrait-header, in [\\_archive/](_archive/))\n> remain as design references only.\n\nUse this mode for pure-embed asks (no rail): brand film, hype, social reel, showcase.\nThe **DNA** locks the visual language (type, palette scheme, blend, motion grammar, hero\nthree-act); **safe-zones v2** parameterizes it to the scene (sampled accent, light-\ndirection contact shadow, depth-match blur); **the agent decides layout only** (planes,\nblocks, per-line typography within the DNA).\n\n## Workflow\n\n1. `bash scripts/prepare.sh <project>` → matte ∥ transcript ∥ envelope → safe-zones v2\n2. Pick a DNA ([../../dna/README.md](../../dna/README.md)): bright hero band → `ink`,\n   else by register (cream / editorial / keynote / documentary / loud). Recommend, let\n   the user pick.\n3. Author `<project>/cinematic.json` — `\"dna\": \"<name>\"` + thought-blocks (schema:\n   `scripts/make-cinematic.cjs` header)\n4. `node scripts/make-cinematic.cjs <project>` → plan.json → engine-compiled index.html\n5. `node scripts/preview-frames.cjs <project>` → § Visual QA (failure checks + the 5\n   positive checks in [../../references/reference-bar.md](../../references/reference-bar.md))\n6. `bash scripts/render-and-composite.sh <project>` → gates → final.mp4\n\n## What the engine generates (never author these)\n\n- word timings from the transcript; accumulate-within-block / page-flip-between-blocks\n- the hero hand-off + **three-act orchestration** (dim → RMS-coupled per-letter entrance\n  → breathe + glow), per the DNA's `hero` block\n- scene tokens: `--accent` (sampled), contact shadow, depth blur\n- reading order, re-slot from measured heights, hero size/collision post-pass\n\n## What you DON'T do\n\n- Override `.cap` color / blend / shadow / filter / motion curves — that's the DNA.\n  Scene fights the look → pick a different DNA (bright → `ink`), never recolor.\n- Hand-position the hero into a clean margin (it belongs ON the subject, ~30–55%\n  occluded — safe-zones `heroBands.best`).\n- Add full-frame grades/textures over the footage (hard rule: the video ships untouched).\n\n## Adding a DNA\n\n`dna/<name>.json` — copy one, change the voice (see [../../dna/README.md](../../dna/README.md)\n§ Adding). The engine consumes it with no code change. A DNA must be a distinct voice\nwith a reason to exist, not a recolor.\n\nFile v1.0.25:themes/README.md\n\n# THEME mode — composed visual constitutions\n\nTheme mode is the third compiler (`scripts/make-theme.cjs`), beside Standard and\nCinematic. It exists because \"mode\" was a bundle of orthogonal axes pretending\nto be one switch. A theme DNA composes its identity from registries implemented\nONCE in the compiler — **paradigms are the unit of code; DNAs are the unit of\nidentity**. A new look is a JSON file; only a genuinely new paradigm/setpiece\n(rare) touches the compiler.\n\n```\ntheme DNA = body PARADIGM   how the transcript surface lives\n          × body LAYER      fg-alpha (rail.html channel) | bg-embed\n          × hero SETPIECE   the climax choreography\n          × front FX        flash / rings / sparks / scanband / crowdflash / paflash (fg, over subject)\n          × PLATE budget    charge-dim (in-page) + punch/shake/grain (_postfx.sh)\n          × LINKAGES        declarative theme interactions\n```\n\nStandard and Cinematic were, in retrospect, two fixed points of this space.\nStandard is now RETIRED (2026-06-12): its rail×embed-climax point is served by\nthe `anchor` theme (rail paradigm × settle setpiece — the quiet default).\nCinematic remains a separate compiler; do not re-implement it as a theme yet.\n\n## Unification roadmap (strangler fig — interface first, engines later)\n\nThe user-facing model is already unified (SKILL.md Step 0): one catalog of\nLOOKS; classic looks pick a DELIVERY (rail | column), themed looks bind their\nown. \"Standard/Cinematic\" are delivery/compiler names, not modes. Remaining\nphases, each gated on need — never rewrite for tidiness alone:\n\n- **Phase 2 — one authoring schema.** `lines`/`minors`/`hero` are already\n  ~90% shared between standard.json and theme.json; a router that translates a\n  single `caption.json` into the engine-specific file removes the last\n  user-visible seam. cinematic.json's blocks/planes are the odd one — map the\n  common fields, pass engine-specific ones through.\n- **Phase 3 — engine convergence.** Port a classic delivery into make-theme\n  ONLY when something forces it (e.g. a classic DNA wants a plate budget or a\n  setpiece). Acceptance bar: blind A/B on the cap_multi regression scenes vs\n  the old compiler — swap engines only when indistinguishable or better. Until\n  then the old compilers are the reference implementation of 8 rounds of\n  validated typography (lockup/orbit, multi-climax, ratio-lock, per-plane\n  legibility, occlusion adjudication) — that machinery is the moat, not debt.\n\n## Body paradigms (registry)\n\n| paradigm      | surface                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                          | layer | entrance verbs                                                                                                                                                                                                                                                                                           | exit verbs                                                                                                                                                                                                                         |\n| ------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |\n| `rail`        | lower-third lines, replace                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                       | fg    | `stamp` (1f hot appear→crush→recoil→cool), `flick` (tube double-flick), `bootflick` (corrupted-cyan boot 1–2f → settles to UI color; minors keep a cyan/red registration shadow). `stamp` minors cool to `palette.minorCool` (default amber)                                                             | `drop` (mass falls), `powercut` (bright→ghost→off)                                                                                                                                                                                 |\n| `panel`       | docked glass console, accumulate + typed                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                         | fg    | `type` (width steps per word, $ prompt, caret blinks)                                                                                                                                                                                                                                                    | hold                                                                                                                                                                                                                               |\n| `poem`        | open-space stanzas, accumulate; letters CONDENSE from seeded dust                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                | fg    | `condense`                                                                                                                                                                                                                                                                                               | `drift` (stanza hand-off) + optional dispersal linkage                                                                                                                                                                             |\n| `takeover`    | full-frame hard-cut cards, plate dimmed throughout                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                               | bg    | `cut` (+ tension `creep` before silences, auto-detected)                                                                                                                                                                                                                                                 | hard cut / blur dissolve                                                                                                                                                                                                           |\n| `lastpage`    | manuscript rail (typed serif) + a seeded field of BLURRED apex-word instances haunting the room from t=0                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                         | bg    | typed rail; field breathes imperceptibly                                                                                                                                                                                                                                                                 | rail fades; field resolves at apex (see `rackfocus`)                                                                                                                                                                               |\n| `flaprail`    | split-flap word tiles on a chyron bar (tag + ROW flip-counter; optional odometer chip docks above the bar while the first emphasis line reads)                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                   | fg    | `splitflap` (3 flips: wrong word → wrong word → real word clacks on hot, squash → elastic settle → cools to bone; emphasis tiles keep amber text + underline)                                                                                                                                            | flip-to-blank cascade left→right (completes before the next line)                                                                                                                                                                  |\n| `ledboard`    | amber LED departure strip docked lower-third (`body.board` geometry); lines page by vertical scroll-snap (steps(6)) inside the board window; furniture: green platform tag, ticking seconds clock (stacked digit reel), ⚠ DELAYED blinks then pages to ● ON TIME ~0.7s after the apex, idle LED shimmer in dead air ≥1.2s                                                                                                                                                                                                                                                                                                                        | fg    | `ledwipe-mini` (per-word L→R LED column wipe, clipPath steps(len), + 1-frame arrival flicker; emphasis words get the LED-refresh double-blink; last word gets one late refresh)                                                                                                                          | scroll-snap page-up (next line pages in at the same instant)                                                                                                                                                                       |\n| `vhsrail`     | camcorder-playback lower third: every word carries a PERMANENT 1px red+cyan misregistration fringe (split clip-path halves); furniture: ▶ PLAY / ◀◀ REW top-left (REW gag in dead air ≥0.7s + the final rewind), ● REC top-right through the apex window, green timestamp bottom-right with 1 Hz blinking colon, idle tape-drift line in long silences, plus apex-coupled artifacts (1-frame white tear band + head-switching noise bar at REC FREEZE, scrub lines at the apex band on the hero rewind)                                                                                                                                          | fg    | `trackglitch` (instant on + 2-frame y-shake + 1-frame band-slice shift of the word's bottom half; emphasis = PAUSE jitter ±1px y oscillation + white noise-band flicker)                                                                                                                                 | REWIND streak left + flat-white echo copy + 2 scrub-line flickers                                                                                                                                                                  |\n| `hudrail`     | arcade-cabinet HUD: whole frame plays through CRT overlays (scanlines `body.scan` + corner vignette); docked HUD bar (`body.bar` geometry, 2px accent border, glow) with `body.tags` 1UP / CREDIT furniture — the 1UP tag blinks on the classic 0.45s attract cadence all clip; HUD yields (`body.yield`) while the boss lands                                                                                                                                                                                                                                                                                                                   | fg    | `coinpop` (instant-on + steps(2) scale pop from the baseline + 1-frame white coin-blip tick over the word; emphasis = 3-frame palette swap body→accent→magenta→body)                                                                                                                                     | checkerboard dissolve (words vanish in 2 alternating 2-frame passes, line off ≤ DUR−0.02)                                                                                                                                          |\n| `carbonstrip` | cold-war case-file transcript strip docked on the clearer side (manila gradient, punch holes, faint red `body.watermark`, typed `body.header` that types on by width steps); lines live on up to `body.rows` fixed slots; PAGES (broken at full rows, dead-air gaps > 0.7s, and after the hero hand-off line — it is filed away while the stamp owns the frame); the strip top-shadow breathes during the stamp's dead-still hold and the strip yields (`body.yield`) while it lands; the final page is pulled with the whole strip at clip end                                                                                                  | fg    | `hammer` (per-char sets 30ms apart, scale 1.35→1 in one frame = the stamp motif at low amplitude; seeded ±1.2px baselines + seeded ribbon-ink opacity; every strike kicks the strip 1px — y-channel locked while the strip itself tweens; emphasis words get a red ink ellipse drawn around them, 0.25s) | X-row over-strike (3 chunks per line, 0.05s apart, lines 0.16s apart) → carriage yank-down feed → return; roomy cadence in dead air, compressed when the next page is close — always completes before the next page's first strike |\n| `laserrail`   | concert lower third: lines centered on a bottom rail (`body.bottomPx`), uppercase laser-green words with a neon text-glow; emphasis words render at `body.minorScale`× with a hotter glow; word centers are BAKED at compile time from the measured char-width table (`assets/fonts/char-widths.json`, ≤3px vs browser ground truth) so the beam aim can't be skewed by font-load timing; per-word furniture = two thin beams (accent from top-left, magenta from top-right) that sweep to CONVERGE on the word 2 frames early + a convergence flare; beams dim ×`yield.beamDim` inside the apex window and the visible line dims to `yield.dim` | fg    | `igniteconverge` (beams converge at st−0.083 → flare pop → word snaps on at opacity 1 with brightness 1.85, crush 0.09s power3.in → 2-frame contact squash → elastic settle → brightness cools; emphasis = single 1-frame strobe off/on + lingering beams)                                               | sweeper beam crosses L→R in 0.13s and each word streaks away (x+44, blur 5) as it passes; line off ≤ DUR−0.02                                                                                                                      |\n\n| `stormrail` | rain-washed lower third: ambient rain = `body.rain.count` seeded thin streaks falling on repeating linear cycle tweens (pure f(t), one slanted wind vector, mid-fall at t=0); a second seeded set DOUBLES the rain for ~1s after the strike; lines sit on a bottom gradient scrim (stable reading surface); the visible line dims to `yield.dim` while the bolt lands and its text-shadow flips toward the strike (`hero.params.side`) for the flash frames | fg | `lightningpop` (the apex motif at low amplitude: instant on at brightness 2.2, y+4, scale 1.06 → 0.12s micro settle; emphasis = sheet-lightning double flash, 2-frame brightness 1.9/1.0/1.8 chain — the bg sky double-kisses at the same beats) | wash downward with the rain (y+22, blur 3, 0.18s power1.in); a line facing dead air washes early at lastWord+1.08s; line off ≤ DUR−0.02 |\n| `holorail` | volumetric projection plate: a scanlined glass slab (`body.plateW`×`plateH`, centered at `H − bottomPx`) with corner brackets + feed tag (`body.tag` / `theme.plateTag` override) boots like a projector turning ON (scaleY 0.03→1 expo + 1-frame flicker) and breathes a finite yoyo luminance cycle sized to the clip length; lines sit centered IN the plate (`lineDy` below plate center) with a constant chromatic-fringe text-shadow (−1.5px accent / +1.5px magenta) and drift slowly in rotationY (it IS a projection); the whole railwrap — plate included — dims to `yield.dim` while the apex lands | fg | `sweepboot` (the apex motif at low amplitude: instant opacity + rising clip-wipe reveal 0.15s + 1-frame x interference glitch at st+0.05/0.09/0.13; emphasis = 2-frame projection DROPOUT at `body.emphDelay`, restores brighter in `palette.em` with a hotter fringe, then settles back — the delay auto-shrinks near the line's exit so dropout+restore always complete) | projector cut — brightness 2.2 pop → collapse to a horizontal slit (scaleY 0.045, 0.07s power3.in) → to a dot (scaleX 0.015, 0.06s) → off at exit+0.15; the plate itself gets the same cut at DUR−0.14 |\n| `planktonrail` | deep-sea lower third on TWO coexisting row slots: even lines ride the upper row (`H − bottomPx − rowDy`), odd lines the lower (`H − bottomPx`) — couplets share the water like drifting strata. A line clears when its row is next needed (line i+2 in − 0.15) or ~0.3s after its couplet finishes speaking (staggered pair exits, +0.12s on odd rows); the pre-apex leftover clears at heroIn−0.21 while the FEEDING line (last to start before the apex) yields to `yield.dim` instead — no restore unless ≥1.2s of hold remains (it exits dimmed, by design); a line entering during the hold arrives at `yield.enter` and restores at I+`yield.post`; every line floats on a gentle y keyframe drift sized to its window. Furniture: `body.motes` = {count, near, seed} plankton motes on seeded drift walks f(t); the first `near` seed a ring around the apex heart (ring radius scales with hero fontPx/120) and get ATTRACTED to the bloom at I−0.19 (pulled 0.72× toward it, brightening to 0.42 at scale 1.7), wiggle through the hold, and sink at clip end | fg | `glowon` (the apex bloom at low amplitude: 2-frame opacity ramp + radial bloom from word center — scale 0.92→1, blur 5→0, brightness 0.45→1, 0.4s sine.out — while a cyan glow text-shadow grows in; emphasis = slow DOUBLE-BEAT propulsion pulse at st+0.52, 5-key brightness/scale 0.55s, guarded ≤ DUR−0.09) | light sinks — y+15 while dimming, 0.3s power1.in; the last line floats to the end of the clip; line off ≤ DUR−0.02 |\n| `sheenrail` | iridescent silk lower third on TWO coexisting row slots: even lines ride the upper row (`H − bottomPx − rowDy`), odd lines the lower (`H − bottomPx`) — couplets read together like a woven pair. Words are aurora-gradient ink (background-clip:text over a rose→teal→violet 300% wash whose backgroundPosition drifts f(t) 0%→300% across the whole clip — the body's hold life) above a blurred dark understroke (top+3px, blur 3); a soft radial band scrim sits under the rows (stable reading surface). Line scheduling mirrors planktonrail: a row clears when next needed (line i+2 in − 0.15) or shortly after its couplet finishes speaking (pairEnd + 0.23 + 0.17 on odd rows); the pre-apex leftover clears at heroIn−0.32 while the FEEDING line stays and dims with the railwrap yield (`yield.dim` at I−`pre`, restore at I+`post` — band included); the last line holds to the end | fg | `flowon` (the ribbon's flow at low amplitude: 2-frame opacity + y8 settle 0.38s sine.out; emphasis = ONE bright sheen sweep — a white highlight marches through the word once at st+0.16 (backgroundPosition −80%→180%, 0.42s power2.inOut) + a brightness 1.55 pulse) | dissolve UPWARD — words to y−12, blur 3, 0.32s power1.in staggered 0.035; each line releases `body.streaks` seeded light streaks (rose/teal/violet verticals) that float up from its row (y 14→−78, 0.62s, start clamped ≤ DUR−0.66) |\n| `scoperail` | oscilloscope lower third: a dark phosphor scope band (`body.bandH`, top border + glass shadow) hosts HUD readouts (`body.hudL` left, sample-rate `· TRIG` + amber trig LED right), a graticule, and an ALWAYS-ALIVE waveform — an SVG polyline rebuilt from baked point-set states (seeded pseudo-RMS envelope at ~12 states/s + 3-frame triplets on word arrivals: speech grows the amplitude, silence flattens it, every word kicks a local gaussian burst at its slot center — burst x BAKED at compile time from the measured char-width table; the hero words kick WIDE bursts spread from `HG.x` toward `HG.x + 0.6·halfW`, where the trace surges to write). Body lines sit centered on the band above the trace with reserved-width slots (sizer + clip — no re-center drift as words arrive); the whole band (lines + wave + HUD) dims to `yield.dim` while the apex writes. At the FINAL transcript word the scope DIES: band x-shake, trig LED panics red then dies, the trace explodes to max-amp chaos for ~0.3s, flatlines, collapses to a single dot (scaleX 0.015) and `body.siglost` stamps with a 2-frame flicker (all clamped ≤ DUR−0.02) | fg | `tracewipe` (the apex pen-write at low amplitude: clip wipes the word in L→R over 0.10s + a 2-frame just-traced flash — flash-white instantly, settle to phosphor green; the death word settles to `palette.em` amber; the trig LED blinks on every arrival; emphasis = a ticking frequency-readout chip above the word, 3 seeded ascending values stepping every 2 frames) | the line collapses INTO the waveform (y+30, 0.13s power2.in) — early into dead air (lastWordStart+0.67), else just before the next line needs the slot; the last line collapses as the trace flatlines (min(NZ+0.33, DUR−0.2)) |\n| `paperrail` | stop-motion craft table: a kraft strip (torn top edge via seeded polygon clip-path, washi-taped corners, handwritten `body.tag` in the tag font) docked lower-third (`body.strip` geometry, centered by default); body words are cream paper chips (`palette.chip` on `palette.chipEdge` border, hard dy3 shadow) on centered replace-lines; the whole strip jitters ±0.7px every 4 frames (handmade stop-motion weave); the hero slot keeps a red paper star (`palette.accent`, rail↔climax hand-off) that pulses 1.18→1.06→1 as the apex chips land; the FINAL transcript word is an accent-red sign-off chip; emphasis words get a washi-tape slap (steps(2) scale 1.7→1 at `word.start + body.tapeDelay`, chip pressed 1 frame); strip + lines yield (`body.yield`) while the apex lands | fg | `place` (the papermat drop at low amplitude: opacity full in 0 frames + seeded ±2.5° rotation, y −16 + scale 1.06→1 steps(2) 0.12s; hold = rotation-only ~5fps wobble) | `peel` (rotation −8°, y −40, fade, steps(3), staggered 0.03s; last line clamped to finish on frame) |\n| `popuprail` | pop-up book page: a cream page strip (`body.strip` geometry, center foldline, top border + paper glow inset) unfolds up from the bottom edge (scaleY 0.02→1 back.out at 0.14 — the stage is built) and hosts TWO reading rows (`body.rowA`/`rowB` px above H) that alternate per line; the whole page breathes a gentle y drift sized to the clip; the hero slot keeps a blue paper star chip (`palette.accent` clip-path star, rail↔climax hand-off — spliced into the line whose window holds heroIn); emphasis words land at `body.emPx` weight 800 with a contact squash (1.06/0.93 set → elastic settle) the strip FEELS (1px dip), then flick-wobble (5-key rotationX, squeezed to fit before the fold) with a blue paper strut leaning in behind (scaleY back.out(2)); the FINAL transcript word is a blue sign-off with a strut; page yields (`body.yield`) while the centerfold lands; on the final word the whole page TILTS 2° +x6 (paper physics, `body.tilt:false` opts out) and the strip folds shut on the last frame (scaleY power3.in ≤ DUR−0.017) | fg | `popup` (the centerfold motif at low amplitude: opacity 1 in 0 frames + rotationX 85→0 baseline hinge, 0.25s back.out(1.4), back-shadow grows as it rises) | `fold` (the page closing: rotationX 88 power2.in 0.12s, shadow off, L→R wave staggered 0.045s, slot hidden at +0.13; fold at next line in / lastWord+`body.maxHold` in silence, wave clamped ≤ DUR−0.017) |\n| `chalkrail` | night-lecture chalk band: a deep-green board strip (`body.strip` geometry, worn top hairline, eraser smudges, handwritten `body.tag` header in the tag font, two chalk sticks bottom-right) docked at the bottom edge hosts `body.rows` reading rows that ACCUMULATE — lines fill the rows top→bottom in phases of rows.length; when the next phase needs the rows an ERASER SWIPE clears the board (blurred light band sweeps across 0.42s at `prevEnd + body.eraseDelay` clamped clear of the next phase, the old words SMEAR out scaleX 1.3 + x+24 + blur(8) staggered 0.06s, faint ghost smudges fade in and STAY — chalk never fully erases); the FINAL transcript word lands in yellow chalk (`palette.em` sign-off); the band yields (`body.yield`) while the apex writes | fg | `scribble` (the chalkwrite verb at low amplitude: 2-frame opacity pop + scribble clip-reveal steps(4) 0.12s + y4 drop-settle + 6 seeded dust specks puffing at the word base, seeded ±2° tilt; emphasis = double underline dashes drawn beneath at st+0.22/+0.38) | eraser-swipe smear between phases (scaleX 1.3, x+24, blur 8, power2.in 0.18s); the final phase holds to clip end |\n| `markerrail` | street handstyle rail: centered marker lines on a bottom rail (`body.bottomPx`), `palette.body` cream paint with a heavy drop text-shadow; emphasis words take ALTERNATING paint colors (`palette.em`/`em2`, yellow/magenta) + a spray-underline SWOOSH (dash-revealed squiggle viewBox-stretched to the word) with 8 seeded overspray dots that pop and fade; the FINAL transcript word lands BIG (1.5×) in `palette.em` with mini apex-impact physics (crush 0.09s power3.in → 2-frame contact squash → elastic settle — it rides the plate punch); the line visible at the tag's onset yields (`body.yield`, restore only with runway before its buff — else it exits dimmed) | fg | `flick` (the spray-write verb at low amplitude: opacity full in 1 frame at scale 1.18, seeded ±6° rotation settling to ±2°, 1-frame squeak skewX 7, back.out(2.2) settle 0.18s) | ROLLER BUFF (gray paint roller scaleX-sweeps the line 0.15s power2.in, words vanish under it, the wet smear lingers at 0.15 + blur(2) before fading); the last line is never buffed — the sign-off stays painted |\n| `brushrail` | washi paper band: a deckle-edged band (`body.strip` geometry, two faintly rotated `paperHi`→`paperLo` under-sheets at 0.42 + a translucent `sheetHi`→`sheetLo` main sheet with warm inset shadow) swipes in L→R behind a traveling brush-head shadow (clipPath wipe 0.2s power2.out + blurred `palette.bh` gradient head — the ink gesture at low amplitude), then the under-sheet breathes faintly (0.32 Hz baked f(t)); body lines are centered replace-lines on the band (`fonts.body` serif at `body.fontPx`/`body.weight`, seeded ±1° tilt); emphasis words get a DOUBLE curved ink underline (`palette.ulInk`, fat 7/0.3 wash + tight 2.6/0.95 core, dash-drawn at st+0.12/+0.15); band + the line under the gesture yield (`body.yield`) while the apex lands; the band exits with the last line on one final R→L wipe | fg | `brush-wipe` (the gesture at low amplitude: clipPath inset reveal L→R 0.16s power2.out with a blurred per-word brush-head shadow riding the reveal edge, then vanishing) | wet-cloth WIPE (the reveal in reverse: the line clipPath wipes R→L over 0.10–0.14s power2.in with an erase brush head riding it, at min(next line − 0.136, lastWord + `body.maxHold`); the LAST line and the band wipe together at DUR−0.142) |\n| `inkrail` | sumi ink on still water: up to THREE fixed reading rows at the bottom center (`body.bottomPx` = last-row center, `body.rowGap` apart; rows cycle in pairs, the closing line takes the third row — the final thought ACCUMULATES instead of replacing); `fonts.body` Mincho serif at `body.fontPx`/`body.weight`/`body.letterSpacing`, `palette.body` ink over a paper-white text-shadow halo (`palette.paper`/`paper2` glow stack — readable on bright skies) with a duplicate blurred dark halo per word; an ink-settling RULE (86%-width gradient hairline, scaleX 0→1 over 0.9s) draws beneath each line at firstWord+0.12; emphasis words shed a faint ink wisp curling off the word top (st+0.55, ±dir by line parity, skipped near clip end); the rows visible under the bloom yield (`body.yield` — dim at heroIn+`delay`, restore at heroIn+`post` only with runway before the exit) | fg | `bleed` (the apex bloom at low amplitude: opacity snaps in 2 frames; blur 12→0 + scale 1.25→1 power2.out carry the bleed — late words bleed faster so they resolve before clip end; the dark halo overshoots scale 1.5→1.04 while its opacity decays 0.5→0.25) | RE-DISSOLVE (blur 9 + scale 1.13 + fade + y−8 power1.in — ink dispersing back into the water) as the line-after-next approaches (= the next occupant of its row, outDur 0.5−0.05·i, never mid-read); the closing TWO lines persist to clip end |\n| `ransomrail` | kidnapper's collage rail: every transcript word is a cutout chip in one of 4 cycled recipes (cream/`fonts.body` Anton, black/`fonts.dark` Inter-800, newsprint/`fonts.news` Special Elite, accent/`fonts.accent` Permanent Marker — emphasis words snap to the accent chip, never two identical chips in a row) PASTED along the bottom rail line (`body.bottomPx`) with seeded crookedness (rot sign alternates with line+word parity, baseline dy leans the other way), seeded ±px sizes (`body.fontPx`−2..+1, emphasis +3) and torn edges (seeded 12-vertex tear polygon, same paper stock as the apex); chips carry a hard dy5 offset drop-shadow; lines breathe (5-key scale keyframes) through their hold; the whole railwrap yields (`body.yield`, container opacity — never contests per-chip channels) while the apex letters slam | fg | `paste` (the apex slam motif at low amplitude: opacity full in 1 frame at scale 1.12 / rot×1.5 / y−8 → power3.in crush 0.08s → 2-frame contact squash 1.05/0.95 → elastic settle; emphasis = 1-frame RE-PASTE lift at `body.emphDelay` — shadow grows to a flight blur, rotation flips sign for good, then crush + squash + elastic again, rip-runway-guarded) | RIPPED off one by one (rotation ±16/17° throw + y−30 + fade, power2.in, staggered 0.045s) with a 1-frame paper-tear flash under each chip — timed so the rip tail overlaps the next line's pastes a beat (collage chaos, by design); the last line tears off clamped to finish on frame |\n\nfg body uses the **rail.html alpha-webm channel** (true alpha — dark panels and\nscrims work; never the screen-blend index_fg path, which can only add light).\n\n## Hero setpieces (registry)\n\n| setpiece      | what happens                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                   | notes                                                                                                                                                                                                                                                                                                                                                                                                                         |\n| ------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |\n| `detonation`  | charge-dim → sheared stencil slices crush in hot → SNAP into register → squash → elastic settle → paint cools to bone → bars/ticks/designation-tag deploy → shear-apart exit                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                   | pairs with fx flash/rings/sparks + plate punch/shake                                                                                                                                                                                                                                                                                                                                                                          |\n| `decode`      | slot-machine glyph reels (steps() ease, seek-safe) lock left→right with RGB jitter → lock snap → CRT power-off exit                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                            | pairs with `redact-until-hero`                                                                                                                                                                                                                                                                                                                                                                                                |\n| `drawon`      | the word is WRITTEN stroke-by-stroke from a single-line font (Hershey) — per-stroke paths revealed sequentially at constant pen speed, nib rides `getPointAtLength`, hops at pen lifts; then hum + buzz dip                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                    | any word, zero tuning: `gen-stroke-path.py` lays out glyph pen-paths at compile time                                                                                                                                                                                                                                                                                                                                          |\n| `cpslam`      | acid-yellow stencil word slams in BEHIND the subject with REAL diagonal notch cuts (SVG mask = true transparency over footage), chromatic split that settles to a PERMANENT cyan/red registration error, seeded glitch ticks, katakana tag, glitch-out exit                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                    | bounded hold (~2.6s default, `params.hold` / `hero.exitAt` override) — a climax is an event, not wallpaper                                                                                                                                                                                                                                                                                                                    |\n| `assembly`    | (inline, poem) apex word condenses BIG while star particles fly into it                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                        |                                                                                                                                                                                                                                                                                                                                                                                                                               |\n| `colorflip`   | (inline, takeover) accent-color crush card + dim kick + squash/settle + loom                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                   | pairs with fx flash + plate punch/shake                                                                                                                                                                                                                                                                                                                                                                                       |\n| `flapboard`   | split-flap departures board takeover: floodlight dim/scrim swing on, the housing clacks down (rotationX, squash, elastic) behind the subject, one giant flap tile per char flip-cycles 3–5 seeded wrong glyphs then LOCKS left→right across `params.lockWindow` (hot flash → cools to bone) under baked clack y-jitter; lock-complete = big clack + amber glow flicker + rule wipe + steps() status-ticker; hold = loom + one tile re-flutter; exit = tiles flip to blank cascade, housing flips away                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                          | pairs with `fx.crowdflash` (pops at lock start/end) and `plate.punchOffset = lockWindow` so the plate punch lands on the lock-complete clack, not the hero onset                                                                                                                                                                                                                                                              |\n| `ledwipe`     | transit LED board takeover: cool scrim + dim charge, an amber dot-matrix panel EXPANDS upward (scaleY steps(6)) behind the subject, header tags + unlit LED ghost boot-flicker alive, then the word ILLUMINATES column-by-column (clipPath L→R, steps ≈ 1.4×chars over `params.wipeDur`) hot → cools to amber; contact = cell squash + elastic settle + PA \"ding\" double brightness pulse; an 8-phase chasing LED-marquee border runs to the exit; hold = glow breathe + seeded-look LED dropouts; exit = word pages up out of the window, panel collapses back down                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                           | pairs with `fx.paflash` (amber PA flash at wipe-complete contact). Panel geometry derives from fontPx (`panelH = 44 + 1.42×fontPx + 18`); `params.panelW` fixed, width-fit shrinks the type only. Plate: grain-only (no punch) is the demo register                                                                                                                                                                           |\n| `vhsosd`      | REC FREEZE camcorder slam: permanent lens vignette (`params.vignette`) + backlight-comp scrim (peak = `plate.charge`, cools to 0.69×) kill the bright sky; the word crushes in as a huge OSD readout — 3 horizontal clip-path tracking bands (`params.bandClips`) sheared (`params.bandShear`) + blurred (power4.in over `params.crush`) SNAP into register on the contact frame, squash → elastic settle; red/cyan misregistration copies cool from ±`misregIn` to ±`misreg` px and keep BREATHING through the hold under tape-weave y jitter (`params.weave`) + slow loom; auto-iris pump (white wash 0.20, dark dip = `plate.dim`); exit at `I + params.hold` = REWIND streak left + flat-white echo copy                                                                                                                                                                                                                                                                                                                                                                   | pairs with the `vhsrail` paradigm, which emits the fg REC-FREEZE artifacts (tear band, head-switching bar, apex scrub lines, ● REC) at the same I/X times. Plate: grain-only (no punch) is the demo register; hold-life keyframes auto-clamp inside [contact, exit]                                                                                                                                                           |\n| `bossintro`   | arcade BOSS INTRO: the room dims (gradient `plate.dim`) behind a CRT edge glow (cyan base wash that pulses on the last emphasis word before the boss and spills magenta on the strobe + the final word); `params.tag` (WARNING) blinks 2× above the word rect; ~COLS×ROWS (`params.rows`, cells 0.7×/0.5× fontPx) seeded \n\nFile v1.0.25:_meta.json\n\n{\n  \"ownerId\": \"kn77d06grj6xqp3dqwkk4bavhn89pegt\",\n  \"slug\": \"embedded-captions\",\n  \"version\": \"1.0.25\",\n  \"publishedAt\": 1791142135378\n}\n\nFile v1.0.25:references/aesthetic-principles.md\n\n# Aesthetic Principles for Cinematic Captions\n\nThe 18 rules that separate \"designed\" caption work from \"preset-generated\" caption work. This is the single most important reference in the skill — every plan.json and every Standard-mode HTML should be checked against these before committing.\n\nThe whole competitive thesis: every AI caption tool in 2026 (Veed, Submagic, Opus Clip, Captions.ai, CapCut) is a **preset picker**. They hand the user a box of crayons. This skill is a **director** — it exercises judgment. Beat them on taste, not feature count.\n\n---\n\n## The 18 rules\n\n### 1. The subject always wins\n\nCaptions exist to serve the face, not compete with it. If the caption draws the eye before the speaker, the caption has failed. **Before shipping any plan.json, ask: \"does the caption or the subject read first?\"** If caption wins, shrink, dim, or move it.\n\n### 2. Occlude, don't hover\n\nWorld-class caption design puts text _into_ the scene. Letters that pass behind a shoulder, mic, or head feel diegetic. Letters that float uniformly above the lower-third feel like PowerPoint. Use the matte pipeline — that's the moat over every competitor.\n\n### 3. Contrast is a hierarchy problem, not a brightness problem\n\nA white box behind text is a failure of taste. Priority order:\n\n1. `mix-blend-mode: overlay` or `screen` picking up scene luminance\n2. 2–3px dark stroke + soft drop shadow\n3. Narrow semi-opaque gradient bar (NOT a solid box)\n4. Dim the background plate by 10–15% locally\n5. (LAST RESORT) A hard white pill box — banned on cinematic directions\n\n### 4. Kill the constant lower-third\n\nFixed-bottom captions are monotone. Caption zone shifts with shot:\n\n- Tight close-up → upper sidebar or crown\n- Mid-shot → embedded on back wall / foam / whiteboard\n- Wide → classic lower-third offset to non-subject side\n\n### 5. One family, two weights maximum\n\nHierarchy lives in **weight** (e.g., 500 → 800), not in **font**. Mixing Montserrat + script + serif in one clip is the #1 amateur tell. Ship Inter, SF Pro, Söhne, GT America, Aktiv Grotesk, or Neue Haas Grotesk with ≥5 weights and compose with weight + size.\n\n### 6. Tracking tightens as size grows\n\nDisplay-size (>40pt) wants negative tracking (-10 to -30 units, or `-0.015em` to `-0.035em`). Body-size (14–20pt equivalent) wants positive tracking (+5 to +15, or `+0.005em` to `+0.015em`). Apple SF Pro's optical-size model is the reference. Submagic defaults do the opposite — they look cheap because of it.\n\n### 7. Cap height ≈ 3.5–5% of frame height for body captions\n\n- 1080p vertical (1920 tall): body 65–95px, emph/hook 130–170px\n- 1080p landscape (1080 tall): body 40–55px, emph 70–100px\n- Hormozi-size (~7–9% of frame) is **hook** size, not body size. Reserve for punchlines.\n\n### 8. Never italicize for emphasis in video text\n\nItalics are a **print** convention for flow inside a paragraph. On 24fps motion they read as \"tilted\" not \"stressed\". Emphasize with weight (extrabold), color (single accent), or size (1.3–1.6×). Italics allowed only for:\n\n- Literal quotation of written material\n- Foreign word\n- Thought vs spoken dichotomy\n\n### 9. Color discipline: one hue + neutrals\n\nPick **one** saturated accent per video for keyword highlights. Hormozi's yellow+green+red works for him because his content is already loud. Cinematic = single accent + white/bone/charcoal. Default palette:\n\n- Warm white `#F5EFE6` on dark\n- Graphite `#1A1A1A` on light\n- Accent chosen from scene sampling\n\n### 10. Animate transform only, never the letter itself\n\n`letter-spacing`, `filter:blur`, `font-weight` animations cause inline-block reflow → line-jumps. Animate `translateY`, `scale`, `opacity`, `clip-path`. This is locked from the embedded-captions debugging.\n\n### 11. Minimum 0.4s per word visible\n\nBBC reading speed is 160–180 wpm (0.33–0.38s/word). Cinematic feel wants more breathing room — 200–220ms stagger per word, hold full phrase 0.4–0.6s before exit. Entries under 150ms read as frantic.\n\n### 12. Stagger is the primary expressive axis\n\nSame phrase, different stagger, totally different feel:\n\n| Stagger | Feel                             |\n| ------- | -------------------------------- |\n| 40ms    | machine-gun, urgent, TikTok-hook |\n| 80ms    | conversational, default          |\n| 150ms   | deliberate, documentary          |\n| 250ms+  | poetic, ceremonial               |\n\nPick from content tone, not default to one value.\n\n### 13. Emphasis escalates _within_ a phrase, not _between_ phrases\n\nEvery word bolded = no emphasis. Structure:\n\n- 70% plain body\n- 20% slight lift (color OR weight, not both)\n- 8% full emphasis (bigger, brighter, held longer)\n- 2% climax (biggest, held 1.5s, breath before + after)\n\n### 14. Break rhythm once per 30s\n\nEvery-word-same-way = eye adapts, stops registering motion. Plant a rhythm-break every ~30s:\n\n- Phrase enters from opposite direction\n- A single word at 2× size\n- A beat of pure silence with no caption\n- A color shift\n\nThis is what separates Submagic-preset work from something designed.\n\n### 15. Caption what adds, cut what restates\n\nTranscribe everything. **Display** 70–85%. Remove:\n\n- Filler (\"um\", \"like\", \"you know\", \"I mean\")\n- Self-corrections (\"I think... I mean actually...\")\n- Obvious visual echoes (\"as you can see here\" while pointing)\n\nEditorial judgment — no existing AI caption tool does this. It's pure upside.\n\n### 16. Segment on breath, not on duration\n\nChunk at natural pauses ≥ 250ms. A caption spanning a breath-break feels wrong. Whisper word-level timestamps make this trivial.\n\n### 17. Safe zones per platform, always\n\n- Use the safe boxes in `/hyperframes-studio` (Safe zones) for wide and vertical\n  framings; that skill owns the values.\n\nBake into the layout solver. Never eyeball.\n\n### 18. Letterbox/pillarbox is caption real estate\n\nBlack bars on 9:16 from 16:9 source? Those bars are the caption home. 2.35:1 cinematic frame on 16:9 export? Serif quotation in the letterbox reads as documentary.\n\n---\n\n## Rhetorical judgment rules (editorial)\n\nThese are the \"what to caption\" decisions. No existing tool exercises them.\n\n- **Filler suppression** — off/light/strong. Default: light for documentary, strong for vlog.\n- **Self-correction folding** — display only the final version.\n- **Breath-group segmentation** — split on silences >250ms, never mid-phrase.\n- **Semantic emphasis** — LLM-pass per phrase: \"which 1–2 words carry the meaning?\" Highlight those. Don't default to stressed syllables or loudest words.\n- **Silence honor** — if speaker pauses 1.5s+ for rhetorical effect, don't back-fill with lingering prior caption. Let silence breathe.\n- **Quote-sensing** — \"he said, quote…\" or air quotes → italic (one of the allowed italic exceptions).\n- **No `[laughs]` / `[sighs]`** — those are accessibility captions. For aesthetic captions, they pollute the frame.\n\n---\n\n## Self-critique checklist (run before rendering)\n\nThe agent's own pre-render pass. Flag violations:\n\n- [ ] Does the caption or subject read first? (Rule #1)\n- [ ] Any hard white pill boxes? (Rule #3 — banned on cinematic)\n- [ ] Still lower-third for a close-up? (Rule #4)\n- [ ] More than 2 weights or more than 1 font family? (Rule #5)\n- [ ] Italic used for emphasis? (Rule #8 — banned)\n- [ ] More than 1 saturated accent color? (Rule #9)\n- [ ] Letter-spacing or filter:blur animating? (Rule #10 — banned)\n- [ ] Stagger same for documentary and vlog? (Rule #12)\n- [ ] Every group emphasized? (Rule #13)\n- [ ] 30s+ with no rhythm-break? (Rule #14)\n- [ ] Displaying every filler word? (Rule #15)\n- [ ] Caption crossing breath-break? (Rule #16)\n- [ ] Caption in platform UI zone? (Rule #17)\n\nAny \"yes\" to a violation question → regenerate the affected segment.\n\nFile v1.0.25:references/anti-patterns.md\n\n# Anti-Patterns\n\nYou default to these. Stop.\n\nEach entry: the bad habit, what it produces, what to do instead. Written in the voice of someone who has watched an agent do this 10 times — because we have.\n\n---\n\n## Layout\n\n### You default to center-aligned crown.\n\n`.crown-plane { left: 0; right: 0; text-align: center }` is the template default, and you'll leave it on every video because it worked once. On a subject that sits right-of-center (Jobs in a 16:9 frame), the body eats the middle 60% of the word and \"THE BEATLES\" becomes \"THE \\_\\_\\_ S\".\n\n**Before using the default crown**, run the three conditions in [layout-heuristics.md § Crown placement](layout-heuristics.md):\n\n1. Subject centered within 10% of frame center\n2. Clean zones ≥ 15% on each side\n3. Crown width > subject width + 400px\n\nIf any one fails, **move the crown to the larger clean zone** with a narrower container and smaller font. Don't keep centering a word that's about to be 60% occluded.\n\n### You compute text leftmost as `plane_left + padding`.\n\nThat's correct for left-aligned. Wrong for the templates you're using, which are **right-aligned** on the main column and **center-aligned** on crowns. The real leftmost depends on the word width.\n\n| Alignment | leftmost_x                                         |\n| --------- | -------------------------------------------------- |\n| Left      | `plane_left + padding_left`                        |\n| Right     | `plane_right − padding_right − longest_line_width` |\n| Center    | `plane_center − longest_line_width / 2`            |\n\nCompute it against the **longest wrapped line**, not the whole phrase. \"four very talented guys\" wraps to 3 lines — the widest is \"talented\" (~8 chars), not 23.\n\n### You treat `plane_left` as the text's leftmost edge.\n\nIt isn't. With `left: 180px` and a right-aligned 468px word, the text starts 104px **before** `plane_left` relative to the plane, but actually lands somewhere inside the plane box. The plane box just sets the coordinate space — the text positions inside it based on alignment. Check the compiled output, not the plane attribute.\n\n### You center text on the frame when the subject is off-center.\n\nLook at the subject's body center, not the frame center. Jobs sits at x=1100 in a 1920 frame. The **scene's** center of gravity is 1100, not 960. Center-aligning text to 960 is center-aligning to the empty left third, not to anything meaningful.\n\n### You copy-paste position values from memory-wall to a new video.\n\n`top: 40, right: 30, width: 720, rotateY: -13` worked for a 1280×720 frame with an acoustic foam wall on the right. It will not work on a 1920×1080 frame with a bookshelf backdrop. Run the 6-item checklist in [scene-types.md](scene-types.md) for the new video before reusing any numbers.\n\n---\n\n## Typography\n\n### You use template default font sizes regardless of column width.\n\nTemplate defaults (66/78/92/140) are tuned for a ~560px column. When you bump the plane to 700px+ and don't touch the fonts, the captions feel underweight — small text swimming in negative space.\n\nUse the [Font-size × column-width matrix](typography-presets.md). For a 700px column, that's 78/108/128/220 — a 30-40% bump across the board. Re-check pillarbox safety after bumping (bigger right-aligned text extends further left).\n\n### You pick `intro` style for every first caption.\n\nIntro = italic, smaller, contemplative. Fine for filler discourse markers (\"You know,\", \"So,\", \"Well,\"). **Not** for the first line when it's actually the thesis (\"I've had this kind of upbringing\"). Read the words semantically, not positionally.\n\n### You skip emphasis entirely because \"every caption looks clean.\"\n\nA story without an emph or crown is typographically flat — viewers can't feel the crescendo. Reserve at least ONE emph per video for the line that lands. Skip it only for truly monotone content (policy statements, warnings).\n\n### You give every group its own style.\n\nStyle is supposed to signal _hierarchy_. Using `intro`/`phrase`/`emph`/`dream`/`crown` all in one 15-second video means none of them signal anything. Pick 2-3 styles max for a short clip.\n\n### You only use the 5 preset class names.\n\n`intro / phrase / emph / dream / crown` are scaffolding, not a closed set. The canonical `memory-wall.html` uses `cap-1 / cap-2 / cap-3 / cap-4` — **position-indexed** with bespoke typography per position. That's how it achieves the 3-line right-aligned cascade on the climax. You can't express \"this cap has a hanging indent at position 2\" with `\"style\": \"dream\"`.\n\nThe `\"style\"` field in plan.json accepts **any string** — it becomes `class=\"cap-<string>\"`. Define the class in `custom_css`:\n\n```json\n\"custom_css\": \".cap-1 { font-size: 78px; ... } .cap-2 { padding-right: 44px; }\",\n\"groups\": [{\"id\": \"cg-0\", \"style\": \"1\", ...}]\n```\n\nWhen the scene needs per-position bespoke typography, do this. Don't force-fit into `intro / phrase / emph`.\n\n### You regenerate from plan.json when a canonical example already has the answer.\n\nWhen scene framing matches `references/example-renders/memory-wall.html` or `champion.html`, clone that HTML into `<project>/index.html` and only swap the GROUPS array. Don't re-derive the design from presets — you'll lose the specific per-position typography that took many iterations to validate. See [bespoke-vs-presets.md § The clone-and-tweak workflow](bespoke-vs-presets.md).\n\n---\n\n## Blending\n\n### You default to `mix-blend-mode: overlay`.\n\nOverlay works on **mid-tone** backgrounds. On dark bookshelves (<60 luminance), overlay makes white text render as black. On sunny backgrounds (>180 luminance), overlay blows out. Check the actual luminance of the caption region in a sampled frame. Pick:\n\n- mid-tone surface (60-180) → `overlay`\n- dark surface (<60) → `screen`\n- bright surface (>180) → `normal` with opaque text\n\n### You animate `letter-spacing` on the word entry.\n\nYou saw a \"typewriter breath\" effect somewhere and added `letterSpacing: \"0.04em\"` to the `fromTo` \"from\" state. Result: inline-block reflow every frame → the whole `.cap` line box recomputes → captions visibly jump between rows. See the \"Some → line 2\" bug from memory-wall.\n\n**Animate only opacity + transform.** If you want a breath effect, use `scale` or `y`, not letter-spacing.\n\n---\n\n## Animation\n\n### You fade both the group container AND each word.\n\nDefault feels safer: fade the container in, then fade each word in. But container.opacity × word.opacity produces a non-linear curve — at progress 40% the combined opacity is 16%, not 40%. Visible result: captions seem to \"pop\" into view around halfway.\n\nSet container opacity to 1 at entrance via `tl.set`, then fade only the words. Single-layer opacity = linear perceived fade.\n\n### You stack captions in a flex column.\n\nFlex-direction: column with multiple `.cap` children looks tidy in source. At runtime, hidden captions (`visibility: hidden`) still reserve flex space. The newly-entering caption shows up at the _bottom_ of the column instead of the designed position — landing in the gesture zone, getting clipped.\n\nUse `position: absolute` for each `.cap` inside the plane so hidden ones don't occupy layout. Both shipped templates do this.\n\n### You start the timeline at t=0.\n\nCaption at exactly t=0 feels like it was there before the video started. Offset 0.1-0.3s (hyperframes motion-principles.md agrees). Same for the very last caption — let it exit before the video fades.\n\n---\n\n## Scene admission\n\n### You trust that \"looks like one speaker\" = \"is one speaker throughout.\"\n\nTV archive clips cut to B-roll mid-sentence. Interview clips insert cutaways to the interviewer. You didn't check frames beyond the first one, so you rendered captions across a shot transition. Result: captions designed for Subject A are placed relative to Subject B's position (or empty frame) for half the render.\n\nSample frames at 20%, 50%, 80% BEFORE planning. If the scene changes, trim to the largest single-subject segment.\n\n### You ignore baked-in captions / pillarbox / watermarks.\n\nYou saw black bars on the sides and didn't computing the safe-zone. Your captions cross the pillarbox. Or: the source already has burned-in subtitles and you added more — two caption systems fighting for attention.\n\nRun the **letterbox probe** first. If the source has existing captions, refuse with: \"source already captioned, adding more would conflict.\"\n\n### You ship Whisper's transcript without checking timings against the beat.\n\nTranscription is Whisper (via `transcribe.cjs`, no API key) — good word timings, but not infallible: a word can land with a near-zero duration or a timestamp a beat off. This skill is verbatim + on-beat, so `check-timing.cjs --strict` (80ms tolerance) is the gate, not a suggestion — fix drift in `plan.json` before rendering, and never pack two transcript words into one timed entry (the second inherits the first's timestamp and fires early).\n\n---\n\n## Matting\n\n### You enable CoreML for the matting ONNX.\n\nIt's the Apple way, obviously faster. No. CoreML partitions the ONNX graph across providers. The mixed-precision boundary produced (observed with the previous RVM engine) alpha=30 inside the subject's face while background correctly reads 0. Captions shine through face. Pin the CPU execution provider only (`onnxruntime-node`). Our `matte.cjs` already does this; don't \"optimize\" it by re-adding CoreML.\n\n### You pick a matte model by \"general vs human.\"\n\nDECISION FLIPPED 2026-06-12 after a 5-model × 6-scene A/B with caption renders: the matte's job here is CAPTION LAYERING, not prop fidelity. `u2net_human_seg` (via hyperframes `remove-background`) usually excludes thin offset furniture (mic boom arms) from the matte — words stop being sliced by booms, which beat PP-MattingV2's prop-preserving behavior on real caption videos. It is NOT surgical: large salient objects near the subject (telescope rigs) can still leak in — always sample frames_fg/. Known cost: HELD products can drop out intermittently (captions pass in front) — route product-demo climaxes away from held objects. `isnet-general-use` lost outright (backlit-hair collapse). birefnet-portrait (MIT) beat everything semantically (keeps held items AND drops furniture) but is 928 MB / ~7 s-per-frame CPU — a future quality tier, not the default.\n\n---\n\n## Grouping\n\n### You caption every word.\n\nTranscripts are verbose — \"you know, um, I, I mean, you know\" is 5 words from a single beat. Captioning all of them clutters the screen and breaks reading rhythm.\n\nEditorially drop filler. Use the rules in [caption-grouping.md](caption-grouping.md): merge short fragments, cut repeated discourse markers, keep the meaning, trim the noise. You are writing typography to support speech, not a court transcript.\n\n### You group by fixed word-count.\n\n\"3 words per caption, always\" makes every caption look the same and fights the natural cadence of speech. Break on sentence boundaries, 250ms+ pauses, and semantic units instead. Caption sizes will naturally vary from 2 words to 5 — that's correct.\n\n---\n\n## The meta anti-pattern\n\n### You read one reference doc and skip the rest.\n\nYou'll read this file alone and feel covered. These anti-patterns reference concepts defined in `layout-heuristics.md`, `typography-presets.md`, `scene-types.md`. If you haven't read those, the fix advice here won't make sense.\n\n**Order of reading for a new video**:\n\n1. SKILL.md (decision gate + pipeline + pre-flight probes)\n2. bespoke-vs-presets.md (**first check if a canonical example fits — clone if so**)\n3. scene-types.md (template selection — all 4 wall conditions)\n4. layout-heuristics.md (positions, sides, crown, font scale, pillarbox formula)\n5. typography-presets.md (font-size × column-width table, starting points)\n6. caption-grouping.md (word → group)\n7. **This file last** (to catch yourself before committing to plan.json)\n\nIf you're pressed for time, still read this one — it flags the failures you're about to make.\n\nFile v1.0.25:references/bespoke-vs-presets.md\n\n# Bespoke design vs. presets — when to override, when to clone\n\nThe 5 preset styles (`intro / phrase / emph / dream / crown`) and the 3 templates (`wall-embed`, `corner-column-crown`, `portrait-header`) are **scaffolds**, not rules. The best renders we've shipped all override presets for specific groups because **typography is a per-scene decision**, not a general rule.\n\nIf you only use presets, your render will look generic. If you only copy existing renders, your skill won't adapt. The right workflow is:\n\n1. **Decide the shape first** (template choice, plane position, blend mode).\n2. **Check if a canonical example is close enough** → clone and tweak words + timings.\n3. **Otherwise start from presets** → override per-group via `custom_css`.\n\n---\n\n## Canonical example renders\n\nFull HTML for two validated renders is in `references/example-renders/`:\n\n| File               | Scene                                                   | What makes it work                                                                                                                                                                                         |\n| ------------------ | ------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |\n| `memory-wall.html` | Introspective monologue, right-side foam wall, mid-tone | Right-aligned cascade, per-group bespoke sizes (`cap-1` 78 italic / `cap-2` 66 italic + right hanging-indent / `cap-3` 72 upright / `cap-4` 90 uppercase). `mix-blend-mode: screen` for the dark-ish foam. |\n| `champion.html`    | Podcast interview, cluttered bookshelf, 1920×1080       | Upper-left column + center-stage crown. Tuned preset class sizes (`cap-intro` 52 / `cap-phrase` 60 / `cap-emph` 70 / `cap-crown` 140). `screen` blend reads the shelves through text.                      |\n\n**When a new scene matches one of these closely** (similar framing, similar subject-center, similar backdrop type): clone the HTML and only replace the GROUPS array + word timings. Don't re-derive the design from presets — you'll lose the specific choices that took many iterations to validate.\n\n---\n\n## When presets are wrong\n\nYou'll reach for presets like `\"style\": \"emph\"` when what the scene really needs is:\n\n### \"This cap is at position N and deserves its own treatment\"\n\n`memory-wall.html` uses `cap-1 / cap-2 / cap-3 / cap-4` — **position-indexed**, not role-indexed. Each one is a bespoke design for a specific phrase at a specific point in the arc:\n\n- cap-1 (soft opener, 4 words): 78px italic 600 — feels like a whisper\n- cap-2 (dreamy modifier, 3 words): 66px italic 500 + `padding-right: 44px` — hanging indent creates ragged right-edge stagger\n- cap-3 (turn, 2 words): 72px upright 700 — the syntactic pivot, no italic\n- cap-4 (climax, 4 words): 90px uppercase 900 — three lines cascade right-aligned\n\n`phrase/emph/intro` can't express \"this cap has a hanging indent\" or \"this cap is the syntactic pivot\". When that matters, invent your own class names:\n\n```json\n{\n  \"template\": \"wall-embed\",\n  \"custom_css\": \"\n    .cap-1 { font-size: 78px; font-weight: 600; font-style: italic;\n             letter-spacing: -0.01em; }\n    .cap-2 { font-size: 66px; font-weight: 500; font-style: italic;\n             padding-right: 44px; }\n    .cap-3 { font-size: 72px; font-weight: 700; letter-spacing: -0.015em; }\n    .cap-4 { font-size: 90px; font-weight: 900; letter-spacing: -0.03em;\n             text-transform: uppercase; line-height: 1.0; }\n  \",\n  \"groups\": [\n    { \"id\": \"cg-0\", \"style\": \"1\", \"words\": [...] },\n    { \"id\": \"cg-1\", \"style\": \"2\", \"words\": [...] },\n    ...\n  ]\n}\n```\n\nThe `\"style\": \"1\"` field becomes `class=\"cap-1\"` on the element — any string works, no validation.\n\n### \"The template's blend doesn't suit this backdrop\" → pick a different template, do NOT override\n\n**Cinematic mode does not override colour or blend.** A template's `mix-blend-mode` + fill are **locked DNA** — `make-composition.cjs` ignores `plan.cap_color` / `blend_mode` / `text_shadow` / `text_filter`. Selecting a template commits to its look; the only agent-authored things are layout (planes/positions) and per-group typography.\n\nSo use the caption-region luminance to **choose** a template that already fits — never to recolour one:\n\n| Region luminance                  | What fits                                                                                     | Why                                                                        |\n| --------------------------------- | --------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------- |\n| < 60 (dark / low-key)             | a cream + `screen` template (`cinematic-cream`, `memory-wall`, `champion`, `portrait-header`) | light text glows, picks up the scene                                       |\n| 60–180 (mid-tone)                 | a cream + `screen` template still reads (add a scrim via Standard if marginal)                | text picks up texture                                                      |\n| > 180 (bright: window, pale wall) | **none of the cream/`screen` Cinematic templates — they wash out**                            | → use **Standard mode** (opaque rail, set per the chosen template) instead |\n\nIf the scene is bright and the cream/`screen` look washes out, that's the signal to switch to **Standard mode** (which sets opaque colour in the HTML), not to recolour a Cinematic template into something it isn't.\n\n### \"Hanging indent / outdent / letter-width tweak\"\n\nThese are per-group affordances you'll occasionally need. Express via `custom_css`:\n\n```css\n#cg-2 {\n  padding-right: 44px;\n} /* right-aligned: shrinks right edge, creating left outdent */\n#cg-2 {\n  padding-left: 44px;\n} /* left-aligned: offset right start */\n.cap-emph .w:first-child {\n  font-size: 110%;\n} /* oversize first word only */\n```\n\nThe `#cg-N` selector always works because `make-composition.cjs` writes `<div id=\"cg-N\" ...>` for every group.\n\n### \"Caps should accumulate (flex stack) instead of swap\"\n\nAll three templates default to `position: absolute` on `.cap` inside their plane — caps stack at one spot and only the active one shows (single-caption swap). This is correct for `portrait-header` and `corner-column-crown`, where each caption replaces the last.\n\n`memory-wall` uses **flex column accumulation** — captions pile up like a poem. The template doesn't default to this, so override via custom_css:\n\n```css\n.wall-plane {\n  display: flex;\n  flex-direction: column;\n  justify-content: center;\n  align-items: flex-end; /* or flex-start for left-aligned */\n  text-align: right;\n  gap: 14px;\n}\n.wall-plane .cap {\n  position: static; /* un-do template's absolute */\n  top: auto;\n  right: auto;\n  max-width: 100%;\n}\n```\n\nWith this + staggered `in` / `out` times, cap-0 fades in at t=0.2, cap-1 at t=2.55 (below cap-0 in flex order), cap-2 at t=4.90 (replaces both as they fade out together at t=4.85) — this is how the memory-wall poem pages work.\n\n### \"Font size doesn't match the scene\"\n\nTemplate preset sizes are tuned for a specific column width + frame size. Don't fight them — override:\n\n```json\n\"custom_css\": \".cap-intro { font-size: 52px; } .cap-phrase { font-size: 60px; }\"\n```\n\nThen check [typography-presets.md § Font-size scales with column width](typography-presets.md) for what to aim at given your plane's actual dimensions.\n\n---\n\n## The clone-and-tweak workflow\n\nFor a new video that's clearly similar to an existing canonical example:\n\n```bash\n# 1. Scaffold the project\nhyperframes init <project> --non-interactive --video <video.mp4> --skill=embedded-captions\n\n# 2. Matte + transcribe\nnode scripts/matte.cjs <project>\nnode scripts/transcribe.cjs <project>\n\n# 3. Copy the canonical HTML instead of writing plan.json\ncp references/example-renders/memory-wall.html <project>/index.html\n\n# 4. Replace GROUPS array with the new transcript's grouping (hand-edit index.html)\n\n# 5. Render directly (skip make-composition.cjs since we're not using plan.json)\nbash scripts/render-and-composite.sh <project>\n```\n\nThis skips the preset-based plan.json entirely. Use when:\n\n- Subject framing, shot composition, and backdrop type are similar to the example\n- You just need to swap the words and timings\n- The bespoke typography from the example is what you want\n\n**Don't clone when**:\n\n- Subject position differs significantly (e.g. centered vs off-center)\n- Scene luminance / blend-mode needs are different\n- You want to experiment with new typography\n\nIn those cases, start with plan.json + custom_css and iterate.\n\n---\n\n## Rendering history\n\n`render-and-composite.sh` now snapshots `index.html` + `plan.json` into `<project>/history/` with a timestamp before every render. If the user says \"the previous one was better\", diff against the latest snapshot:\n\n```bash\nls <project>/history/\ndiff <project>/history/index-20260422-203947.html <project>/index.html\n```\n\nThis lets you recover a design you iterated away from, without re-reading the agent transcript.\n\nFile v1.0.25:references/caption-grouping.md\n\n# Caption Grouping\n\nHow to turn the Whisper word-level transcript into the `groups[]` array of plan.json.\n\n## Goal\n\n**Each group is one visual phrase**: enters, reveals word-by-word, exits. Rule of thumb: 1 group ≈ 1 comma-to-comma clause or 1 breath of speech.\n\n## Input\n\n`transcript.json` from `transcribe.cjs`:\n\n```json\n{\n  \"words\": [\n    { \"text\": \"Some\", \"start\": 0.24, \"end\": 0.44, \"type\": \"word\" },\n    { \"text\": \" \", \"start\": 0.44, \"end\": 0.48, \"type\": \"spacing\" },\n    { \"text\": \"memories\", \"start\": 0.48, \"end\": 0.82, \"type\": \"word\" }\n  ]\n}\n```\n\nDrop `type: \"spacing\"` entries; you only need words.\n\n## Break boundaries\n\nCut a new group at ANY of:\n\n1. **Pause ≥ 500ms** (gap between word.end[i] and word.start[i+1]) — speaker took a breath.\n2. **Sentence terminator** — word ends with `.`, `?`, `!`, or an em-dash-like pause.\n3. **Strong comma** — `,` followed by pause ≥ 250ms.\n4. **Discourse reset** — words like \"but\", \"so\", \"and then\", \"you know\" starting a clause often merit their own or new group.\n5. **Group reaches 6 words OR 2.5 seconds** — whichever first. Long groups feel like subtitles, not embedded typography.\n\nHard constraints:\n\n- Minimum 2 words per group (1-word exceptions: interjections like \"Wait.\" or the crown line).\n- Minimum 0.5s on screen. If a group is less, merge into neighbor.\n- No overlapping groups — at most one visible at a time.\n\n## Timing the group\n\nFor a group with words `w[0]..w[n-1]`:\n\n- `in` = `w[0].start - 0.08` (enter slightly before first word)\n- `out` = `min(next_group.in - 0.05, w[n-1].end + 0.6)` (linger ~0.6s after last word, but don't collide with next)\n\nThe last group extends to the video end if needed.\n\n## Style & tone (cross-reference)\n\nSee `typography-presets.md` for how to pick `style` and `tone` per group. Work left-to-right through the groups and:\n\n1. First group: default `intro` + `soft`.\n2. Watch for emphasis signals (ALL CAPS in transcript is rare but possible; more often it's semantic — superlatives, proper nouns).\n3. Escalate tone into `present` once the monologue shifts from setup to statement.\n4. Reserve `crown` for at most ONE group, typically the final line.\n\n## Editorial surgery is allowed\n\nYou do NOT have to caption every word. It's fine to:\n\n- **Drop filler** like extra \"you know\"s, \"um\"s, \"I mean\"s if they bloat the visual pace.\n- **Condense** a 6-word run into 4 by cutting function words, as long as the meaning and timing remain truthful.\n- **Skip the whole thing** during obvious silence or non-speech (laugh, music interlude).\n\nEditorial rule: you are writing typography to support the speech, not a court transcript. Keep meaning, trim noise.\n\n## Example (champion)\n\nTranscript: \"You know, for me I've had this kind of upbringing, had the great foundation and, you know, I've achieved incredible things. I was dreaming of becoming number one in the world and becoming a Wimbledon champion\"\n\nGroups after editorial pass:\n\n```json\n[\n  {\n    \"id\": \"cg-0\",\n    \"style\": \"intro\",\n    \"tone\": \"soft\",\n    \"words\": [\"You\", \"know\", \"for\", \"me\"],\n    \"in\": 0.1,\n    \"out\": 1.45\n  },\n  {\n    \"id\": \"cg-1\",\n    \"style\": \"phrase\",\n    \"tone\": \"soft\",\n    \"words\": [\"I've\", \"had\", \"this\", \"kind\", \"of\", \"upbringing\"],\n    \"in\": 1.4,\n    \"out\": 3.35\n  },\n  {\n    \"id\": \"cg-2\",\n    \"style\": \"phrase\",\n    \"tone\": \"soft\",\n    \"words\": [\"the\", \"great\", \"foundation\"],\n    \"in\": 3.5,\n    \"out\": 5.35\n  },\n  {\n    \"id\": \"cg-3\",\n    \"style\": \"emph\",\n    \"tone\": \"present\",\n    \"words\": [\"I've\", \"achieved\", \"incredible\", \"things\"],\n    \"in\": 6.05,\n    \"out\": 8.3\n  },\n  {\n    \"id\": \"cg-4\",\n    \"style\": \"dream\",\n    \"tone\": \"present\",\n    \"words\": [\"dreaming\", \"of\", \"becoming\", \"number\", \"one\"],\n    \"in\": 8.5,\n    \"out\": 10.4\n  }\n]\n```\n\nPlus the crown:\n\n```json\n{ \"id\": \"cg-crown\", \"style\": \"crown\", \"words\": [\"Wimbledon\", \"Champion\"], \"in\": 10.8, \"out\": 12.08 }\n```\n\nNotice \"had\" was dropped from cg-2 (\"had the great foundation\" → \"the great foundation\"), \"I was\" was dropped from cg-4, and \"a\" was dropped from crown — all for visual cadence.\n\n## word.start/end inside groups\n\nPass through the original timestamps from the transcript. Don't retime individual words — only the group `in`/`out`. The word-level karaoke reveal inside each group uses the original w.start.\n\nFile v1.0.25:references/composition-craft.md\n\n# Composition craft — EMBED track only\n\nThis is the deep \"how to lay a caption INTO the scene\" manual — it governs the **embed**\ntrack only (captions composited behind the subject). The default **rail** track (standard\nlower-third subtitle, where most text lives) has its own, much simpler spec → [rail.md](rail.md).\nRead this before authoring any **promoted** phrase, in either authoring mode (template\nCinematic `plan.json` or Standard `index.html`).\n\nTopics: phrase grouping, planes & clean-zone anchoring, zone coherence, climax pop,\nreadability, edge breathing, the occlusion 3-step judgement, accumulation, and persistence.\n\n> **Two-track model (SKILL.md § Caption model).** Render states are `drop` / `rail` / `embed`.\n> The `bg / fg / hybrid` axis used throughout the text below is **superseded**: read **\"bg\" as\n> embed** (behind subject / in-scene), **\"fg\" as rail or promote-out** (in front), and \"hybrid\"\n> as simply rail + embed coexisting. Everything here applies to the **embed** track — don't\n> apply this craft to the rail.\n\n(Granular sub-topics also have their own files — see SKILL.md § Shared knowledge.)\n\n---\n\n### Caption layer: BG vs FG (aesthetic axis)\n\nSeparately from template choice, decide whether captions are **embedded** (behind subject, matte occludes — default \"embed\" cinematic feel) or **announced** (in front of subject, no occlusion — poster / lower-third feel):\n\n- **bg** — classic embed. Captions sit behind the matte, subject's body partially occludes letters. **Default for any cap that fits the frame.** Partial occlusion is the cinematic feature — small narrator caps with face nibbling 20-30% of letters reads as embedded texture; climax words with face crossing 30-40% reads as Vogue masthead. The eating is what sells the embed. Don't shrink to avoid it.\n- **fg** — captions float on top, subject can't occlude. **Fallback for when bg cannot work.** Two scenarios:\n  - **Climax fg**: the climax word is so big at impactful size that subject + frame width together leave no place for bg to render at all (subject fills frame AND climax phrase is long; bg would be eaten past readability into broken).\n  - **All-fg fallback**: rare, only when the entire scene has no usable bg zone for any caption.\n- **hybrid** (per-group `layer:` field) — when one cap (usually the climax) hits the \"no place to fit bg\" wall but the others (smaller body caps) still work fine in bg. **Body bg + climax fg** is the canonical hybrid pattern.\n\n**Rule of thumb:**\n\n- Small narrator/emphasis captions: bg almost always. Subject eating a few letter-edges = embed feature.\n- Climax: try bg first, sized to maximally cross subject. If that size won't fit frame width, AND no clean zone elsewhere, use fg.\n- Don't invert the hybrid: body fg + climax bg is the WRONG direction (sacrifices the easy bg embed wins on body, while the climax that needed protection from over-occlusion gets sent into the fire).\n\n**Hybrid worked example — Startup_Host (subject fills frame).**\n\nSubject fills the frame top-to-bottom (head_top y=126 / 10% from top). For body captions there's no large clean zone, but small italic narrator caps in upper corners get eaten ~25% by subject — that's the embed showcase. For the climax \"SUCCESSFUL STARTUP\" at maximally cinematic size, bg would be eaten 40-50%+ → past cinematic into broken. Climax goes fg.\n\n```json\n{\n  \"caption_layer\": \"bg\",\n  \"groups\": [\n    {\n      \"id\": \"cg-0\",\n      \"plane\": \"body\",\n      \"layer\": \"bg\",\n      \"css\": \"font-size: calc(0.055 * var(--h)); font-style: italic; ...\"\n    }, // narrator bg — face eats edges, embed texture\n    {\n      \"id\": \"cg-1\",\n      \"plane\": \"body\",\n      \"layer\": \"bg\",\n      \"css\": \"font-size: calc(0.085 * var(--h)); font-weight: 800; text-transform: uppercase; ...\"\n    }, // emphasis bg — Vogue masthead\n    {\n      \"id\": \"cg-5\",\n      \"plane\": \"crown\",\n      \"layer\": \"fg\",\n      \"css\": \"font-size: calc(0.09 * var(--h)); font-weight: 900; text-transform: uppercase; ...\"\n    }\n    // ↑ climax fg: subject too dense for bg to land readably; fg fallback. Sized to fit frame width fully (single line).\n  ]\n}\n```\n\nResult: body bg captures embed cinematic across the subject (eaten edges = texture not bug), climax fg sits on top fully visible because there was nowhere else for it to go. make-composition.cjs emits both `index.html` (bg groups) and `index_fg.html` (fg groups); render-and-composite.sh runs two parallel Chromium passes and ffmpeg screen-blends fg over bg+matte.\n\nSet via `caption_layer: \"bg\" | \"fg\"` plan-level (default for groups without explicit `layer:`), and `layer: \"bg\" | \"fg\"` per-group to override for hybrid.\n\n**Decision rule — bg is default, fg is fallback, hybrid is for mixed semantic content.**\n\nFor BODY captions (narrator + emphasis):\n\n1. **Find the actual clean zone.** Sample frames at 20/50/80% and mentally trace where the subject is (head, shoulders, gesturing hands). The clean zone is what's left. Top band above the head, bottom band below the shoulders, or off-center side column.\n2. **Anchor the plane in that zone.** Body caps fit comfortably with NO subject overlap → `layer: \"bg\"`.\n3. **If body cap stack inherently overlaps subject heavily (>40% per check-occlusion peak)** even at smallest reasonable fonts → body `layer: \"fg\"`. Subject fills frame too much for embed.\n\nFor CLIMAX caption (the payoff):\n\n1. **Default: `layer: \"bg\"`, sized to maximally cross subject for embed effect.** The face/body eating part of the climax word IS the visual win. Don't shrink to avoid occlusion. Use the largest size that fits frame WIDTH (sub-pixel readable letters fully on-screen) — vertical position should land it across the subject's hair / forehead / face.\n2. **`layer: \"fg\"` for climax** ONLY when frame width physically can't hold a font big enough to feel climactic AND there's no compositional way to break the word across multiple lines that bg embed could carry.\n3. **Hybrid (body bg + climax fg)** is the canonical pattern for subject-fills-frame scenes. Body small caps stay bg — subject eats letter edges = embed texture (cinematic, not a bug). Climax goes fg ONLY because at impactful size + subject filling frame there's no place where bg climax could land readably (would be 50%+ eaten = past cinematic into broken). The opposite direction (body fg + climax bg) sacrifices the easy bg embed wins on small caps and forces the climax into a slot it can't survive — never do that.\n\n**Editorial split — the spatial form of body bg + climax fg.** When the hybrid above is laid out with **bg in the upper half and fg in the lower-third**, you get a magazine-cover composition for free — no separate template needed. The rule:\n\n- **Body caps (BG)** — placed in the upper half (top: 6-30%), at the head/hair zone. They get partially occluded by hair/forehead = vogue-masthead embed texture.\n- **Climax cap (FG)** — placed in lower-third (top: 70-85%), in front of subject's chest/below. It's fully clear, fully readable, dominates the bottom band.\n\nThe viewer's eye reads top-to-bottom: first the partially-veiled body line (\"decorative, atmospheric\"), then the clean climax statement (\"the point\"). Same Z-axis layering as Vogue / Apple Keynote (decorative top behind subject + clear bottom in front), but driven by cinematic-cream's existing per-cap `layer:` field — no new template, no separate masthead track.\n\nTwo timing modes inside this layout:\n\n- **Sequential cross-fade (default)** — body fades out as climax fades in over 0.3-0.5s. The brief overlap window IS the editorial dual-layer moment. Standard cinematic-cream behavior.\n- **Sustained overlap** — extend body's `out` to coincide with or extend past climax's `in` so both are visible together for 1-2s. Use sparingly: works when body line is short (≤3 words) and acts as a \"kicker\" above\n\nArchive v1.0.24: 100 files, 1397513 bytes\n\nFiles: assets/fonts/char-widths.json (39965b), assets/strokefonts/HersheyScript1.svg (60060b), assets/strokefonts/HersheyScriptMed.svg (71046b), CATALOG.md (54574b), dna/chrome.json (1769b), dna/cream.json (1537b), dna/documentary.json (1441b), dna/editorial.json (1674b), dna/glitch.json (1805b), dna/ink.json (1529b), dna/keynote.json (1400b), dna/loud.json (1912b), dna/neon.json (1567b), dna/README.md (12207b), dna/velocity.json (1968b), modes/cinematic/_archive/champion/spec.md (4718b), modes/cinematic/_archive/champion/template.html (5784b), modes/cinematic/_archive/memory-wall/spec.md (5878b), modes/cinematic/_archive/memory-wall/template.html (5624b), modes/cinematic/_archive/portrait-header/spec.md (3621b), modes/cinematic/_archive/portrait-header/template.html (5312b), modes/cinematic/cinematic-cream/spec.md (1139b), modes/cinematic/cinematic-cream/template.html (10266b), modes/cinematic/engine.html (17093b), modes/cinematic/README.md (2677b), modes/standard/_anatomy.md (11200b), modes/standard/_motion.md (18518b), modes/standard/fonts/build-fonts-css.cjs (3597b), modes/standard/fonts/fonts.css (1203094b), references/aesthetic-principles.md (7779b), references/anti-patterns.md (12023b), references/bespoke-vs-presets.md (9286b), references/caption-grouping.md (4283b), references/composition-craft.md (55818b), references/direction-catalog.md (7855b), references/example-renders/champion.html (11122b), references/example-renders/memory-wall.html (10707b), references/failure-modes.md (11303b), references/layout-heuristics.md (13202b), references/motion-vocabulary.md (5609b), references/rail.md (4576b), references/reference-bar.md (2689b), references/scene-types.md (6215b), references/typographic-moves.md (8655b), references/typography-presets.md (4714b), scripts/audio-envelope.cjs (3093b), scripts/check-occlusion.cjs (10490b), scripts/check-overflow.cjs (6817b), scripts/check-rail-climax.cjs (8577b), scripts/check-timing.cjs (6354b), scripts/fill-timings.cjs (4587b), scripts/fit-fonts.cjs (6326b), scripts/fixtures/heroless/theme.json (159b), scripts/gen-stroke-path.py (2503b), scripts/hf-cli.cjs (1258b), scripts/inject-fonts.cjs (5901b), scripts/lib-dna.cjs (7410b), scripts/make-cinematic.cjs (56867b), scripts/make-composition.cjs (16750b), scripts/make-theme.cjs (460163b), scripts/make-theme.test.mjs (4623b), scripts/matte.cjs (9125b), scripts/measure-layout.cjs (10839b), scripts/prepare.sh (1881b), scripts/preview-frames.cjs (11515b), scripts/preview-frames.test.mjs (1160b), scripts/render-and-composite.sh (23051b), scripts/render-theme.sh (1786b), scripts/safe-zones.cjs (30395b), scripts/transcribe.cjs (12471b), scripts/transcribe.test.mjs (2847b), skill-card.md (1903b), SKILL.md (34624b), themes/anchor.json (1090b), themes/arcade.json (2031b), themes/aurora.json (1967b), themes/biolume.json (1924b), themes/brush.json (2668b), themes/chalkboard.json (2040b), themes/dossier.json (1905b)\n\nArchive v1.0.23: 100 files, 1397515 bytes\n\nFiles: assets/fonts/char-widths.json (39965b), assets/strokefonts/HersheyScript1.svg (60060b), assets/strokefonts/HersheyScriptMed.svg (71046b), CATALOG.md (54574b), dna/chrome.json (1769b), dna/cream.json (1537b), dna/documentary.json (1441b), dna/editorial.json (1674b), dna/glitch.json (1805b), dna/ink.json (1529b), dna/keynote.json (1400b), dna/loud.json (1912b), dna/neon.json (1567b), dna/README.md (12207b), dna/velocity.json (1968b), modes/cinematic/_archive/champion/spec.md (4718b), modes/cinematic/_archive/champion/template.html (5784b), modes/cinematic/_archive/memory-wall/spec.md (5878b), modes/cinematic/_archive/memory-wall/template.html (5624b), modes/cinematic/_archive/portrait-header/spec.md (3621b), modes/cinematic/_archive/portrait-header/template.html (5312b), modes/cinematic/cinematic-cream/spec.md (1139b), modes/cinematic/cinematic-cream/template.html (10266b), modes/cinematic/engine.html (17093b), modes/cinematic/README.md (2677b), modes/standard/_anatomy.md (11200b), modes/standard/_motion.md (18518b), modes/standard/fonts/build-fonts-css.cjs (3597b), modes/standard/fonts/fonts.css (1203094b), references/aesthetic-principles.md (7779b), references/anti-patterns.md (12023b), references/bespoke-vs-presets.md (9286b), references/caption-grouping.md (4283b), references/composition-craft.md (55818b), references/direction-catalog.md (7855b), references/example-renders/champion.html (11122b), references/example-renders/memory-wall.html (10707b), references/failure-modes.md (11303b), references/layout-heuristics.md (13202b), references/motion-vocabulary.md (5609b), references/rail.md (4576b), references/reference-bar.md (2689b), references/scene-types.md (6215b), references/typographic-moves.md (8655b), references/typography-presets.md (4714b), scripts/audio-envelope.cjs (3093b), scripts/check-occlusion.cjs (10490b), scripts/check-overflow.cjs (6817b), scripts/check-rail-climax.cjs (8577b), scripts/check-timing.cjs (6354b), scripts/fill-timings.cjs (4587b), scripts/fit-fonts.cjs (6326b), scripts/fixtures/heroless/theme.json (159b), scripts/gen-stroke-path.py (2503b), scripts/hf-cli.cjs (1258b), scripts/inject-fonts.cjs (5901b), scripts/lib-dna.cjs (7410b), scripts/make-cinematic.cjs (56867b), scripts/make-composition.cjs (16750b), scripts/make-theme.cjs (460163b), scripts/make-theme.test.mjs (4623b), scripts/matte.cjs (9125b), scripts/measure-layout.cjs (10839b), scripts/prepare.sh (1881b), scripts/preview-frames.cjs (11515b), scripts/preview-frames.test.mjs (1160b), scripts/render-and-composite.sh (23051b), scripts/render-theme.sh (1786b), scripts/safe-zones.cjs (30395b), scripts/transcribe.cjs (12471b), scripts/transcribe.test.mjs (2847b), skill-card.md (1891b), SKILL.md (34639b), themes/anchor.json (1090b), themes/arcade.json (2031b), themes/aurora.json (1967b), themes/biolume.json (1924b), themes/brush.json (2668b), themes/chalkboard.json (2040b), themes/dossier.json (1905b)\n\nArchive v1.0.22: 100 files, 1397518 bytes\n\nFiles: assets/fonts/char-widths.json (39965b), assets/strokefonts/HersheyScript1.svg (60060b), assets/strokefonts/HersheyScriptMed.svg (71046b), CATALOG.md (54574b), dna/chrome.json (1769b), dna/cream.json (1537b), dna/documentary.json (1441b), dna/editorial.json (1674b), dna/glitch.json (1805b), dna/ink.json (1529b), dna/keynote.json (1400b), dna/loud.json (1912b), dna/neon.json (1567b), dna/README.md (12207b), dna/velocity.json (1968b), modes/cinematic/_archive/champion/spec.md (4718b), modes/cinematic/_archive/champion/template.html (5784b), modes/cinematic/_archive/memory-wall/spec.md (5875b), modes/cinematic/_archive/memory-wall/template.html (5624b), modes/cinematic/_archive/portrait-header/spec.md (3621b), modes/cinematic/_archive/portrait-header/template.html (5312b), modes/cinematic/cinematic-cream/spec.md (1139b), modes/cinematic/cinematic-cream/template.html (10266b), modes/cinematic/engine.html (17093b), modes/cinematic/README.md (2677b), modes/standard/_anatomy.md (11200b), modes/standard/_motion.md (18518b), modes/standard/fonts/build-fonts-css.cjs (3597b), modes/standard/fonts/fonts.css (1203094b), references/aesthetic-principles.md (7779b), references/anti-patterns.md (12023b), references/bespoke-vs-presets.md (9286b), references/caption-grouping.md (4283b), references/composition-craft.md (55818b), references/direction-catalog.md (7855b), references/example-renders/champion.html (11122b), references/example-renders/memory-wall.html (10707b), references/failure-modes.md (11303b), references/layout-heuristics.md (13202b), references/motion-vocabulary.md (5609b), references/rail.md (4576b), references/reference-bar.md (2689b), references/scene-types.md (6215b), references/typographic-moves.md (8655b), references/typography-presets.md (4714b), scripts/audio-envelope.cjs (3093b), scripts/check-occlusion.cjs (10490b), scripts/check-overflow.cjs (6817b), scripts/check-rail-climax.cjs (8577b), scripts/check-timing.cjs (6354b), scripts/fill-timings.cjs (4587b), scripts/fit-fonts.cjs (6326b), scripts/fixtures/heroless/theme.json (159b), scripts/gen-stroke-path.py (2503b), scripts/hf-cli.cjs (1258b), scripts/inject-fonts.cjs (5901b), scripts/lib-dna.cjs (7410b), scripts/make-cinematic.cjs (56867b), scripts/make-composition.cjs (16750b), scripts/make-theme.cjs (460163b), scripts/make-theme.test.mjs (4623b), scripts/matte.cjs (9125b), scripts/measure-layout.cjs (10839b), scripts/prepare.sh (1881b), scripts/preview-frames.cjs (11515b), scripts/preview-frames.test.mjs (1160b), scripts/render-and-composite.sh (23051b), scripts/render-theme.sh (1786b), scripts/safe-zones.cjs (30395b), scripts/transcribe.cjs (12471b), scripts/transcribe.test.mjs (2847b), skill-card.md (2062b), SKILL.md (34639b), themes/anchor.json (1090b), themes/arcade.json (2031b), themes/aurora.json (1967b), themes/biolume.json (1924b), themes/brush.json (2668b), themes/chalkboard.json (2040b), themes/dossier.json (1905b)\n\nArchive v1.0.21: 100 files, 1397548 bytes\n\nFiles: assets/fonts/char-widths.json (39965b), assets/strokefonts/HersheyScript1.svg (60060b), assets/strokefonts/HersheyScriptMed.svg (71046b), CATALOG.md (54574b), dna/chrome.json (1769b), dna/cream.json (1537b), dna/documentary.json (1441b), dna/editorial.json (1674b), dna/glitch.json (1805b), dna/ink.json (1529b), dna/keynote.json (1400b), dna/loud.json (1912b), dna/neon.json (1567b), dna/README.md (12207b), dna/velocity.json (1968b), modes/cinematic/_archive/champion/spec.md (4718b), modes/cinematic/_archive/champion/template.html (5784b), modes/cinematic/_archive/memory-wall/spec.md (5875b), modes/cinematic/_archive/memory-wall/template.html (5624b), modes/cinematic/_archive/portrait-header/spec.md (3621b), modes/cinematic/_archive/portrait-header/template.html (5312b), modes/cinematic/cinematic-cream/spec.md (1139b), modes/cinematic/cinematic-cream/template.html (10266b), modes/cinematic/engine.html (17093b), modes/cinematic/README.md (2677b), modes/standard/_anatomy.md (11132b), modes/standard/_motion.md (18518b), modes/standard/fonts/build-fonts-css.cjs (3597b), modes/standard/fonts/fonts.css (1203094b), references/aesthetic-principles.md (7779b), references/anti-patterns.md (12023b), references/bespoke-vs-presets.md (9286b), references/caption-grouping.md (4283b), references/composition-craft.md (55818b), references/direction-catalog.md (7855b), references/example-renders/champion.html (11122b), references/example-renders/memory-wall.html (10707b), references/failure-modes.md (11303b), references/layout-heuristics.md (13202b), references/motion-vocabulary.md (5609b), references/rail.md (4576b), references/reference-bar.md (2689b), references/scene-types.md (6215b), references/typographic-moves.md (8655b), references/typography-presets.md (4714b), scripts/audio-envelope.cjs (3093b), scripts/check-occlusion.cjs (10490b), scripts/check-overflow.cjs (6817b), scripts/check-rail-climax.cjs (8577b), scripts/check-timing.cjs (6354b), scripts/fill-timings.cjs (4587b), scripts/fit-fonts.cjs (6326b), scripts/fixtures/heroless/theme.json (159b), scripts/gen-stroke-path.py (2503b), scripts/hf-cli.cjs (1258b), scripts/inject-fonts.cjs (5901b), scripts/lib-dna.cjs (7410b), scripts/make-cinematic.cjs (56867b), scripts/make-composition.cjs (16750b), scripts/make-theme.cjs (460163b), scripts/make-theme.test.mjs (4623b), scripts/matte.cjs (9125b), scripts/measure-layout.cjs (10839b), scripts/prepare.sh (1881b), scripts/preview-frames.cjs (11515b), scripts/preview-frames.test.mjs (1160b), scripts/render-and-composite.sh (23051b), scripts/render-theme.sh (1786b), scripts/safe-zones.cjs (30395b), scripts/transcribe.cjs (12471b), scripts/transcribe.test.mjs (2847b), skill-card.md (2009b), SKILL.md (34639b), themes/anchor.json (1090b), themes/arcade.json (2031b), themes/aurora.json (1967b), themes/biolume.json (1924b), themes/brush.json (2668b), themes/chalkboard.json (2040b), themes/dossier.json (1905b)\n\nArchive v1.0.20: 100 files, 1397407 bytes\n\nFiles: assets/fonts/char-widths.json (39965b), assets/strokefonts/HersheyScript1.svg (60060b), assets/strokefonts/HersheyScriptMed.svg (71046b), CATALOG.md (54574b), dna/chrome.json (1769b), dna/cream.json (1537b), dna/documentary.json (1441b), dna/editorial.json (1674b), dna/glitch.json (1805b), dna/ink.json (1529b), dna/keynote.json (1400b), dna/loud.json (1912b), dna/neon.json (1567b), dna/README.md (12207b), dna/velocity.json (1968b), modes/cinematic/_archive/champion/spec.md (4718b), modes/cinematic/_archive/champion/template.html (5784b), modes/cinematic/_archive/memory-wall/spec.md (5875b), modes/cinematic/_archive/memory-wall/template.html (5624b), modes/cinematic/_archive/portrait-header/spec.md (3621b), modes/cinematic/_archive/portrait-header/template.html (5312b), modes/cinematic/cinematic-cream/spec.md (1139b), modes/cinematic/cinematic-cream/template.html (10444b), modes/cinematic/engine.html (17288b), modes/cinematic/README.md (2677b), modes/standard/_anatomy.md (11132b), modes/standard/_motion.md (18518b), modes/standard/fonts/build-fonts-css.cjs (3597b), modes/standard/fonts/fonts.css (1203094b), references/aesthetic-principles.md (7779b), references/anti-patterns.md (12023b), references/bespoke-vs-presets.md (9286b), references/caption-grouping.md (4283b), references/composition-craft.md (55818b), references/direction-catalog.md (7855b), references/example-renders/champion.html (11122b), references/example-renders/memory-wall.html (10707b), references/failure-modes.md (11303b), references/layout-heuristics.md (13202b), references/motion-vocabulary.md (5609b), references/rail.md (4576b), references/reference-bar.md (2689b), references/scene-types.md (6215b), references/typographic-moves.md (8655b), references/typography-presets.md (4714b), scripts/audio-envelope.cjs (3093b), scripts/check-occlusion.cjs (10490b), scripts/check-overflow.cjs (6817b), scripts/check-rail-climax.cjs (8577b), scripts/check-timing.cjs (6354b), scripts/fill-timings.cjs (4587b), scripts/fit-fonts.cjs (6326b), scripts/fixtures/heroless/theme.json (159b), scrip...","readmeExcerpt":"Skill: embedded-captions Owner: heygen-com Summary: Add captions or subtitles to an existing single-subject talking-head video without editing the footage. Use for plain verbatim captions, cinematic captions embedded behind the subject, VFX captions, “炸/特效/酷炫字幕,” or a named identity from the 35-style catalog. Route by visual identity, not by backend engine. The quiet anchor rail is the default; embed every word only ","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"npm install --prefix <project> --save-dev --save-exact sharp@0.35.3 puppeteer@25.8.0 gsap@3.15.0"},{"language":"bash","snippet":"ffprobe <video.mp4>                    # specs\nffmpeg -ss <t> -i <video.mp4> -vframes 1 sample.png   # at 20/50/80%"},{"language":"text","snippet":"1. hyperframes init <project> --non-interactive --video <video.mp4> --skill=embedded-captions\n2. bash scripts/prepare.sh <project>       # matte ∥ transcribe (parallel) → safe-zones. One command.\n                                           #   → frames_fg/ transcript.json safe-zones.json\n3. [AGENT STEP — the only creative step] author a small JSON; see below by mode\n   Cinematic: author cinematic.json → node scripts/make-cinematic.cjs <project>\n   Theme:     author theme.json → bash scripts/render-theme.sh <project>   (compiles + renders + plate fx)\n4. node scripts/preview-frames.cjs <project>   # ~2s/frame composite previews → § Visual QA (BEFORE the render)\n5. bash scripts/render-and-composite.sh <project>  # gates → final.mp4 + history/ snapshot\n   (Theme mode: SKIP steps 3b/5 — render-theme.sh already runs compile + render-and-composite\n    + _postfx.sh; the deliverable is final_fx.mp4, final.mp4 is pre-plate-reaction)"},{"language":"text","snippet":"theme DNA = body PARADIGM   how the transcript surface lives\n          × body LAYER      fg-alpha (rail.html channel) | bg-embed\n          × hero SETPIECE   the climax choreography\n          × front FX        flash / rings / sparks / scanband / crowdflash / paflash (fg, over subject)\n          × PLATE budget    charge-dim (in-page) + punch/shake/grain (_postfx.sh)\n          × LINKAGES        declarative theme interactions"},{"language":"json","snippet":"\"custom_css\": \".cap-1 { font-size: 78px; ... } .cap-2 { padding-right: 44px; }\",\n\"groups\": [{\"id\": \"cg-0\", \"style\": \"1\", ...}]"},{"language":"json","snippet":"{\n  \"template\": \"wall-embed\",\n  \"custom_css\": \"\n    .cap-1 { font-size: 78px; font-weight: 600; font-style: italic;\n             letter-spacing: -0.01em; }\n    .cap-2 { font-size: 66px; font-weight: 500; font-style: italic;\n             padding-right: 44px; }\n    .cap-3 { font-size: 72px; font-weight: 700; letter-spacing: -0.015em; }\n    .cap-4 { font-size: 90px; font-weight: 900; letter-spacing: -0.03em;\n             text-transform: uppercase; line-height: 1.0; }\n  \",\n  \"groups\": [\n    { \"id\": \"cg-0\", \"style\": \"1\", \"words\": [...] },\n    { \"id\": \"cg-1\", \"style\": \"2\", \"words\": [...] },\n    ...\n  ]\n}"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: embedded-captions\ndescription: >\n  Add captions or subtitles to an existing single-subject talking-head video without editing the\n  footage. Use for plain verbatim captions, cinematic captions embedded behind the subject, VFX\n  captions, “炸/特效/酷炫字幕,” or a named identity from the 35-style catalog. Route by visual\n  identity, not by backend engine. The quiet `anchor` rail is the default; embed every word only\n  when the user explicitly wants a fully cinematic treatment. The workflow runs locally end to end,\n  including transcription and subject matting; split multi-shot footage before applying it.\n---\n\n**Plugin installs:** Before setup or freshness commands, follow [plugin execution rules](../hyperframes/references/plugin-installation.md) when this skill is inside a HyperFrames plugin. Standalone installs keep the update instructions below.\n\n> **First, keep this skill fresh — confirm with the user before running:** `npx hyperframes skills update embedded-captions`. A fast no-op when everything is current; otherwise it refreshes this skill plus the core domain skills it depends on before you rely on them.\n\n# Embedded Captions\n\n**One catalog, picked up front** ([CATALOG.md](CATALOG.md) — 35 identities; the engines behind it are backend detail). **Standard** (default) builds a clean verbatim **rail** (lower-third subtitle carrying most text) + an **embed** climax composited _into_ the scene behind the subject at the peak. **Cinematic** is pure embed — no rail, every caption composited behind the subject (hero typography, accumulation, occlusion as the effect). **Theme** is a complete themed constitution — body paradigm × hero setpiece × front fx × plate reaction, composed from registries ([themes/README.md](themes/README.md)): `ordnance` `terminal` `neonsign` `stardust` `stomp`. Most explainer / voiceover is **Standard**; **embed is the scarce, earned peak** — embedding every word is the common mistake; Theme is for VFX-grade asks (\"炸\", \"特效\", \"像 AE 做的\").\n\n---\n\n## Runtime prerequisites\n\nPlugin installs use the bundled, manifest-pinned CLI for matting, transcription,\nand rendering; no source checkout is required. The local preview and caption\nmeasurement helpers also need Sharp, Puppeteer (with its Chromium browser), and\nGSAP. Install these in the **caption project**, not inside the read-only plugin:\n\n```bash\nnpm install --prefix <project> --save-dev --save-exact sharp@0.35.3 puppeteer@25.8.0 gsap@3.15.0\n```\n\nKeep the project's lockfile. If these dependencies already exist, use its locked\nversions instead of overwriting them. Bash and FFmpeg/ffprobe must be on PATH.\nMatting and transcription may download their own models on first use.\n\nRendering waits for the CLI to exit successfully before compositing. The old\n`HF_TIMEOUT_S` shell watchdog is no longer used: a large partial file is not proof\nthat rendering finished. An explicit built-checkout argument or `HYPERFRAMES_ROOT`\nselects the contributor CLI instead of the plugin pin. Cancel a stalle"},{"path":"dna/README.md","content":"# DNA registry — pick a visual language, not a preset\n\nA **DNA** is a complete, art-directed visual language: typeface, palette logic, motion\ngrammar, and hero orchestration. It **parameterizes per scene** instead of shipping a\nfixed look: the accent color is sampled from THIS scene, the contact shadow falls along\nTHIS scene's light, embed text blur matches THIS scene's depth-of-field, and the hero's\nentrance amplitude follows how hard the word was actually spoken (RMS).\n\nThis replaces the template grab-bag. Six deep languages × scene adaptation beats 54\nshallow presets — every render is already fitted to its footage.\n\n## Category lock (deliveries field, enforced by the compilers)\n\nEvery classic DNA's **home is Cinematic (column)** — that is where all ten were\nbuilt and validated. (Standard/rail mode was retired 2026-06-12; the verbatim-rail\nneed is served by the `anchor` theme. The old rail combos are archived outside\nthis repo and are not distributed with the skill.)\n\n## The ten\n\n| DNA             | Register       | Scene fit                                       | Voice                                                                                                                                                                       |\n| --------------- | -------------- | ----------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |\n| **cream**       | premium-warm   | dark / mid warm scenes (band luma < 150)        | Inter, warm cream, screen blend, glowing emergence hero. The poetic default.                                                                                                |\n| **ink**         | premium        | **bright scenes (band luma > 150)**             | Inter, near-black, multiply blend — type reads as _printed on_ the wall. Fixes the bright-scene hole.                                                                       |\n| **editorial**   | editorial-luxe | introspective / fashion / poetic, mid-dark      | Bodoni Moda, bone, _lowercase italic hero_ — magazine elegance over shout.                                                                                                  |\n| **keynote**     | tech-premium   | product / launch / founder updates              | Inter 800, opaque white, line-wipe reveals, hero wipes UP. Stillness = confidence.                                                                                          |\n| **documentary** | formal         | interviews, serious subject matter              | Inter, bone, **burn-in reveals**, no hero. Gravitas IS the style.                                                                                                           |\n| **loud**        | loud           | hype / sport / music / social                   | Anton, scene-sampled accent hero, single-unit slam + caption-layer ripple; **body announc"},{"path":"modes/cinematic/README.md","content":"# Cinematic mode (pure embed) — one engine, six DNAs\n\n> Cinematic mode compiles **[../../dna/](../../dna/README.md)** through\n> **[engine.html](engine.html)** (`make-composition.cjs`). The old per-template HTML\n> shells are retired — `cinematic-cream` maps to `dna: \"cream\"` automatically; the other\n> archived templates (memory-wall / champion / portrait-header, in [\\_archive/](_archive/))\n> remain as design references only.\n\nUse this mode for pure-embed asks (no rail): brand film, hype, social reel, showcase.\nThe **DNA** locks the visual language (type, palette scheme, blend, motion grammar, hero\nthree-act); **safe-zones v2** parameterizes it to the scene (sampled accent, light-\ndirection contact shadow, depth-match blur); **the agent decides layout only** (planes,\nblocks, per-line typography within the DNA).\n\n## Workflow\n\n1. `bash scripts/prepare.sh <project>` → matte ∥ transcript ∥ envelope → safe-zones v2\n2. Pick a DNA ([../../dna/README.md](../../dna/README.md)): bright hero band → `ink`,\n   else by register (cream / editorial / keynote / documentary / loud). Recommend, let\n   the user pick.\n3. Author `<project>/cinematic.json` — `\"dna\": \"<name>\"` + thought-blocks (schema:\n   `scripts/make-cinematic.cjs` header)\n4. `node scripts/make-cinematic.cjs <project>` → plan.json → engine-compiled index.html\n5. `node scripts/preview-frames.cjs <project>` → § Visual QA (failure checks + the 5\n   positive checks in [../../references/reference-bar.md](../../references/reference-bar.md))\n6. `bash scripts/render-and-composite.sh <project>` → gates → final.mp4\n\n## What the engine generates (never author these)\n\n- word timings from the transcript; accumulate-within-block / page-flip-between-blocks\n- the hero hand-off + **three-act orchestration** (dim → RMS-coupled per-letter entrance\n  → breathe + glow), per the DNA's `hero` block\n- scene tokens: `--accent` (sampled), contact shadow, depth blur\n- reading order, re-slot from measured heights, hero size/collision post-pass\n\n## What you DON'T do\n\n- Override `.cap` color / blend / shadow / filter / motion curves — that's the DNA.\n  Scene fights the look → pick a different DNA (bright → `ink`), never recolor.\n- Hand-position the hero into a clean margin (it belongs ON the subject, ~30–55%\n  occluded — safe-zones `heroBands.best`).\n- Add full-frame grades/textures over the footage (hard rule: the video ships untouched).\n\n## Adding a DNA\n\n`dna/<name>.json` — copy one, change the voice (see [../../dna/README.md](../../dna/README.md)\n§ Adding). The engine consumes it with no code change. A DNA must be a distinct voice\nwith a reason to exist, not a recolor."},{"path":"themes/README.md","content":"# THEME mode — composed visual constitutions\n\nTheme mode is the third compiler (`scripts/make-theme.cjs`), beside Standard and\nCinematic. It exists because \"mode\" was a bundle of orthogonal axes pretending\nto be one switch. A theme DNA composes its identity from registries implemented\nONCE in the compiler — **paradigms are the unit of code; DNAs are the unit of\nidentity**. A new look is a JSON file; only a genuinely new paradigm/setpiece\n(rare) touches the compiler.\n\n```\ntheme DNA = body PARADIGM   how the transcript surface lives\n          × body LAYER      fg-alpha (rail.html channel) | bg-embed\n          × hero SETPIECE   the climax choreography\n          × front FX        flash / rings / sparks / scanband / crowdflash / paflash (fg, over subject)\n          × PLATE budget    charge-dim (in-page) + punch/shake/grain (_postfx.sh)\n          × LINKAGES        declarative theme interactions\n```\n\nStandard and Cinematic were, in retrospect, two fixed points of this space.\nStandard is now RETIRED (2026-06-12): its rail×embed-climax point is served by\nthe `anchor` theme (rail paradigm × settle setpiece — the quiet default).\nCinematic remains a separate compiler; do not re-implement it as a theme yet.\n\n## Unification roadmap (strangler fig — interface first, engines later)\n\nThe user-facing model is already unified (SKILL.md Step 0): one catalog of\nLOOKS; classic looks pick a DELIVERY (rail | column), themed looks bind their\nown. \"Standard/Cinematic\" are delivery/compiler names, not modes. Remaining\nphases, each gated on need — never rewrite for tidiness alone:\n\n- **Phase 2 — one authoring schema.** `lines`/`minors`/`hero` are already\n  ~90% shared between standard.json and theme.json; a router that translates a\n  single `caption.json` into the engine-specific file removes the last\n  user-visible seam. cinematic.json's blocks/planes are the odd one — map the\n  common fields, pass engine-specific ones through.\n- **Phase 3 — engine convergence.** Port a classic delivery into make-theme\n  ONLY when something forces it (e.g. a classic DNA wants a plate budget or a\n  setpiece). Acceptance bar: blind A/B on the cap_multi regression scenes vs\n  the old compiler — swap engines only when indistinguishable or better. Until\n  then the old compilers are the reference implementation of 8 rounds of\n  validated typography (lockup/orbit, multi-climax, ratio-lock, per-plane\n  legibility, occlusion adjudication) — that machinery is the moat, not debt.\n\n## Body paradigms (registry)\n\n| paradigm      | surface                                                                                                                                                                                                                                                                                                                                                                                                                                                                                          "},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn77d06grj6xqp3dqwkk4bavhn89pegt\",\n  \"slug\": \"embedded-captions\",\n  \"version\": \"1.0.25\",\n  \"publishedAt\": 1791142135378\n}"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Add captions or subtitles to an existing single-subject talking-head video without editing the footage. Use for plain verbatim captions, cinematic captions embedded behind the subject, VFX captions, “炸/特效/酷炫字幕,” or a named identity from the 35-style catalog. Route by visual identity, not by backend engine. The quiet `anchor` rail is the default; embed every word only when the user explicitly wants a fully cinematic treatment. The workflow runs locally end to end, including transcription and subject matting; split multi-shot footage before applying it. Skill: embedded-captions Owner: heygen-com Summary: Add captions or subtitles to an existing single-subject talking-head video without editing the footage. Use for plain verbatim captions, cinematic captions embedded behind the subject, VFX captions, “炸/特效/酷炫字幕,” or a named identity from the 35-style catalog. Route by visual identity, not by backend engine. The quiet anchor rail is the default; embed every word only","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1831,"uniquenessScore":50,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T16:46:35.121Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T16:46:35.121Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T06:43:38.102Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}