{"id":"281e00f1-13a5-4ddb-97e4-5948ed313205","entityType":"agent","slug":"clawhub-tobewin-make-motion-comic","name":"动态漫画制作","canonicalUrl":"https://www.xpersona.co/agent/clawhub-tobewin-make-motion-comic","canonicalPath":"/agent/clawhub-tobewin-make-motion-comic","generatedAt":"2026-10-11T15:15:29.204Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-11T12:58:34.857Z","emptyReason":null},"description":"从剧本、统一角色图到 Edge TTS 与防抖竖屏成片 Skill: 动态漫画制作 Owner: tobewin Summary: 从剧本、统一角色图到 Edge TTS 与防抖竖屏成片 Tags: latest:0.1.0 Version history: v0.1.0 | 2026-08-26T03:06:00.241Z | auto Initial release of the make-motion-comic skill. - Create low-cost motion-comic videos from story/script using AI-generated keyframes, multi-character Chinese TTS, captions, and FFmpeg assembly. - Ensures character identity consistency, professional voice quality, smooth motion","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.1K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s174rz3f0862tcfw7pzfh5w8kn83hv2z:make-motion-comic","sourceUrl":"https://clawhub.ai/tobewin/make-motion-comic","homepage":"https://clawhub.ai/tobewin/skills/make-motion-comic","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/tobewin/make-motion-comic","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/tobewin/skills/make-motion-comic","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":61,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"从剧本、统一角色图到 Edge TTS 与防抖竖屏成片 Skill: 动态漫画制作 Owner: tobewin Summary: 从剧本、统一角色图到 Edge TTS 与防抖竖屏成片 Tags: latest:0.1.0 Version history: v0.1.0 | 2026-08-26T03:06:00.2"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T12:58:34.857Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T12:58:34.857Z","emptyReason":null},"stars":null,"forks":null,"downloads":1061,"packageName":null,"latestVersion":"0.1.0","tractionLabel":"1.1K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T12:58:34.781Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T12:58:34.857Z","lastCrawledAt":"2026-10-11T12:58:34.781Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T12:58:34.781Z","lastVerifiedAt":null,"highlights":[{"version":"0.1.0","createdAt":"2026-08-26T03:06:00.241Z","changelog":"Initial release of the make-motion-comic skill. - Create low-cost motion-comic videos from story/script using AI-generated keyframes, multi-character Chinese TTS, captions, and FFmpeg assembly. - Ensures character identity consistency, professional voice quality, smooth motion, and independently editable assets. - Provides a structured workflow: script/shot writing, identity locking, keyframe generation, TTS voice production, precise timeline/timing, and final video rendering. - Default video format is 9:16, 1080×1920, 30 fps, 45–90s, with 6–12 keyframes and post-rendered captions. - Includes quality checks for image consistency, hand anatomy, subtitle safety, audio balance, and deliberate/non-jittery motion. - Requires dependency check before media production, and supports reviewable, reusable production assets for episodic work.","fileCount":23,"zipByteSize":95057}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s174rz3f0862tcfw7pzfh5w8kn83hv2z:make-motion-comic","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tobewin-make-motion-comic/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tobewin-make-motion-comic/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tobewin-make-motion-comic/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-tobewin-make-motion-comic/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-tobewin-make-motion-comic/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-tobewin-make-motion-comic/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T15:15:29.203Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tobewin-make-motion-comic/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tobewin-make-motion-comic/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tobewin-make-motion-comic/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tobewin-make-motion-comic/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-11T12:58:34.857Z","emptyReason":null},"readme":"Skill: 动态漫画制作\n\nOwner: tobewin\n\nSummary: 从剧本、统一角色图到 Edge TTS 与防抖竖屏成片\n\nTags: latest:0.1.0\n\nVersion history:\n\nv0.1.0 | 2026-08-26T03:06:00.241Z | auto\n\nInitial release of the make-motion-comic skill.\n\n- Create low-cost motion-comic videos from story/script using AI-generated keyframes, multi-character Chinese TTS, captions, and FFmpeg assembly.\n- Ensures character identity consistency, professional voice quality, smooth motion, and independently editable assets.\n- Provides a structured workflow: script/shot writing, identity locking, keyframe generation, TTS voice production, precise timeline/timing, and final video rendering.\n- Default video format is 9:16, 1080×1920, 30 fps, 45–90s, with 6–12 keyframes and post-rendered captions.\n- Includes quality checks for image consistency, hand anatomy, subtitle safety, audio balance, and deliberate/non-jittery motion.\n- Requires dependency check before media production, and supports reviewable, reusable production assets for episodic work.\n\nArchive index:\n\nArchive v0.1.0: 23 files, 95057 bytes\n\nFiles: .gitignore (55b), agents (0b), agents/openai.yaml (231b), assets (0b), assets/templates (0b), assets/templates/tts-script.json (277b), docs (0b), docs/demo-cover.jpg (72573b), LICENSE (1064b), README.md (9390b), references (0b), references/audio-and-tts.md (2369b), references/image-consistency.md (1962b), references/motion-and-qc.md (2265b), references/story-and-shots.md (1881b), scripts (0b), scripts/build_timeline.py (4372b), scripts/preflight.sh (996b), scripts/render_still_clip.sh (1874b), scripts/synthesize_edge_tts.py (3552b), skill-card.md (2653b), SKILL.md (8425b), _meta.json (136b)\n\nFile v0.1.0:SKILL.md\n\n---\nname: make-motion-comic\ndescription: Create or revise low-cost motion-comic videos from a story or script using consistent AI-generated keyframes, multi-character Chinese Edge TTS, captions, synthesized or licensed audio, and FFmpeg assembly. Use for 动态漫画、漫剧、条漫视频、animated manga/comic, narrated image-story shorts, vertical story videos, or when Codex must turn generated still images into a polished video without a generative video model; also use to diagnose or fix character drift, robotic TTS, subtitle timing, micro-jitter, shaky zoompan motion, audio balance, covers, and reusable episode production assets.\n---\n\n# Make Motion Comic\n\nProduce a complete motion-comic episode from a script while keeping image identity, voice quality, motion smoothness, and source assets independently editable.\n\n## Required route\n\n1. Use the built-in Image Generator through the available `imagegen` skill for character sheets, keyframes, image corrections, and covers. Follow that skill's reference-image and save-path rules.\n2. Use Edge TTS for Chinese production voice by default. Do not use macOS `say` for a final deliverable unless the user explicitly chooses its offline quality tradeoff.\n3. Use FFmpeg for deterministic motion, audio mixing, captions, encoding, and inspection. A user may explicitly choose another video framework.\n4. Run `scripts/preflight.sh` before producing media. Report missing required dependencies before continuing.\n\nEdge TTS uses an unofficial client for Microsoft's online speech endpoint. It is free in normal use but needs network access and has no service guarantee. Retry transient failures; do not silently substitute a worse voice.\n\n## Default production brief\n\nUse these defaults when the user says “开始”“直接做” or otherwise authorizes an autonomous first pass:\n\n- Format: 9:16, 1080×1920, 30 fps, H.264/AAC.\n- Length: 45–90 seconds.\n- Visuals: 6–12 keyframes; use a new image when story state changes, not at arbitrary time intervals.\n- Structure: hook in 0–3 seconds, rule or dilemma, escalation, emotional reversal, final serial cliffhanger.\n- Captions: render in post; never ask the image model to typeset dialogue.\n- Audio: multi-character Neural voices, light ambience/SFX, no unverified copyrighted BGM.\n- Review: inspect the character sheet, raw keyframe contact sheet, and final-video snapshots.\n\nRead [references/story-and-shots.md](references/story-and-shots.md) before writing a new episode. Read [references/image-consistency.md](references/image-consistency.md) before generating images.\n\n## Workflow\n\n### 1. Establish the production package\n\nCreate a project-local working folder and a user-facing output folder. Preserve:\n\n- script and shot table;\n- character/world visual bible;\n- one prompt per keyframe;\n- raw keyframes;\n- TTS manifest and voice-only mix;\n- subtitle file and timed timeline;\n- final mix, cover, contact sheet, and final video.\n\nKeep temporary render fragments outside the user-facing output folder.\n\n### 2. Write for motion comics\n\nWrite narration and dialogue before generating images. Assign every spoken line to a shot. Prefer two or three recurring characters and a small number of reusable locations. Use visual reveals, poses, props, lighting changes, and cuts rather than animation-dependent action.\n\nUse the structure and duration guidance in [references/story-and-shots.md](references/story-and-shots.md).\n\n### 3. Lock identity before keyframes\n\nGenerate one production visual bible with neutral full-body views, face closeups, signature clothing/accessories, and a world inset. Treat it as the identity source of truth.\n\nFor every keyframe:\n\n- reference the original visual bible;\n- optionally reference the previous shot only for pose or continuity;\n- explicitly preserve face, hair, clothing, proportions, and signature props;\n- label every reference image's role in the prompt;\n- request no generated captions, speech bubbles, logos, or watermarks.\n\nDo not build a pure A→B→C reference chain. It accumulates identity drift. Follow [references/image-consistency.md](references/image-consistency.md).\n\n### 4. Generate and inspect keyframes\n\nGenerate each distinct shot with a separate built-in image call. Save final selected images into the project. Inspect full-size images and a contact sheet for:\n\n- character identity and clothing;\n- hand anatomy and person count;\n- prop continuity;\n- color and lighting continuity;\n- focal area and subtitle-safe space;\n- accidental text or watermarks.\n\nRegenerate a failed shot with one targeted correction. Never continue from a visibly drifted reference.\n\n### 5. Produce voices before final timing\n\nCreate a JSON TTS manifest from `assets/templates/tts-script.json`. Give recurring characters stable voices and stable rate/pitch settings. Use `scripts/synthesize_edge_tts.py` to generate one file per line with retries and resumability.\n\nFor Mandarin, start with:\n\n- `zh-CN-XiaoxiaoNeural`: warm narrator;\n- `zh-CN-XiaoyiNeural`: adult woman or a lightly raised-pitch child;\n- `zh-CN-YunyangNeural`: controlled or authoritative man;\n- `zh-CN-YunxiNeural`: younger, urgent man.\n\nPunctuation controls acting. Use commas and ellipses sparingly; excessive ellipses make an episode drag. Read [references/audio-and-tts.md](references/audio-and-tts.md) before casting voices or mixing.\n\n### 6. Build the real timeline\n\nMeasure generated voice files with `ffprobe`; voice duration overrides estimates. Build subtitle and shot timing from those durations plus deliberate pauses. Retiming must update all of:\n\n- shot boundaries;\n- subtitle in/out times;\n- SFX placement;\n- final card timing.\n\nNever hardcode subtitle timing from the draft script.\n\nUse `scripts/build_timeline.py` to create `voice.wav`, `subtitles.srt`, and `timeline.json` from the same TTS manifest:\n\n```bash\npython3 scripts/build_timeline.py \\\n  --manifest tts-script.json \\\n  --audio-dir audio/lines \\\n  --out timeline\n```\n\n### 7. Render deliberate, non-jittery motion\n\nUse motion only when it supports attention or emotion:\n\n- push in for realization or threat;\n- pull out for isolation or consequence;\n- pan to reveal information;\n- hold still for shock, grief, or the final reveal.\n\nDo not add camera shake as a generic “dynamic” effect. Do not run low-resolution `zoompan` directly at delivery size.\n\nRender still-image motion with `scripts/render_still_clip.sh`. It uses an oversized working canvas, eased movement, Lanczos downsampling, and 30 fps defaults to prevent integer-coordinate stepping and line-art shimmer. Read [references/motion-and-qc.md](references/motion-and-qc.md) before implementing motion or diagnosing jitter.\n\n### 8. Mix captions and audio\n\nKeep captions separate from generated art. Use a stable bottom safe area, high contrast, and at most two short lines. Do not cover faces or required props.\n\nMix voice first, then ambience, transitions, and music. Duck beds under dialogue. Target approximately:\n\n- integrated loudness: −16 LUFS for social video;\n- true peak: no higher than −1.5 dBTP;\n- clear dialogue at phone-speaker volume.\n\nUse synthesized ambience/SFX or media with explicit reusable licensing. Record provenance for downloaded audio.\n\n### 9. Verify before handoff\n\nRun all applicable checks:\n\n```bash\nffprobe -v error -show_entries format=duration:stream=codec_name,width,height,r_frame_rate,sample_rate,channels -of json final.mp4\nffmpeg -v error -i final.mp4 -f null -\nffmpeg -hide_banner -i final.mp4 -vf blackdetect=d=0.3:pix_th=0.05 -an -f null -\nffmpeg -hide_banner -i final.mp4 -af loudnorm=I=-16:LRA=9:TP=-1.5:print_format=summary -f null -\n```\n\nExtract and inspect snapshots from the hook, each major reversal, the last spoken line, and the final card. Confirm the last subtitle does not overlap the final card.\n\nDeliver the final video plus cover, script/shot list, subtitles, character bible, prompt set, keyframes/contact sheet, and reusable audio mix.\n\n## Quality gates\n\nDo not call the episode complete when any of these remain:\n\n- system-quality or novelty TTS in a production track;\n- character identity drift across keyframes;\n- generated Chinese dialogue inside images;\n- arbitrary camera motion on every frame;\n- visible one-pixel stepping, shimmer, or unintended shake;\n- subtitle/face overlap or subtitle/final-card overlap;\n- clipped audio, unverified copyrighted music, missing codec/audio stream, or black gaps.\n\nFile v0.1.0:README.md\n\n# Make Motion Comic\n\n用 AI 图片、Edge TTS 与 FFmpeg 制作低成本、高一致性、可维护的动态漫画。\n\n[![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE)\n[![Codex Skill](https://img.shields.io/badge/Codex-Skill-111827)](SKILL.md)\n[![FFmpeg](https://img.shields.io/badge/FFmpeg-Required-007808)](https://ffmpeg.org/)\n[![Edge TTS](https://img.shields.io/badge/Edge%20TTS-Neural%20Voice-2563EB)](https://github.com/rany2/edge-tts)\n\n`make-motion-comic` 是一个面向 Codex 的动态漫画制作 Skill。它把“剧本 → 角色视觉基准 → 逐镜图片 → 多角色配音 → 防抖镜头 → 字幕与混音 → 成片质检”固化为可复用的生产流程，不依赖生成式视频模型。\n\n它适合：\n\n- 动态漫画、漫剧、条漫视频；\n- 竖屏悬疑、情感、科普和连续短剧；\n- 用少量关键帧制作有声音、有节奏的故事视频；\n- 修复角色漂移、机器人配音、字幕错位和 `zoompan` 微抖动。\n\n> 当前版本重点支持普通话、9:16 竖屏和 FFmpeg 工作流。\n\n![动态漫画效果预览](docs/demo-cover.jpg)\n\n## 为什么做这个 Skill\n\n图片生成模型已经能够产出质量很高的漫画关键帧，但将图片直接拼成视频通常会出现四类问题：\n\n1. 人物在镜头之间变脸、换衣服或改变年龄；\n2. 系统 TTS 缺乏情绪，破坏画面建立的氛围；\n3. 低分辨率 `zoompan` 产生一像素跳动和细线闪烁；\n4. 配音、字幕、镜头和片尾各自计时，返工后全部错位。\n\n本项目为这些问题提供明确的生产规则和可执行脚本：\n\n- 始终使用原始角色视觉基准作为身份锚点；\n- 默认使用多角色中文 Neural Edge TTS；\n- 根据真实语音文件重建时间轴；\n- 在 2 倍工作画布上进行缓动，再用 Lanczos 缩回 1080p；\n- 对编码、黑帧、响度、字幕冲突和片尾衔接进行验收。\n\n## 工作流\n\n```mermaid\nflowchart LR\n    A[\"剧本与钩子\"] --> B[\"分镜与台词\"]\n    B --> C[\"角色 / 世界视觉基准\"]\n    C --> D[\"逐镜生成关键帧\"]\n    B --> E[\"Edge TTS 多角色配音\"]\n    E --> F[\"实测语音时间轴\"]\n    D --> G[\"防抖镜头渲染\"]\n    F --> G\n    G --> H[\"字幕、音效与混音\"]\n    H --> I[\"封面与最终导出\"]\n    I --> J[\"画面 / 音频 / 编码质检\"]\n```\n\n默认成片规格：\n\n- 1080×1920，9:16；\n- 30 fps，H.264 / AAC；\n- 45–90 秒；\n- 6–12 张关键帧；\n- 约 −16 LUFS，True Peak 不高于 −1.5 dBTP。\n\n## 安装\n\n### 1. 克隆到长期维护目录\n\n```bash\ngit clone https://github.com/ToBeWin/make-motion-comic.git\ncd make-motion-comic\n```\n\n### 2. 让 Codex 发现 Skill\n\n推荐使用符号链接，维护源码时安装版本会同步更新：\n\n```bash\nmkdir -p \"${CODEX_HOME:-$HOME/.codex}/skills\"\nln -s \"$(pwd)\" \"${CODEX_HOME:-$HOME/.codex}/skills/make-motion-comic\"\n```\n\n如果目标位置已经存在，请先确认它是否是需要保留的安装版本，不要直接覆盖。\n\n### 3. 安装依赖\n\n需要：\n\n- Codex，以及可用的内置 Image Generator；\n- Python 3.9+；\n- FFmpeg / FFprobe；\n- Zsh；\n- [`edge-tts`](https://github.com/rany2/edge-tts)。\n\n安装 Edge TTS：\n\n```bash\npython3 -m pip install edge-tts\n```\n\n运行环境检查：\n\n```bash\n./scripts/preflight.sh\n```\n\n返回以下结果即可开始制作：\n\n```json\n{\"ok\":true,\"warnings\":[]}\n```\n\n## 快速使用\n\n在 Codex 中：\n\n```text\n使用 $make-motion-comic，把这个故事制作成一集 60 秒的竖屏动态漫画：\n一名女孩每天都会收到已经去世的哥哥发来的天气预报。\n```\n\nSkill 会依次处理：\n\n1. 剧情结构、钩子和反转；\n2. 角色视觉基准与分镜提示词；\n3. 逐镜图片生成与一致性检查；\n4. Edge TTS 多角色配音；\n5. 实测时间轴、字幕与镜头时长；\n6. 防抖推拉、平移和静止镜头；\n7. 环境声、音效、混音、封面与成片；\n8. 编码、黑帧、响度和视觉检查。\n\n## 核心原则\n\n### 角色一致性\n\n- 先生成包含正面、侧面、全身、表情和服装的视觉基准图；\n- 每个镜头都回到原始基准图获取身份；\n- 上一镜头只能作为姿势或构图参考；\n- 禁止单纯采用 A → B → C 的连续参考链；\n- 对人物、服装、手部、人数、关键道具和意外文字逐项验收。\n\n详细说明见 [`references/image-consistency.md`](references/image-consistency.md)。\n\n### 配音与时间轴\n\n- 正式成片默认不使用 macOS `say`；\n- 每句台词单独生成，角色长期保持固定音色；\n- Edge TTS 网络错误采用有界重试，不静默降级；\n- 修改台词、音色或语速后自动失效旧缓存；\n- 最终时长以 `ffprobe` 测量结果为准。\n\n详细说明见 [`references/audio-and-tts.md`](references/audio-and-tts.md)。\n\n### 防抖运动\n\n慢速 `zoompan` 每帧可能移动不足一个像素。整数取整会形成“停顿—跳一像素”的微抖动，而漫画细线会进一步放大这种现象。\n\n本项目使用：\n\n- 2592×4608 源工作画布；\n- 2160×3840 运动输出；\n- 余弦缓入缓出；\n- 30 fps；\n- Lanczos 缩放至 1080×1920；\n- 单镜头只使用一个主要运动参数；\n- 情绪反转镜头优先静止。\n\n详细说明见 [`references/motion-and-qc.md`](references/motion-and-qc.md)。\n\n## 脚本\n\n### 批量生成 Edge TTS\n\n从模板创建配音清单：\n\n```bash\ncp assets/templates/tts-script.json ./tts-script.json\n```\n\n生成独立语音文件：\n\n```bash\npython3 scripts/synthesize_edge_tts.py \\\n  --manifest tts-script.json \\\n  --out audio/lines \\\n  --speed 1.00\n```\n\n特点：\n\n- 每句独立 MP3 / WAV；\n- 网络错误自动重试；\n- 通过内容指纹安全复用缓存；\n- 支持无变调语速微调；\n- 48 kHz 单声道输出。\n\n### 构建真实时间轴\n\n```bash\npython3 scripts/build_timeline.py \\\n  --manifest tts-script.json \\\n  --audio-dir audio/lines \\\n  --out timeline\n```\n\n输出：\n\n- `timeline/voice.wav`\n- `timeline/subtitles.srt`\n- `timeline/timeline.json`\n- `timeline/voice-concat.txt`\n\n### 渲染防抖镜头\n\n```bash\n./scripts/render_still_clip.sh \\\n  --input images/shot-01.png \\\n  --output video/shot-01.mp4 \\\n  --duration 6.5 \\\n  --motion push \\\n  --fps 30 \\\n  --strength 0.035\n```\n\n可用运动：\n\n- `push`\n- `pull`\n- `pan-left`\n- `pan-right`\n- `hold`\n\n推拉强度通常保持在 `0.02`–`0.05`。\n\n## 项目结构\n\n```text\nmake-motion-comic/\n├── SKILL.md\n├── README.md\n├── LICENSE\n├── agents/\n│   └── openai.yaml\n├── assets/\n│   └── templates/\n│       └── tts-script.json\n├── docs/\n│   └── demo-cover.jpg\n├── references/\n│   ├── audio-and-tts.md\n│   ├── image-consistency.md\n│   ├── motion-and-qc.md\n│   └── story-and-shots.md\n└── scripts/\n    ├── build_timeline.py\n    ├── preflight.sh\n    ├── render_still_clip.sh\n    └── synthesize_edge_tts.py\n```\n\n## 验收\n\nSkill 要求至少检查：\n\n```bash\nffprobe -v error \\\n  -show_entries format=duration:stream=codec_name,width,height,r_frame_rate,sample_rate,channels \\\n  -of json final.mp4\n\nffmpeg -v error -i final.mp4 -f null -\n\nffmpeg -hide_banner -i final.mp4 \\\n  -vf blackdetect=d=0.3:pix_th=0.05 \\\n  -an -f null -\n\nffmpeg -hide_banner -i final.mp4 \\\n  -af loudnorm=I=-16:LRA=9:TP=-1.5:print_format=summary \\\n  -f null -\n```\n\n还需要人工检查：\n\n- 人物身份、服装和关键道具是否漂移；\n- 手部和人数是否正确；\n- 细线、头发、眼睛和瓶口是否抖动或闪烁；\n- 字幕是否遮挡脸部；\n- 最后一句字幕是否与片尾卡重叠；\n- 手机扬声器音量下是否仍能听清对白。\n\n## 已知限制\n\n- Edge TTS 是微软在线语音端点的非官方客户端，需要网络且没有 SLA；\n- 当前脚本重点针对普通话与 9:16 竖屏；\n- Image Generator 的角色一致性仍需人工视觉验收；\n- 当前版本提供可靠的生产组件，还不是完全一键式渲染器；\n- BGM 必须自行生成或使用明确允许复用的素材。\n\n## 路线图\n\n- 统一的 `episode.json` 单一数据源；\n- 一键式全片编排与导出；\n- 自动字幕排版和片尾安全区检查；\n- BGM 自动 ducking 与 SFX 时间轴；\n- 横屏、方形和多平台规格；\n- 可插拔 TTS 供应商；\n- 自动角色漂移检测；\n- 自动生成 QC 报告和版本对比视频。\n\n## 贡献\n\n欢迎提交 Issue 和 Pull Request。贡献代码时请：\n\n1. 保持 `SKILL.md` 简洁，把细节放入 `references/`；\n2. 为新增脚本提供可复现的运行示例；\n3. 运行环境预检和 Skill 验证；\n4. 不提交第三方版权音乐、密钥、Token 或生成缓存；\n5. 说明改动如何改善稳定性、质量或可维护性。\n\n## English summary\n\n`make-motion-comic` is a Codex Skill for producing low-cost motion comics from AI-generated keyframes, multi-character Mandarin Edge TTS, captions, sound design, and deterministic FFmpeg assembly.\n\nIt focuses on four recurring production problems:\n\n- character identity drift across generated shots;\n- robotic or inconsistent speech;\n- subtitle and timing desynchronization;\n- micro-jitter caused by low-resolution integer-rounded `zoompan`.\n\nThe default pipeline targets 1080×1920, 30 fps vertical videos without requiring a generative video model.\n\n## License\n\nReleased under the [MIT License](LICENSE).\n\nFile v0.1.0:_meta.json\n\n{\n  \"ownerId\": \"kn75z6gevjsyrznm7dg2ez6sen82h8sz\",\n  \"slug\": \"make-motion-comic\",\n  \"version\": \"0.1.0\",\n  \"publishedAt\": 1787713560241\n}\n\nFile v0.1.0:references/audio-and-tts.md\n\n# Audio and Edge TTS\n\n## Default voice route\n\nUse Edge TTS for zero-cost online Mandarin Neural speech when its service is available. It is an unofficial client and requires internet access. Run:\n\n```bash\nedge-tts --list-voices | rg '^zh-CN'\n```\n\nDo not silently fall back to macOS `say`; it is suitable for drafts and accessibility, not emotional drama.\n\nIf Edge TTS is unavailable after bounded retries, report the failure and offer a quality-preserving alternative such as an authenticated Neural TTS provider or a local expressive model.\n\n## Cast voices\n\nKeep voice identity stable across episodes. Recommended starting points:\n\n| Role | Voice | Treatment |\n|---|---|---|\n| Warm narrator | `zh-CN-XiaoxiaoNeural` | rate −6% to +2%, pitch −3 to 0 Hz |\n| Adult woman | `zh-CN-XiaoyiNeural` | rate −12% to 0%, pitch −5 to 0 Hz |\n| Controlled man | `zh-CN-YunyangNeural` | rate −14% to −4%, pitch −10 to −4 Hz |\n| Younger urgent man | `zh-CN-YunxiNeural` | rate 0% to +8%, pitch −4 to 0 Hz |\n| Child | `zh-CN-XiaoyiNeural` | rate −12% to −4%, pitch +8 to +16 Hz |\n\nTreat these as starting points. Generate a short audition when a new recurring cast is created.\n\n## Acting through text\n\n- Use full stops for conviction.\n- Use a comma for a short breath.\n- Use one ellipsis only for meaningful hesitation.\n- Avoid repeated ellipses; Edge TTS may create long dead air.\n- Split long exposition into separate line assets.\n- Give urgent lines a faster rate instead of using exclamation marks everywhere.\n\n## Manifest\n\nUse `assets/templates/tts-script.json` and save one audio file per line. This enables:\n\n- individual voice replacement;\n- exact subtitle timing;\n- per-line gain and pacing;\n- resumable online synthesis.\n\nRun:\n\n```bash\npython3 scripts/synthesize_edge_tts.py \\\n  --manifest tts-script.json \\\n  --out audio/lines \\\n  --speed 1.00\n```\n\nUse a small, pitch-preserving `--speed` adjustment only after hearing the generated pacing. Prefer 0.96–1.10. Do not accelerate poor acting into acceptable duration.\n\n## Mix\n\n- Normalize and lightly compress each voice asset.\n- Keep ambience 12–20 dB beneath voice.\n- Place SFX relative to measured line/shot times.\n- Use sidechain ducking or keyframed bed volume under speech.\n- Measure the finished program, not only the voice bus.\n\nTarget approximately −16 LUFS integrated and ≤−1.5 dBTP.\n\nFile v0.1.0:references/image-consistency.md\n\n# Image consistency\n\n## Build the visual bible\n\nCreate one reference image containing:\n\n- neutral full-body front, side, and back views;\n- face closeups with neutral and high-value expressions;\n- signature clothes and accessories;\n- palette and material details;\n- a small environment/world inset.\n\nUse distinct silhouette and color cues. Avoid two leads with nearly identical hair, coat, or face shape.\n\n## Reference hierarchy\n\nFor every shot:\n\n1. Use the visual bible as the identity reference.\n2. Use an approved environment frame when location continuity matters.\n3. Use the previous shot only as a pose/composition reference.\n\nNever rely only on the immediately previous generated frame. Small deviations compound across a chain.\n\nLabel input roles in the image prompt:\n\n```text\nInput images:\n- Image 1: identity and wardrobe source of truth.\n- Image 2: environment and lighting reference only.\n- Image 3: pose continuity reference only.\n```\n\n## Prompt skeleton\n\n```text\nUse case: illustration-story\nAsset type: vertical 9:16 motion-comic keyframe\nPrimary request: <single story moment>\nInput images: <labeled roles>\nSubject: <named characters with locked visual traits>\nScene/backdrop: <location and continuity details>\nStyle/medium: <series art direction>\nComposition/framing: <shot size, focal subject, caption-safe area>\nLighting/mood: <series palette plus local change>\nConstraints: preserve exact identities, wardrobe, proportions, and signature props\nAvoid: text, captions, speech bubbles, logos, watermarks, duplicate people, extra fingers\n```\n\n## Review\n\nInspect at full size. Reject or correct:\n\n- changed hairstyle, age, facial proportions, or clothing;\n- missing signature accessory;\n- duplicate or fused people;\n- wrong hand count;\n- changed bottle, weapon, photo, or other story-critical prop;\n- accidental letters;\n- focal face placed beneath planned captions.\n\nWhen a shot fails, return to the visual bible and correct one issue at a time.\n\nFile v0.1.0:references/motion-and-qc.md\n\n# Motion and quality control\n\n## Why micro-jitter happens\n\nFFmpeg `zoompan` evaluates crop position on a pixel grid. A slow move may request less than one source pixel per output frame. Integer rounding then produces a repeating pattern: hold, hold, jump one pixel. Fine comic linework makes the jump look like camera shake or texture shimmer.\n\nAdditional causes:\n\n- rendering motion directly at 1080×1920 with little overscan;\n- 24 fps linear movement across a small distance;\n- simultaneous zoom and pan with unrelated timing;\n- high-frequency paper grain or ink lines resampled with a weak filter;\n- H.264 bitrate too low for moving line art.\n\n## Required prevention\n\nUse `scripts/render_still_clip.sh`, or reproduce all of its principles:\n\n1. Scale to a working canvas at least twice delivery resolution.\n2. Animate crop/zoom coordinates on that oversized canvas.\n3. Use cosine ease-in/ease-out.\n4. Downsample with Lanczos.\n5. Default to 30 fps.\n6. Move only one dominant camera parameter per shot.\n7. Keep zoom strength around 2–5%.\n8. Include intentional holds.\n\nDo not use shake, random motion, unseeded noise, or frame-by-frame position randomness unless the story explicitly calls for an impact.\n\n## Motion grammar\n\n| Story purpose | Motion |\n|---|---|\n| Realization, danger, intimacy | Slow push |\n| Isolation, consequence | Slow pull |\n| Reveal a prop or second character | Eased pan |\n| Shock, grief, final line | Hold |\n| Entering a location | Short pan or push, not both |\n\nAlternate movement with holds. If every shot moves, motion stops carrying meaning.\n\n## Validation\n\nInspect the final encoded video, not only the raw source:\n\n- play on a phone-sized window and full size;\n- inspect eyes, hair strands, bottle edges, and high-contrast lines;\n- step frame-by-frame through the slowest move;\n- extract 2–3 consecutive frames and compare crop movement;\n- check 100% and 50% playback speed.\n\nTechnical checks:\n\n```bash\nffmpeg -v error -i final.mp4 -f null -\nffmpeg -hide_banner -i final.mp4 -vf blackdetect=d=0.3:pix_th=0.05 -an -f null -\nffprobe -v error -show_entries format=duration,size:stream=codec_name,width,height,r_frame_rate,pix_fmt -of json final.mp4\n```\n\nUse CRF 16–19 for detailed line art and `yuv420p` for broad compatibility.\n\nFile v0.1.0:references/story-and-shots.md\n\n# Story and shot design\n\n## Episode architecture\n\nFor a 45–90 second serial motion comic, use:\n\n1. **0–3 s — Hook:** an impossible question, alarming image, or contradiction.\n2. **3–20 s — Situation:** identify the character, need, and immediate constraint.\n3. **20–40 s — Rule or price:** reveal what the character must risk.\n4. **40–65 s — Consequence:** show the choice changing reality.\n5. **Final 10–20 s — Double turn:** resolve the episode's emotional question, then reveal a larger series mystery.\n\nThe hook must be understandable without prior episodes. The final beat must add information rather than merely saying “未完待续”.\n\n## Shot selection\n\nCreate a new keyframe only when at least one changes:\n\n- story location;\n- speaker or point of view;\n- emotional state;\n- decisive prop state;\n- revealed information;\n- time.\n\nDo not create a new image because a fixed number of seconds elapsed.\n\nTypical 60–80 second episode:\n\n- 8–10 keyframes;\n- 4–9 seconds per keyframe;\n- 1–3 characters per frame;\n- no more than two main locations.\n\n## Writing constraints\n\n- Write dialogue for speech, not prose.\n- Put exposition in short narration sentences.\n- Give different characters different sentence rhythm.\n- Avoid explaining what the frame already proves.\n- Keep character names and key terms pronunciation-friendly.\n- Reserve silence after a reversal; do not fill every second with speech.\n\n## Shot table fields\n\nRecord:\n\n| Field | Purpose |\n|---|---|\n| `shot` | Stable shot number |\n| `story beat` | What changes |\n| `visual` | Subject, action, framing, light |\n| `spoken lines` | IDs from TTS manifest |\n| `motion` | hold, push, pull, pan-left, pan-right |\n| `SFX` | Event and intended timestamp |\n| `caption safe area` | Top, lower third, or custom |\n\nWrite the complete spoken script and shot table before starting image generation.\n\nFile v0.1.0:skill-card.md\n\n## Description:\n\nCreate or revise low-cost motion-comic videos from a story or script using consistent AI-generated keyframes, multi-character Chinese Edge TTS, captions, synthesized or licensed audio, and FFmpeg assembly.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[tobewin](https://clawhub.ai/user/tobewin)\n\n### License/Terms of Use:\n\nMIT\n\n## Use Case:\n\nDevelopers and creators use this skill to turn scripts or story ideas into vertical motion-comic episodes with reusable story, image, voice, subtitle, timing, and render assets. It is also used to diagnose and revise common production issues such as character drift, robotic TTS, subtitle timing problems, jittery motion, and audio balance.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Online TTS may send script text outside the local machine.\n\nMitigation: Avoid confidential scripts unless this data flow is acceptable, or use a reviewed alternative TTS route that matches the release requirements.\n\nRisk: Helper scripts can write outside the chosen output folder when given unsafe paths or untrusted manifests.\n\nMitigation: Run helper scripts only with trusted manifests and reviewed project-local paths until path validation and no-overwrite safeguards are added.\n\n## Reference(s):\n\n- [Server-resolved GitHub provenance: ToBeWin/make-motion-comic](https://github.com/ToBeWin/make-motion-comic)\n- [ClawHub skill page](https://clawhub.ai/tobewin/skills/make-motion-comic)\n- [Story and Shots](references/story-and-shots.md)\n- [Image Consistency](references/image-consistency.md)\n- [Audio and Edge TTS](references/audio-and-tts.md)\n- [Motion and QC](references/motion-and-qc.md)\n- [Edge TTS project](https://github.com/rany2/edge-tts)\n- [FFmpeg](https://ffmpeg.org/)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, code, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown guidance with JSON manifests, shell commands, generated prompts, subtitles, timelines, and media-production file paths]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May produce project-local production assets such as scripts, shot lists, keyframe prompts, TTS manifests, subtitle files, timeline JSON, audio mixes, covers, contact sheets, and final video outputs.]\n\n## Skill Version(s):\n\n0.1.0 (source: server-resolved release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v0.1.0:assets/templates/tts-script.json\n\n{\n  \"lines\": [\n    {\n      \"id\": \"01\",\n      \"shot\": 1,\n      \"speaker\": \"旁白\",\n      \"voice\": \"zh-CN-XiaoxiaoNeural\",\n      \"rate\": \"-4%\",\n      \"pitch\": \"-2Hz\",\n      \"volume\": \"+0%\",\n      \"text\": \"把第一句台词放在这里。\",\n      \"pause_after\": 0.5\n    }\n  ]\n}\n\nFile v0.1.0:agents/openai.yaml\n\ninterface:\n  display_name: \"动态漫画制作\"\n  short_description: \"从剧本、统一角色图到 Edge TTS 与防抖竖屏成片\"\n  default_prompt: \"使用 $make-motion-comic 把这个故事制作成一集竖屏动态漫画。\"\n\nFile v0.1.0:LICENSE\n\nMIT License\n\nCopyright (c) 2026 ToBeWin\n\nPermission is hereby granted, free of charge, to any person obtaining a copy\nof this software and associated documentation files (the \"Software\"), to deal\nin the Software without restriction, including without limitation the rights\nto use, copy, modify, merge, publish, distribute, sublicense, and/or sell\ncopies of the Software, and to permit persons to whom the Software is\nfurnished to do so, subject to the following conditions:\n\nThe above copyright notice and this permission notice shall be included in all\ncopies or substantial portions of the Software.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR\nIMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,\nFITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE\nAUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER\nLIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,\nOUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE\nSOFTWARE.","readmeExcerpt":"Skill: 动态漫画制作 Owner: tobewin Summary: 从剧本、统一角色图到 Edge TTS 与防抖竖屏成片 Tags: latest:0.1.0 Version history: v0.1.0 | 2026-08-26T03:06:00.241Z | auto Initial release of the make-motion-comic skill. - Create low-cost motion-comic videos from story/script using AI-generated keyframes, multi-character Chinese TTS, captions, and FFmpeg assembly. - Ensures character identity consistency, professional voice quality, smooth motion","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"python3 scripts/build_timeline.py \\\n  --manifest tts-script.json \\\n  --audio-dir audio/lines \\\n  --out timeline"},{"language":"bash","snippet":"ffprobe -v error -show_entries format=duration:stream=codec_name,width,height,r_frame_rate,sample_rate,channels -of json final.mp4\nffmpeg -v error -i final.mp4 -f null -\nffmpeg -hide_banner -i final.mp4 -vf blackdetect=d=0.3:pix_th=0.05 -an -f null -\nffmpeg -hide_banner -i final.mp4 -af loudnorm=I=-16:LRA=9:TP=-1.5:print_format=summary -f null -"},{"language":"mermaid","snippet":"flowchart LR\n    A[\"剧本与钩子\"] --> B[\"分镜与台词\"]\n    B --> C[\"角色 / 世界视觉基准\"]\n    C --> D[\"逐镜生成关键帧\"]\n    B --> E[\"Edge TTS 多角色配音\"]\n    E --> F[\"实测语音时间轴\"]\n    D --> G[\"防抖镜头渲染\"]\n    F --> G\n    G --> H[\"字幕、音效与混音\"]\n    H --> I[\"封面与最终导出\"]\n    I --> J[\"画面 / 音频 / 编码质检\"]"},{"language":"bash","snippet":"git clone https://github.com/ToBeWin/make-motion-comic.git\ncd make-motion-comic"},{"language":"bash","snippet":"mkdir -p \"${CODEX_HOME:-$HOME/.codex}/skills\"\nln -s \"$(pwd)\" \"${CODEX_HOME:-$HOME/.codex}/skills/make-motion-comic\""},{"language":"bash","snippet":"python3 -m pip install edge-tts"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: make-motion-comic\ndescription: Create or revise low-cost motion-comic videos from a story or script using consistent AI-generated keyframes, multi-character Chinese Edge TTS, captions, synthesized or licensed audio, and FFmpeg assembly. Use for 动态漫画、漫剧、条漫视频、animated manga/comic, narrated image-story shorts, vertical story videos, or when Codex must turn generated still images into a polished video without a generative video model; also use to diagnose or fix character drift, robotic TTS, subtitle timing, micro-jitter, shaky zoompan motion, audio balance, covers, and reusable episode production assets.\n---\n\n# Make Motion Comic\n\nProduce a complete motion-comic episode from a script while keeping image identity, voice quality, motion smoothness, and source assets independently editable.\n\n## Required route\n\n1. Use the built-in Image Generator through the available `imagegen` skill for character sheets, keyframes, image corrections, and covers. Follow that skill's reference-image and save-path rules.\n2. Use Edge TTS for Chinese production voice by default. Do not use macOS `say` for a final deliverable unless the user explicitly chooses its offline quality tradeoff.\n3. Use FFmpeg for deterministic motion, audio mixing, captions, encoding, and inspection. A user may explicitly choose another video framework.\n4. Run `scripts/preflight.sh` before producing media. Report missing required dependencies before continuing.\n\nEdge TTS uses an unofficial client for Microsoft's online speech endpoint. It is free in normal use but needs network access and has no service guarantee. Retry transient failures; do not silently substitute a worse voice.\n\n## Default production brief\n\nUse these defaults when the user says “开始”“直接做” or otherwise authorizes an autonomous first pass:\n\n- Format: 9:16, 1080×1920, 30 fps, H.264/AAC.\n- Length: 45–90 seconds.\n- Visuals: 6–12 keyframes; use a new image when story state changes, not at arbitrary time intervals.\n- Structure: hook in 0–3 seconds, rule or dilemma, escalation, emotional reversal, final serial cliffhanger.\n- Captions: render in post; never ask the image model to typeset dialogue.\n- Audio: multi-character Neural voices, light ambience/SFX, no unverified copyrighted BGM.\n- Review: inspect the character sheet, raw keyframe contact sheet, and final-video snapshots.\n\nRead [references/story-and-shots.md](references/story-and-shots.md) before writing a new episode. Read [references/image-consistency.md](references/image-consistency.md) before generating images.\n\n## Workflow\n\n### 1. Establish the production package\n\nCreate a project-local working folder and a user-facing output folder. Preserve:\n\n- script and shot table;\n- character/world visual bible;\n- one prompt per keyframe;\n- raw keyframes;\n- TTS manifest and voice-only mix;\n- subtitle file and timed timeline;\n- final mix, cover, contact sheet, and final video.\n\nKeep temporary render fragments outside the user-facing output folder.\n\n### 2. Write for motion comics"},{"path":"README.md","content":"# Make Motion Comic\n\n用 AI 图片、Edge TTS 与 FFmpeg 制作低成本、高一致性、可维护的动态漫画。\n\n[![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE)\n[![Codex Skill](https://img.shields.io/badge/Codex-Skill-111827)](SKILL.md)\n[![FFmpeg](https://img.shields.io/badge/FFmpeg-Required-007808)](https://ffmpeg.org/)\n[![Edge TTS](https://img.shields.io/badge/Edge%20TTS-Neural%20Voice-2563EB)](https://github.com/rany2/edge-tts)\n\n`make-motion-comic` 是一个面向 Codex 的动态漫画制作 Skill。它把“剧本 → 角色视觉基准 → 逐镜图片 → 多角色配音 → 防抖镜头 → 字幕与混音 → 成片质检”固化为可复用的生产流程，不依赖生成式视频模型。\n\n它适合：\n\n- 动态漫画、漫剧、条漫视频；\n- 竖屏悬疑、情感、科普和连续短剧；\n- 用少量关键帧制作有声音、有节奏的故事视频；\n- 修复角色漂移、机器人配音、字幕错位和 `zoompan` 微抖动。\n\n> 当前版本重点支持普通话、9:16 竖屏和 FFmpeg 工作流。\n\n![动态漫画效果预览](docs/demo-cover.jpg)\n\n## 为什么做这个 Skill\n\n图片生成模型已经能够产出质量很高的漫画关键帧，但将图片直接拼成视频通常会出现四类问题：\n\n1. 人物在镜头之间变脸、换衣服或改变年龄；\n2. 系统 TTS 缺乏情绪，破坏画面建立的氛围；\n3. 低分辨率 `zoompan` 产生一像素跳动和细线闪烁；\n4. 配音、字幕、镜头和片尾各自计时，返工后全部错位。\n\n本项目为这些问题提供明确的生产规则和可执行脚本：\n\n- 始终使用原始角色视觉基准作为身份锚点；\n- 默认使用多角色中文 Neural Edge TTS；\n- 根据真实语音文件重建时间轴；\n- 在 2 倍工作画布上进行缓动，再用 Lanczos 缩回 1080p；\n- 对编码、黑帧、响度、字幕冲突和片尾衔接进行验收。\n\n## 工作流\n\n```mermaid\nflowchart LR\n    A[\"剧本与钩子\"] --> B[\"分镜与台词\"]\n    B --> C[\"角色 / 世界视觉基准\"]\n    C --> D[\"逐镜生成关键帧\"]\n    B --> E[\"Edge TTS 多角色配音\"]\n    E --> F[\"实测语音时间轴\"]\n    D --> G[\"防抖镜头渲染\"]\n    F --> G\n    G --> H[\"字幕、音效与混音\"]\n    H --> I[\"封面与最终导出\"]\n    I --> J[\"画面 / 音频 / 编码质检\"]\n```\n\n默认成片规格：\n\n- 1080×1920，9:16；\n- 30 fps，H.264 / AAC；\n- 45–90 秒；\n- 6–12 张关键帧；\n- 约 −16 LUFS，True Peak 不高于 −1.5 dBTP。\n\n## 安装\n\n### 1. 克隆到长期维护目录\n\n```bash\ngit clone https://github.com/ToBeWin/make-motion-comic.git\ncd make-motion-comic\n```\n\n### 2. 让 Codex 发现 Skill\n\n推荐使用符号链接，维护源码时安装版本会同步更新：\n\n```bash\nmkdir -p \"${CODEX_HOME:-$HOME/.codex}/skills\"\nln -s \"$(pwd)\" \"${CODEX_HOME:-$HOME/.codex}/skills/make-motion-comic\"\n```\n\n如果目标位置已经存在，请先确认它是否是需要保留的安装版本，不要直接覆盖。\n\n### 3. 安装依赖\n\n需要：\n\n- Codex，以及可用的内置 Image Generator；\n- Python 3.9+；\n- FFmpeg / FFprobe；\n- Zsh；\n- [`edge-tts`](https://github.com/rany2/edge-tts)。\n\n安装 Edge TTS：\n\n```bash\npython3 -m pip install edge-tts\n```\n\n运行环境检查：\n\n```bash\n./scripts/preflight.sh\n```\n\n返回以下结果即可开始制作：\n\n```json\n{\"ok\":true,\"warnings\":[]}\n```\n\n## 快速使用\n\n在 Codex 中：\n\n```text\n使用 $make-motion-comic，把这个故事制作成一集 60 秒的竖屏动态漫画：\n一名女孩每天都会收到已经去世的哥哥发来的天气预报。\n```\n\nSkill 会依次处理：\n\n1. 剧情结构、钩子和反转；\n2. 角色视觉基准与分镜提示词；\n3. 逐镜图片生成与一致性检查；\n4. Edge TTS 多角色配音；\n5. 实测时间轴、字幕与镜头时长；\n6. 防抖推拉、平移和静止镜头；\n7. 环境声、音效、混音、封面与成片；\n8. 编码、黑帧、响度和视觉检查。\n\n## 核心原则\n\n### 角色一致性\n\n- 先生成包含正面、侧面、全身、表情和服装的视觉基准图；\n- 每个镜头都回到原始基准图获取身份；\n- 上一镜头只能作为姿势或构图参考；\n- 禁止单纯采用 A → B → C 的连续参考链；\n- 对人物、服装、手部、人数、关键道具和意外文字逐项验收。\n\n详细说明见 [`references/image-consistency.md`](references/image-consistency.md)。\n\n### 配音与时间轴\n\n- 正式成片默认不使用 macOS `say`；\n- 每句台词单独生成，角色长期保持固定音色；\n- Edge TTS 网络错误采用有界重试，不静默降级；\n- 修改台词、音色或语速后自动失效旧缓存；\n- 最终时长以 `ffprobe` 测量结果为准。\n\n详细说明见 [`references/audio-and-tts.md`](references/audio-and-tts.md)。\n\n### 防抖运动\n\n慢速 `zoompan` 每帧可能移动不足一个像素。整数取整会形成“停顿—跳一像素”的微抖动，而漫画细线会进一步放大这种现象。\n\n本项目使用：\n\n- 2592×4608 源工作画布；\n- 2160×3840 运动输出；\n- 余弦缓入缓出；\n- 30 fps；\n- Lanczos 缩放至 1080×1920；\n- 单镜头只使用一个主要运动参数；\n- 情绪反转镜头优先静止。\n\n详细说明见 [`references/motion-a"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn75z6gevjsyrznm7dg2ez6sen82h8sz\",\n  \"slug\": \"make-motion-comic\",\n  \"version\": \"0.1.0\",\n  \"publishedAt\": 1787713560241\n}"},{"path":"references/audio-and-tts.md","content":"# Audio and Edge TTS\n\n## Default voice route\n\nUse Edge TTS for zero-cost online Mandarin Neural speech when its service is available. It is an unofficial client and requires internet access. Run:\n\n```bash\nedge-tts --list-voices | rg '^zh-CN'\n```\n\nDo not silently fall back to macOS `say`; it is suitable for drafts and accessibility, not emotional drama.\n\nIf Edge TTS is unavailable after bounded retries, report the failure and offer a quality-preserving alternative such as an authenticated Neural TTS provider or a local expressive model.\n\n## Cast voices\n\nKeep voice identity stable across episodes. Recommended starting points:\n\n| Role | Voice | Treatment |\n|---|---|---|\n| Warm narrator | `zh-CN-XiaoxiaoNeural` | rate −6% to +2%, pitch −3 to 0 Hz |\n| Adult woman | `zh-CN-XiaoyiNeural` | rate −12% to 0%, pitch −5 to 0 Hz |\n| Controlled man | `zh-CN-YunyangNeural` | rate −14% to −4%, pitch −10 to −4 Hz |\n| Younger urgent man | `zh-CN-YunxiNeural` | rate 0% to +8%, pitch −4 to 0 Hz |\n| Child | `zh-CN-XiaoyiNeural` | rate −12% to −4%, pitch +8 to +16 Hz |\n\nTreat these as starting points. Generate a short audition when a new recurring cast is created.\n\n## Acting through text\n\n- Use full stops for conviction.\n- Use a comma for a short breath.\n- Use one ellipsis only for meaningful hesitation.\n- Avoid repeated ellipses; Edge TTS may create long dead air.\n- Split long exposition into separate line assets.\n- Give urgent lines a faster rate instead of using exclamation marks everywhere.\n\n## Manifest\n\nUse `assets/templates/tts-script.json` and save one audio file per line. This enables:\n\n- individual voice replacement;\n- exact subtitle timing;\n- per-line gain and pacing;\n- resumable online synthesis.\n\nRun:\n\n```bash\npython3 scripts/synthesize_edge_tts.py \\\n  --manifest tts-script.json \\\n  --out audio/lines \\\n  --speed 1.00\n```\n\nUse a small, pitch-preserving `--speed` adjustment only after hearing the generated pacing. Prefer 0.96–1.10. Do not accelerate poor acting into acceptable duration.\n\n## Mix\n\n- Normalize and lightly compress each voice asset.\n- Keep ambience 12–20 dB beneath voice.\n- Place SFX relative to measured line/shot times.\n- Use sidechain ducking or keyframed bed volume under speech.\n- Measure the finished program, not only the voice bus.\n\nTarget approximately −16 LUFS integrated and ≤−1.5 dBTP."},{"path":"references/image-consistency.md","content":"# Image consistency\n\n## Build the visual bible\n\nCreate one reference image containing:\n\n- neutral full-body front, side, and back views;\n- face closeups with neutral and high-value expressions;\n- signature clothes and accessories;\n- palette and material details;\n- a small environment/world inset.\n\nUse distinct silhouette and color cues. Avoid two leads with nearly identical hair, coat, or face shape.\n\n## Reference hierarchy\n\nFor every shot:\n\n1. Use the visual bible as the identity reference.\n2. Use an approved environment frame when location continuity matters.\n3. Use the previous shot only as a pose/composition reference.\n\nNever rely only on the immediately previous generated frame. Small deviations compound across a chain.\n\nLabel input roles in the image prompt:\n\n```text\nInput images:\n- Image 1: identity and wardrobe source of truth.\n- Image 2: environment and lighting reference only.\n- Image 3: pose continuity reference only.\n```\n\n## Prompt skeleton\n\n```text\nUse case: illustration-story\nAsset type: vertical 9:16 motion-comic keyframe\nPrimary request: <single story moment>\nInput images: <labeled roles>\nSubject: <named characters with locked visual traits>\nScene/backdrop: <location and continuity details>\nStyle/medium: <series art direction>\nComposition/framing: <shot size, focal subject, caption-safe area>\nLighting/mood: <series palette plus local change>\nConstraints: preserve exact identities, wardrobe, proportions, and signature props\nAvoid: text, captions, speech bubbles, logos, watermarks, duplicate people, extra fingers\n```\n\n## Review\n\nInspect at full size. Reject or correct:\n\n- changed hairstyle, age, facial proportions, or clothing;\n- missing signature accessory;\n- duplicate or fused people;\n- wrong hand count;\n- changed bottle, weapon, photo, or other story-critical prop;\n- accidental letters;\n- focal face placed beneath planned captions.\n\nWhen a shot fails, return to the visual bible and correct one issue at a time."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"从剧本、统一角色图到 Edge TTS 与防抖竖屏成片 Skill: 动态漫画制作 Owner: tobewin Summary: 从剧本、统一角色图到 Edge TTS 与防抖竖屏成片 Tags: latest:0.1.0 Version history: v0.1.0 | 2026-08-26T03:06:00.241Z | auto Initial release of the make-motion-comic skill. - Create low-cost motion-comic videos from story/script using AI-generated keyframes, multi-character Chinese TTS, captions, and FFmpeg assembly. - Ensures character identity consistency, professional voice quality, smooth motion","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1425,"uniquenessScore":55,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T12:58:34.857Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T12:58:34.857Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T15:15:29.204Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}