{"id":"2beb5431-7ee4-4202-8e0c-f0b79edb4e68","entityType":"agent","slug":"clawhub-darknoah-video-learning-notes","name":"视频自动笔记制作","canonicalUrl":"https://www.xpersona.co/agent/clawhub-darknoah-video-learning-notes","canonicalPath":"/agent/clawhub-darknoah-video-learning-notes","generatedAt":"2026-10-10T17:33:31.353Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-10T14:00:57.329Z","emptyReason":null},"description":"Use this skill when the user provides a video URL and wants a complete Markdown learning note. It downloads the original video, transcribes audio with qwen-a... Skill: 视频自动笔记制作 Owner: darknoah Summary: Use this skill when the user provides a video URL and wants a complete Markdown learning note. It downloads the original video, transcribes audio with qwen-a... Tags: latest:1.0.2 Version history: v1.0.2 | 2026-05-19T15:55:02.074Z | user No user-facing or workflow changes in this version. - No files were changed. - Documentation, functionality, and output remain the same as th","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.4K downloads reported by the source. Last updated 10/10/2026.","installCommand":"clawhub skill install s17ej6pe6hc67ygkxpdwcwpfd583gryy:video-learning-notes","sourceUrl":"https://clawhub.ai/darknoah/video-learning-notes","homepage":"https://clawhub.ai/darknoah/skills/video-learning-notes","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/darknoah/video-learning-notes","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/darknoah/skills/video-learning-notes","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":63,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Use this skill when the user provides a video URL and wants a complete Markdown learning note. It downloads the original video, transcribes audio with qwen-a..."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-10T14:00:57.329Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T14:00:57.329Z","emptyReason":null},"stars":null,"forks":null,"downloads":1399,"packageName":null,"latestVersion":"1.0.2","tractionLabel":"1.4K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T14:00:57.329Z","emptyReason":null},"lastUpdatedAt":"2026-10-10T14:00:57.329Z","lastCrawledAt":"2026-10-10T14:00:57.329Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-11T14:00:57.329Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.2","createdAt":"2026-05-19T15:55:02.074Z","changelog":"No user-facing or workflow changes in this version. - No files were changed. - Documentation, functionality, and output remain the same as the previous release.","fileCount":5,"zipByteSize":8994},{"version":"1.0.1","createdAt":"2026-05-19T15:24:52.550Z","changelog":"- Updated output requirements: Clarified that the final Markdown file should use relative paths for the video and images. - Small edit in documentation to highlight that the Markdown should cite screenshot timestamps and use relative links for the source video and images. - No workflow steps or logic changed; no code changes detected.","fileCount":4,"zipByteSize":7756},{"version":"1.0.0","createdAt":"2026-05-19T12:54:17.805Z","changelog":"video-learning-notes v1.0.0 - Initial release of the skill. - Converts a video URL or local file into a complete Markdown learning note with structured subtitles and key screenshots. - Downloads the original video, extracts and transcribes audio with qwen-audio/STT, and uses ffmpeg for timestamped frame extraction. - Filters key frames by visually analyzing screenshots alongside transcript to select only useful learning images. - Outputs a self-contained directory with the video, transcript, candidate/selected frames, and a final illustrated Markdown note.","fileCount":4,"zipByteSize":7673}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17ej6pe6hc67ygkxpdwcwpfd583gryy:video-learning-notes","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-darknoah-video-learning-notes/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-darknoah-video-learning-notes/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-darknoah-video-learning-notes/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-darknoah-video-learning-notes/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-darknoah-video-learning-notes/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-darknoah-video-learning-notes/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T17:33:31.352Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-darknoah-video-learning-notes/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-darknoah-video-learning-notes/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-darknoah-video-learning-notes/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-darknoah-video-learning-notes/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-10T14:00:57.329Z","emptyReason":null},"readme":"Skill: 视频自动笔记制作\n\nOwner: darknoah\n\nSummary: Use this skill when the user provides a video URL and wants a complete Markdown learning note. It downloads the original video, transcribes audio with qwen-a...\n\nTags: latest:1.0.2\n\nVersion history:\n\nv1.0.2 | 2026-05-19T15:55:02.074Z | user\n\nNo user-facing or workflow changes in this version.\n\n- No files were changed.\n- Documentation, functionality, and output remain the same as the previous release.\n\nv1.0.1 | 2026-05-19T15:24:52.550Z | user\n\n- Updated output requirements: Clarified that the final Markdown file should use relative paths for the video and images.\n- Small edit in documentation to highlight that the Markdown should cite screenshot timestamps and use relative links for the source video and images.\n- No workflow steps or logic changed; no code changes detected.\n\nv1.0.0 | 2026-05-19T12:54:17.805Z | user\n\nvideo-learning-notes v1.0.0\n\n- Initial release of the skill.\n- Converts a video URL or local file into a complete Markdown learning note with structured subtitles and key screenshots.\n- Downloads the original video, extracts and transcribes audio with qwen-audio/STT, and uses ffmpeg for timestamped frame extraction.\n- Filters key frames by visually analyzing screenshots alongside transcript to select only useful learning images.\n- Outputs a self-contained directory with the video, transcript, candidate/selected frames, and a final illustrated Markdown note.\n\nArchive index:\n\nArchive v1.0.2: 5 files, 8994 bytes\n\nFiles: references/api_reference.md (822b), scripts/prepare_video_learning_assets.py (6296b), skill-card.md (2275b), SKILL.md (10228b), _meta.json (139b)\n\nFile v1.0.2:SKILL.md\n\n---\nname: video-learning-notes\ndescription: Use this skill when the user provides a video URL and wants a complete Markdown learning note. It downloads the original video, transcribes audio with qwen-audio/STT, extracts timestamped frames with ffmpeg, reads and filters key screenshots one by one in combination with subtitles, and finally generates an illustrated learning note.\n---\n\n# Video Learning Notes\n\n## Overview\n\nConvert a video URL or local video file into a complete Markdown learning note. The note should be structured from the STT subtitle content and include selected key screenshots. Use this skill for requests such as “turn this video into learning notes”, “download this video and transcribe/analyze it”, or similar video-to-learning-note tasks.\n\n## Required Output\n\nCreate a self-contained output directory containing:\n\n- The downloaded original video file.\n- `transcript.srt` generated by qwen-audio/STT.\n- Timestamped frames extracted by ffmpeg under `frames/`.\n- Manually selected key screenshots under `selected_frames/`.\n- The final Markdown file, usually named `video_learning_notes.md`, using relative paths for the source video and images, and citing screenshot timestamps.\n\n## Workflow\n\n### 1. Create a workspace\n\nCreate a dedicated output directory for each video note. Prefer the current task directory or a stable path such as `./<note-title>/`. Keep all generated files inside this directory; do not scatter outputs into shared default folders.\n\n### 2. Download the original video\n\nIf the source is an online video, use the `yt-dlp-downloader` skill/workflow to download the user-provided video URL. Preserve the original or best available quality when possible, and write the video into the current workspace.\n\nCheck dependencies when needed before downloading:\n\n```bash\nwhich yt-dlp || echo \"yt-dlp not installed. Install with: pip install yt-dlp\"\nwhich ffmpeg || echo \"ffmpeg not installed. Install with: brew install ffmpeg\"\n```\n\nRecommended commands:\n\n```bash\n# Generic: download best quality into the workspace\nyt-dlp -P \"/path/to/workspace\" -o \"%(title)s.%(ext)s\" \"VIDEO_URL\"\n\n# YouTube: use browser cookies by default to reduce 403 errors\nyt-dlp -P \"/path/to/workspace\" --cookies-from-browser chrome -o \"%(title)s.%(ext)s\" \"YOUTUBE_URL\"\n\n# Download subtitles when available; still run qwen-audio/STT unless the user only wants official subtitles\nyt-dlp -P \"/path/to/workspace\" --write-subs --sub-langs all -o \"%(title)s.%(ext)s\" \"VIDEO_URL\"\n```\n\nPlatform handling principles:\n\n- YouTube / YouTube Music: use `--cookies-from-browser chrome` by default. Supported browser cookie sources include `chrome`, `firefox`, `safari`, `edge`, `brave`, and `opera`.\n- Bilibili, Twitter/X, TikTok, Douyin, Vimeo, Twitch, and most other platforms: try direct download first.\n- Playlist URLs: ask the user whether to process the entire playlist, one specific video, or a specific range.\n- Quality selection: default to the best available quality. If the user specifies a quality, use format selectors such as `bestvideo[height<=1080]+bestaudio/best[height<=1080]`.\n\nAfter downloading, identify the actual video file path, such as `.mp4`, `.mkv`, `.mov`, `.webm`, etc. If multiple files are produced, choose the main video as the source for the learning note, while keeping subtitles, thumbnails, and other files as supporting assets.\n\nTroubleshooting:\n\n- `HTTP 403 Forbidden`: retry with `--cookies-from-browser chrome` or another browser where the user is logged in.\n- `Video unavailable`, private videos, or geo-restricted videos: ask the user for login access, cookies, or an accessible environment; do not bypass access restrictions.\n- `Format not available`: run `yt-dlp -F \"VIDEO_URL\"` to list available formats, then choose one.\n- Interrupted downloads: retry; yt-dlp can usually resume partial downloads.\n- `yt-dlp: command not found`: install `yt-dlp` or ask the user to install it.\n\nIf `yt-dlp-downloader` / `yt-dlp` is unavailable, or if the video requires login/authentication, stop and ask the user to provide the missing access requirement instead of silently switching to unreliable tools.\n\n### 3. Transcribe with qwen-audio\n\nRun qwen-audio/STT on the downloaded video or extracted audio, and save the result as `transcript.srt`.\n\nFor large videos, first use ffmpeg to extract compressed mono audio, then transcribe the smaller audio file:\n\n```bash\nffmpeg -y -i input.mp4 -vn -ac 1 -ar 16000 -b:a 32k audio_for_stt.mp3\n```\n\nPreserve timestamp information as much as possible. Prefer SRT format. If STT only produces plain text, create `transcript.txt` and clearly note in the final output that exact subtitle timing is unavailable.\n\n### 4. Extract timestamped candidate frames with ffmpeg\n\nAfter confirming the video path, use `scripts/prepare_video_learning_assets.py`. The script generates timestamped candidate screenshots and a manifest file:\n\n```bash\npython3 \"$SKILL_DIR/scripts/prepare_video_learning_assets.py\" \\\n  --video /path/to/video.mp4 \\\n  --out /path/to/workspace \\\n  --scene-threshold 0.3\n```\n\nBy default, the script extracts frames only from ffmpeg scene changes; it does not take one screenshot every 30 seconds. Use `--interval <seconds>` only when regular interval screenshots are explicitly needed.\n\nFor most learning videos, the recommended `--scene-threshold` range is `0.1`–`0.3`:\n\n- Lower thresholds produce more frames and capture smaller visual changes.\n- Higher thresholds produce fewer frames and keep only more obvious scene changes.\n- After running the script, check the frame count in `frames_manifest.json` and adjust the threshold so the number of candidate frames is suitable for manual review.\n\nThe script writes:\n\n- `frames/frame_000001__HH-MM-SS.jpg`\n- `frames_manifest.json`\n- `video_learning_notes.skeleton.md`\n\nIf scene detection misses important content, add `--interval <seconds>` as a supplement. Use 10–15 seconds for slide-heavy or fast-changing instructional videos, and 45–60 seconds for talking-head videos.\n\n### 5. Read and filter key frames one by one with STT context\n\nUse the Read tool's visual analysis capability to inspect extracted screenshots. You must check candidate frames one by one in chronological order, and decide whether each frame is a key frame by combining the image content with nearby STT/SRT text.\n\nFor each candidate image:\n\n1. Read the image file content with vision; do not judge only from the filename or timestamp.\n2. Locate the subtitles around the same timestamp in `transcript.srt`, usually the preceding and following 15–30 seconds.\n3. Decide whether the image provides learning value beyond the transcript.\n4. Keep the frame as a key screenshot only when it helps explain, supplement, or preserve important information.\n\nPrioritize frames containing:\n\n- Slides, diagrams, charts, code, formulas, tables, whiteboard content, UI operation screens, or definitions.\n- Scene changes that introduce a new topic.\n- Visual examples explicitly referenced by the subtitles.\n- Important on-screen text not fully captured by STT.\n\nSkip frames that are:\n\n- Near-duplicates.\n- Blurry.\n- Pure talking-head shots without useful learning information.\n- Loading screens, ads, intros/outros, or irrelevant overlays.\n\nCopy selected key images into `selected_frames/`, preserving the original timestamped filenames. Keep only enough screenshots to support learning; do not keep every candidate frame. For most videos, 8–30 screenshots are enough. Use more only when the video is highly visual.\n\n### 6. Generate the learning note from screenshots and SRT\n\nRead `transcript.srt`, the selected screenshot filenames/timestamps, and the visual notes produced while reading images one by one. Create `video_learning_notes.md` only after the key-frame selection step is complete.\n\nWhen writing, embed key screenshots into the corresponding time-based sections, and explain why each screenshot matters based on both the STT context and the image content. Use the following structure:\n\n```markdown\n# <Video title or topic>\n\n<video src=\"relative/path/to/video.mp4\" controls ></video>\n\n- Original video: <URL>\n- Video file: [<video-title>](relative/path/to/video.mp4)\n- Generated date: YYYY-MM-DD\n\n## Core Summary\n\n<Summarize the most important ideas in 5–10 bullet points.>\n\n## Learning Objectives\n\n<List the concepts or operations the learner should understand after watching.>\n\n## Section-by-Section Notes\n\n### 00:00:00–00:03:20 <Section title>\n\n<Convert the subtitles into readable learning notes. Do not dump the raw transcript.>\n\n![00:01:30 key screenshot](selected_frames/frame_000003__00-01-30.jpg)\n\n## Key Concepts / Terms\n\n| Term | Explanation | Timestamp |\n|---|---|---|\n\n## Steps or Methodology\n\n<If the video is a tutorial, organize the steps. If it is a course, organize the conceptual framework.>\n\n## Review Checklist\n\n- [ ] <Question or checkpoint>\n```\n\n### Writing standards\n\n- Write in Chinese unless the user asks otherwise.\n- Convert the transcript into structured learning notes; do not dump a raw transcript.\n- Include screenshot timestamps in captions or nearby text.\n- At the beginning of the Markdown, embed the source video with the same syntax style as an image: `<video src=\"relative/path/to/video.mp4\" controls ></video>`.\n- Use relative paths for both the source video and screenshots so the Markdown still displays correctly when the folder is moved.\n- Ground all claims in the subtitle content or image content. Use wording such as “possibly” or “appears to” for uncertain visual interpretations.\n- If the video contains code, formulas, financial charts, UI operations, or domain-specific terminology, preserve readable on-screen text as accurately as possible.\n\n## Tool notes\n\n- Use Bash for `yt-dlp-downloader`, `ffmpeg`, `ffprobe`, qwen-audio/STT commands, and necessary file operations.\n- Use Read with vision to review images; do not rely only on filenames or timestamps.\n- Use the bundled script for deterministic candidate frame extraction and skeleton generation; the final Markdown should be manually organized after subtitle and image analysis.\n- At the end of the task, use Message `files_preview` when appropriate to show the final Markdown and output directory.\n\nFile v1.0.2:_meta.json\n\n{\n  \"ownerId\": \"kn7acsrxymzj0gf0zr5t820f9x82644f\",\n  \"slug\": \"video-learning-notes\",\n  \"version\": \"1.0.2\",\n  \"publishedAt\": 1779206102074\n}\n\nFile v1.0.2:references/api_reference.md\n\n# Video Learning Notes Reference\n\n## Dependency expectations\n\n- `yd-dlp-downloader` is responsible for authenticated/video-site downloads.\n- `ffmpeg` and `ffprobe` are required for audio extraction, duration probing, scene detection, and screenshots.\n- qwen-audio/STT should produce `transcript.srt` whenever timestamps are available.\n- The Read tool should be used on extracted frame images to decide which screenshots are educationally useful.\n\n## Frame selection heuristics\n\nPrioritize slides, diagrams, formulas, charts, tables, code, UI operation screens, whiteboard content, and any timestamp where the visual content adds information missing from transcript text.\n\nAvoid selecting frames that are blurry, duplicated, purely decorative, ads/intros/outros, or only show a speaker with no relevant visual information.\n\nFile v1.0.2:skill-card.md\n\n## Description:\n\nConverts a video URL or local video file into a Chinese Markdown learning note by downloading and transcribing media, extracting timestamped frames, selecting key screenshots, and organizing section-by-section notes.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[darknoah](https://clawhub.ai/user/darknoah)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nExternal users, educators, learners, and developers use this skill to turn authorized online or local videos into illustrated study notes with transcripts, selected screenshots, and review structure.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The workflow downloads online media and writes video, transcript, frame, and note files locally.\n\nMitigation: Run it in a dedicated workspace and process only videos the user is authorized to access.\n\nRisk: Using logged-in browser cookies for video downloads can expose sensitive session material.\n\nMitigation: Require explicit user approval before using browser cookies; prefer public URLs, manual downloads, dedicated browser profiles, or narrowly exported cookies.\n\nRisk: Generated notes and screenshot interpretations can be incomplete or misleading if transcription or visual review misses context.\n\nMitigation: Review the transcript, selected frames, and final Markdown before relying on the notes.\n\n## Reference(s):\n\n- [Video Learning Notes Reference](references/api_reference.md)\n- [ClawHub skill page](https://clawhub.ai/darknoah/skills/video-learning-notes)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, shell commands, configuration]\n\n**Output Format:** [Markdown learning note with local video, SRT transcript, timestamped image files, and JSON frame manifest]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Uses relative paths, timestamped screenshots, and Chinese prose by default unless the user asks otherwise.]\n\n## Skill Version(s):\n\n1.0.2 (source: server release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v1.0.1: 4 files, 7756 bytes\n\nFiles: references/api_reference.md (822b), scripts/prepare_video_learning_assets.py (6296b), SKILL.md (10204b), _meta.json (139b)\n\nFile v1.0.1:SKILL.md\n\n---\nname: video-learning-notes\ndescription: Use this skill when the user provides a video URL and wants a complete Markdown learning note. It downloads the original video, transcribes audio with qwen-audio/STT, extracts timestamped frames with ffmpeg, reads and filters key screenshots one by one in combination with subtitles, and finally generates an illustrated learning note.\n---\n\n# Video Learning Notes\n\n## Overview\n\nConvert a video URL or local video file into a complete Markdown learning note. The note should be structured from the STT subtitle content and include selected key screenshots. Use this skill for requests such as “turn this video into learning notes”, “download this video and transcribe/analyze it”, or similar video-to-learning-note tasks.\n\n## Required Output\n\nCreate a self-contained output directory containing:\n\n- The downloaded original video file.\n- `transcript.srt` generated by qwen-audio/STT.\n- Timestamped frames extracted by ffmpeg under `frames/`.\n- Manually selected key screenshots under `selected_frames/`.\n- The final Markdown file, usually named `video_learning_notes.md`, using relative paths for the source video and images, and citing screenshot timestamps.\n\n## Workflow\n\n### 1. Create a workspace\n\nCreate a dedicated output directory for each video note. Prefer the current task directory or a stable path such as `./<note-title>/`. Keep all generated files inside this directory; do not scatter outputs into shared default folders.\n\n### 2. Download the original video\n\nIf the source is an online video, use the `yt-dlp-downloader` skill/workflow to download the user-provided video URL. Preserve the original or best available quality when possible, and write the video into the current workspace.\n\nCheck dependencies when needed before downloading:\n\n```bash\nwhich yt-dlp || echo \"yt-dlp not installed. Install with: pip install yt-dlp\"\nwhich ffmpeg || echo \"ffmpeg not installed. Install with: brew install ffmpeg\"\n```\n\nRecommended commands:\n\n```bash\n# Generic: download best quality into the workspace\nyt-dlp -P \"/path/to/workspace\" -o \"%(title)s.%(ext)s\" \"VIDEO_URL\"\n\n# YouTube: use browser cookies by default to reduce 403 errors\nyt-dlp -P \"/path/to/workspace\" --cookies-from-browser chrome -o \"%(title)s.%(ext)s\" \"YOUTUBE_URL\"\n\n# Download subtitles when available; still run qwen-audio/STT unless the user only wants official subtitles\nyt-dlp -P \"/path/to/workspace\" --write-subs --sub-langs all -o \"%(title)s.%(ext)s\" \"VIDEO_URL\"\n```\n\nPlatform handling principles:\n\n- YouTube / YouTube Music: use `--cookies-from-browser chrome` by default. Supported browser cookie sources include `chrome`, `firefox`, `safari`, `edge`, `brave`, and `opera`.\n- Bilibili, Twitter/X, TikTok, Douyin, Vimeo, Twitch, and most other platforms: try direct download first.\n- Playlist URLs: ask the user whether to process the entire playlist, one specific video, or a specific range.\n- Quality selection: default to the best available quality. If the user specifies a quality, use format selectors such as `bestvideo[height<=1080]+bestaudio/best[height<=1080]`.\n\nAfter downloading, identify the actual video file path, such as `.mp4`, `.mkv`, `.mov`, `.webm`, etc. If multiple files are produced, choose the main video as the source for the learning note, while keeping subtitles, thumbnails, and other files as supporting assets.\n\nTroubleshooting:\n\n- `HTTP 403 Forbidden`: retry with `--cookies-from-browser chrome` or another browser where the user is logged in.\n- `Video unavailable`, private videos, or geo-restricted videos: ask the user for login access, cookies, or an accessible environment; do not bypass access restrictions.\n- `Format not available`: run `yt-dlp -F \"VIDEO_URL\"` to list available formats, then choose one.\n- Interrupted downloads: retry; yt-dlp can usually resume partial downloads.\n- `yt-dlp: command not found`: install `yt-dlp` or ask the user to install it.\n\nIf `yt-dlp-downloader` / `yt-dlp` is unavailable, or if the video requires login/authentication, stop and ask the user to provide the missing access requirement instead of silently switching to unreliable tools.\n\n### 3. Transcribe with qwen-audio\n\nRun qwen-audio/STT on the downloaded video or extracted audio, and save the result as `transcript.srt`.\n\nFor large videos, first use ffmpeg to extract compressed mono audio, then transcribe the smaller audio file:\n\n```bash\nffmpeg -y -i input.mp4 -vn -ac 1 -ar 16000 -b:a 32k audio_for_stt.mp3\n```\n\nPreserve timestamp information as much as possible. Prefer SRT format. If STT only produces plain text, create `transcript.txt` and clearly note in the final output that exact subtitle timing is unavailable.\n\n### 4. Extract timestamped candidate frames with ffmpeg\n\nAfter confirming the video path, use `scripts/prepare_video_learning_assets.py`. The script generates timestamped candidate screenshots and a manifest file:\n\n```bash\npython3 \"$SKILL_DIR/scripts/prepare_video_learning_assets.py\" \\\n  --video /path/to/video.mp4 \\\n  --out /path/to/workspace \\\n  --scene-threshold 0.3\n```\n\nBy default, the script extracts frames only from ffmpeg scene changes; it does not take one screenshot every 30 seconds. Use `--interval <seconds>` only when regular interval screenshots are explicitly needed.\n\nFor most learning videos, the recommended `--scene-threshold` range is `0.1`–`0.3`:\n\n- Lower thresholds produce more frames and capture smaller visual changes.\n- Higher thresholds produce fewer frames and keep only more obvious scene changes.\n- After running the script, check the frame count in `frames_manifest.json` and adjust the threshold so the number of candidate frames is suitable for manual review.\n\nThe script writes:\n\n- `frames/frame_000001__HH-MM-SS.jpg`\n- `frames_manifest.json`\n- `video_learning_notes.skeleton.md`\n\nIf scene detection misses important content, add `--interval <seconds>` as a supplement. Use 10–15 seconds for slide-heavy or fast-changing instructional videos, and 45–60 seconds for talking-head videos.\n\n### 5. Read and filter key frames one by one with STT context\n\nUse the Read tool's visual analysis capability to inspect extracted screenshots. You must check candidate frames one by one in chronological order, and decide whether each frame is a key frame by combining the image content with nearby STT/SRT text.\n\nFor each candidate image:\n\n1. Read the image file content with vision; do not judge only from the filename or timestamp.\n2. Locate the subtitles around the same timestamp in `transcript.srt`, usually the preceding and following 15–30 seconds.\n3. Decide whether the image provides learning value beyond the transcript.\n4. Keep the frame as a key screenshot only when it helps explain, supplement, or preserve important information.\n\nPrioritize frames containing:\n\n- Slides, diagrams, charts, code, formulas, tables, whiteboard content, UI operation screens, or definitions.\n- Scene changes that introduce a new topic.\n- Visual examples explicitly referenced by the subtitles.\n- Important on-screen text not fully captured by STT.\n\nSkip frames that are:\n\n- Near-duplicates.\n- Blurry.\n- Pure talking-head shots without useful learning information.\n- Loading screens, ads, intros/outros, or irrelevant overlays.\n\nCopy selected key images into `selected_frames/`, preserving the original timestamped filenames. Keep only enough screenshots to support learning; do not keep every candidate frame. For most videos, 8–30 screenshots are enough. Use more only when the video is highly visual.\n\n### 6. Generate the learning note from screenshots and SRT\n\nRead `transcript.srt`, the selected screenshot filenames/timestamps, and the visual notes produced while reading images one by one. Create `video_learning_notes.md` only after the key-frame selection step is complete.\n\nWhen writing, embed key screenshots into the corresponding time-based sections, and explain why each screenshot matters based on both the STT context and the image content. Use the following structure:\n\n```markdown\n# <Video title or topic>\n\n![<title>](relative/path/to/video.mp4)\n\n- Original video: <URL>\n- Video file: `relative/path/to/video.mp4`\n- Transcript: `transcript.srt`\n- Generated date: YYYY-MM-DD\n\n## Core Summary\n\n<Summarize the most important ideas in 5–10 bullet points.>\n\n## Learning Objectives\n\n<List the concepts or operations the learner should understand after watching.>\n\n## Section-by-Section Notes\n\n### 00:00:00–00:03:20 <Section title>\n\n<Convert the subtitles into readable learning notes. Do not dump the raw transcript.>\n\n![00:01:30 key screenshot](selected_frames/frame_000003__00-01-30.jpg)\n\n## Key Concepts / Terms\n\n| Term | Explanation | Timestamp |\n|---|---|---|\n\n## Steps or Methodology\n\n<If the video is a tutorial, organize the steps. If it is a course, organize the conceptual framework.>\n\n## Review Checklist\n\n- [ ] <Question or checkpoint>\n```\n\n### Writing standards\n\n- Write in Chinese unless the user asks otherwise.\n- Convert the transcript into structured learning notes; do not dump a raw transcript.\n- Include screenshot timestamps in captions or nearby text.\n- At the beginning of the Markdown, embed the source video with the same syntax style as an image: `![<title>](relative/path/to/video.mp4)`.\n- Use relative paths for both the source video and screenshots so the Markdown still displays correctly when the folder is moved.\n- Ground all claims in the subtitle content or image content. Use wording such as “possibly” or “appears to” for uncertain visual interpretations.\n- If the video contains code, formulas, financial charts, UI operations, or domain-specific terminology, preserve readable on-screen text as accurately as possible.\n\n## Tool notes\n\n- Use Bash for `yt-dlp-downloader`, `ffmpeg`, `ffprobe`, qwen-audio/STT commands, and necessary file operations.\n- Use Read with vision to review images; do not rely only on filenames or timestamps.\n- Use the bundled script for deterministic candidate frame extraction and skeleton generation; the final Markdown should be manually organized after subtitle and image analysis.\n- At the end of the task, use Message `files_preview` when appropriate to show the final Markdown and output directory.\n\nFile v1.0.1:_meta.json\n\n{\n  \"ownerId\": \"kn7acsrxymzj0gf0zr5t820f9x82644f\",\n  \"slug\": \"video-learning-notes\",\n  \"version\": \"1.0.1\",\n  \"publishedAt\": 1779204292550\n}\n\nFile v1.0.1:references/api_reference.md\n\n# Video Learning Notes Reference\n\n## Dependency expectations\n\n- `yd-dlp-downloader` is responsible for authenticated/video-site downloads.\n- `ffmpeg` and `ffprobe` are required for audio extraction, duration probing, scene detection, and screenshots.\n- qwen-audio/STT should produce `transcript.srt` whenever timestamps are available.\n- The Read tool should be used on extracted frame images to decide which screenshots are educationally useful.\n\n## Frame selection heuristics\n\nPrioritize slides, diagrams, formulas, charts, tables, code, UI operation screens, whiteboard content, and any timestamp where the visual content adds information missing from transcript text.\n\nAvoid selecting frames that are blurry, duplicated, purely decorative, ads/intros/outros, or only show a speaker with no relevant visual information.\n\nArchive v1.0.0: 4 files, 7673 bytes\n\nFiles: references/api_reference.md (822b), scripts/prepare_video_learning_assets.py (6242b), SKILL.md (9960b), _meta.json (139b)\n\nFile v1.0.0:SKILL.md\n\n---\nname: video-learning-notes\ndescription: Use this skill when the user provides a video URL and wants a complete Markdown learning note. It downloads the original video, transcribes audio with qwen-audio/STT, extracts timestamped frames with ffmpeg, reads and filters key screenshots one by one in combination with subtitles, and finally generates an illustrated learning note.\n---\n\n# Video Learning Notes\n\n## Overview\n\nConvert a video URL or local video file into a complete Markdown learning note. The note should be structured from the STT subtitle content and include selected key screenshots. Use this skill for requests such as “turn this video into learning notes”, “download this video and transcribe/analyze it”, or similar video-to-learning-note tasks.\n\n## Required Output\n\nCreate a self-contained output directory containing:\n\n- The downloaded original video file.\n- `transcript.srt` generated by qwen-audio/STT.\n- Timestamped frames extracted by ffmpeg under `frames/`.\n- Manually selected key screenshots under `selected_frames/`.\n- The final Markdown file, usually named `video_learning_notes.md`, using relative image paths and citing screenshot timestamps.\n\n## Workflow\n\n### 1. Create a workspace\n\nCreate a dedicated output directory for each video note. Prefer the current task directory or a stable path such as `./<note-title>/`. Keep all generated files inside this directory; do not scatter outputs into shared default folders.\n\n### 2. Download the original video\n\nIf the source is an online video, use the `yt-dlp-downloader` skill/workflow to download the user-provided video URL. Preserve the original or best available quality when possible, and write the video into the current workspace.\n\nCheck dependencies when needed before downloading:\n\n```bash\nwhich yt-dlp || echo \"yt-dlp not installed. Install with: pip install yt-dlp\"\nwhich ffmpeg || echo \"ffmpeg not installed. Install with: brew install ffmpeg\"\n```\n\nRecommended commands:\n\n```bash\n# Generic: download best quality into the workspace\nyt-dlp -P \"/path/to/workspace\" -o \"%(title)s.%(ext)s\" \"VIDEO_URL\"\n\n# YouTube: use browser cookies by default to reduce 403 errors\nyt-dlp -P \"/path/to/workspace\" --cookies-from-browser chrome -o \"%(title)s.%(ext)s\" \"YOUTUBE_URL\"\n\n# Download subtitles when available; still run qwen-audio/STT unless the user only wants official subtitles\nyt-dlp -P \"/path/to/workspace\" --write-subs --sub-langs all -o \"%(title)s.%(ext)s\" \"VIDEO_URL\"\n```\n\nPlatform handling principles:\n\n- YouTube / YouTube Music: use `--cookies-from-browser chrome` by default. Supported browser cookie sources include `chrome`, `firefox`, `safari`, `edge`, `brave`, and `opera`.\n- Bilibili, Twitter/X, TikTok, Douyin, Vimeo, Twitch, and most other platforms: try direct download first.\n- Playlist URLs: ask the user whether to process the entire playlist, one specific video, or a specific range.\n- Quality selection: default to the best available quality. If the user specifies a quality, use format selectors such as `bestvideo[height<=1080]+bestaudio/best[height<=1080]`.\n\nAfter downloading, identify the actual video file path, such as `.mp4`, `.mkv`, `.mov`, `.webm`, etc. If multiple files are produced, choose the main video as the source for the learning note, while keeping subtitles, thumbnails, and other files as supporting assets.\n\nTroubleshooting:\n\n- `HTTP 403 Forbidden`: retry with `--cookies-from-browser chrome` or another browser where the user is logged in.\n- `Video unavailable`, private videos, or geo-restricted videos: ask the user for login access, cookies, or an accessible environment; do not bypass access restrictions.\n- `Format not available`: run `yt-dlp -F \"VIDEO_URL\"` to list available formats, then choose one.\n- Interrupted downloads: retry; yt-dlp can usually resume partial downloads.\n- `yt-dlp: command not found`: install `yt-dlp` or ask the user to install it.\n\nIf `yt-dlp-downloader` / `yt-dlp` is unavailable, or if the video requires login/authentication, stop and ask the user to provide the missing access requirement instead of silently switching to unreliable tools.\n\n### 3. Transcribe with qwen-audio\n\nRun qwen-audio/STT on the downloaded video or extracted audio, and save the result as `transcript.srt`.\n\nFor large videos, first use ffmpeg to extract compressed mono audio, then transcribe the smaller audio file:\n\n```bash\nffmpeg -y -i input.mp4 -vn -ac 1 -ar 16000 -b:a 32k audio_for_stt.mp3\n```\n\nPreserve timestamp information as much as possible. Prefer SRT format. If STT only produces plain text, create `transcript.txt` and clearly note in the final output that exact subtitle timing is unavailable.\n\n### 4. Extract timestamped candidate frames with ffmpeg\n\nAfter confirming the video path, use `scripts/prepare_video_learning_assets.py`. The script generates timestamped candidate screenshots and a manifest file:\n\n```bash\npython3 \"$SKILL_DIR/scripts/prepare_video_learning_assets.py\" \\\n  --video /path/to/video.mp4 \\\n  --out /path/to/workspace \\\n  --scene-threshold 0.3\n```\n\nBy default, the script extracts frames only from ffmpeg scene changes; it does not take one screenshot every 30 seconds. Use `--interval <seconds>` only when regular interval screenshots are explicitly needed.\n\nFor most learning videos, the recommended `--scene-threshold` range is `0.1`–`0.3`:\n\n- Lower thresholds produce more frames and capture smaller visual changes.\n- Higher thresholds produce fewer frames and keep only more obvious scene changes.\n- After running the script, check the frame count in `frames_manifest.json` and adjust the threshold so the number of candidate frames is suitable for manual review.\n\nThe script writes:\n\n- `frames/frame_000001__HH-MM-SS.jpg`\n- `frames_manifest.json`\n- `video_learning_notes.skeleton.md`\n\nIf scene detection misses important content, add `--interval <seconds>` as a supplement. Use 10–15 seconds for slide-heavy or fast-changing instructional videos, and 45–60 seconds for talking-head videos.\n\n### 5. Read and filter key frames one by one with STT context\n\nUse the Read tool's visual analysis capability to inspect extracted screenshots. You must check candidate frames one by one in chronological order, and decide whether each frame is a key frame by combining the image content with nearby STT/SRT text.\n\nFor each candidate image:\n\n1. Read the image file content with vision; do not judge only from the filename or timestamp.\n2. Locate the subtitles around the same timestamp in `transcript.srt`, usually the preceding and following 15–30 seconds.\n3. Decide whether the image provides learning value beyond the transcript.\n4. Keep the frame as a key screenshot only when it helps explain, supplement, or preserve important information.\n\nPrioritize frames containing:\n\n- Slides, diagrams, charts, code, formulas, tables, whiteboard content, UI operation screens, or definitions.\n- Scene changes that introduce a new topic.\n- Visual examples explicitly referenced by the subtitles.\n- Important on-screen text not fully captured by STT.\n\nSkip frames that are:\n\n- Near-duplicates.\n- Blurry.\n- Pure talking-head shots without useful learning information.\n- Loading screens, ads, intros/outros, or irrelevant overlays.\n\nCopy selected key images into `selected_frames/`, preserving the original timestamped filenames. Keep only enough screenshots to support learning; do not keep every candidate frame. For most videos, 8–30 screenshots are enough. Use more only when the video is highly visual.\n\n### 6. Generate the learning note from screenshots and SRT\n\nRead `transcript.srt`, the selected screenshot filenames/timestamps, and the visual notes produced while reading images one by one. Create `video_learning_notes.md` only after the key-frame selection step is complete.\n\nWhen writing, embed key screenshots into the corresponding time-based sections, and explain why each screenshot matters based on both the STT context and the image content. Use the following structure:\n\n```markdown\n# <Video title or topic>\n\n- Original video: <URL>\n- Video file: `relative/path/to/video.mp4`\n- Transcript: `transcript.srt`\n- Generated date: YYYY-MM-DD\n\n## Core Summary\n\n<Summarize the most important ideas in 5–10 bullet points.>\n\n## Learning Objectives\n\n<List the concepts or operations the learner should understand after watching.>\n\n## Section-by-Section Notes\n\n### 00:00:00–00:03:20 <Section title>\n\n<Convert the subtitles into readable learning notes. Do not dump the raw transcript.>\n\n![00:01:30 key screenshot](selected_frames/frame_000003__00-01-30.jpg)\n\n## Key Concepts / Terms\n\n| Term | Explanation | Timestamp |\n|---|---|---|\n\n## Steps or Methodology\n\n<If the video is a tutorial, organize the steps. If it is a course, organize the conceptual framework.>\n\n## Review Checklist\n\n- [ ] <Question or checkpoint>\n```\n\n### Writing standards\n\n- Write in Chinese unless the user asks otherwise.\n- Convert the transcript into structured learning notes; do not dump a raw transcript.\n- Include screenshot timestamps in captions or nearby text.\n- Use relative image paths so the Markdown still displays correctly when the folder is moved.\n- Ground all claims in the subtitle content or image content. Use wording such as “possibly” or “appears to” for uncertain visual interpretations.\n- If the video contains code, formulas, financial charts, UI operations, or domain-specific terminology, preserve readable on-screen text as accurately as possible.\n\n## Tool notes\n\n- Use Bash for `yt-dlp-downloader`, `ffmpeg`, `ffprobe`, qwen-audio/STT commands, and necessary file operations.\n- Use Read with vision to review images; do not rely only on filenames or timestamps.\n- Use the bundled script for deterministic candidate frame extraction and skeleton generation; the final Markdown should be manually organized after subtitle and image analysis.\n- At the end of the task, use Message `files_preview` when appropriate to show the final Markdown and output directory.\n\nFile v1.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn7acsrxymzj0gf0zr5t820f9x82644f\",\n  \"slug\": \"video-learning-notes\",\n  \"version\": \"1.0.0\",\n  \"publishedAt\": 1779195257805\n}\n\nFile v1.0.0:references/api_reference.md\n\n# Video Learning Notes Reference\n\n## Dependency expectations\n\n- `yd-dlp-downloader` is responsible for authenticated/video-site downloads.\n- `ffmpeg` and `ffprobe` are required for audio extraction, duration probing, scene detection, and screenshots.\n- qwen-audio/STT should produce `transcript.srt` whenever timestamps are available.\n- The Read tool should be used on extracted frame images to decide which screenshots are educationally useful.\n\n## Frame selection heuristics\n\nPrioritize slides, diagrams, formulas, charts, tables, code, UI operation screens, whiteboard content, and any timestamp where the visual content adds information missing from transcript text.\n\nAvoid selecting frames that are blurry, duplicated, purely decorative, ads/intros/outros, or only show a speaker with no relevant visual information.","readmeExcerpt":"Skill: 视频自动笔记制作 Owner: darknoah Summary: Use this skill when the user provides a video URL and wants a complete Markdown learning note. It downloads the original video, transcribes audio with qwen-a... Tags: latest:1.0.2 Version history: v1.0.2 | 2026-05-19T15:55:02.074Z | user No user-facing or workflow changes in this version. - No files were changed. - Documentation, functionality, and output remain the same as th","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"which yt-dlp || echo \"yt-dlp not installed. Install with: pip install yt-dlp\"\nwhich ffmpeg || echo \"ffmpeg not installed. Install with: brew install ffmpeg\""},{"language":"bash","snippet":"# Generic: download best quality into the workspace\nyt-dlp -P \"/path/to/workspace\" -o \"%(title)s.%(ext)s\" \"VIDEO_URL\"\n\n# YouTube: use browser cookies by default to reduce 403 errors\nyt-dlp -P \"/path/to/workspace\" --cookies-from-browser chrome -o \"%(title)s.%(ext)s\" \"YOUTUBE_URL\"\n\n# Download subtitles when available; still run qwen-audio/STT unless the user only wants official subtitles\nyt-dlp -P \"/path/to/workspace\" --write-subs --sub-langs all -o \"%(title)s.%(ext)s\" \"VIDEO_URL\""},{"language":"bash","snippet":"ffmpeg -y -i input.mp4 -vn -ac 1 -ar 16000 -b:a 32k audio_for_stt.mp3"},{"language":"bash","snippet":"python3 \"$SKILL_DIR/scripts/prepare_video_learning_assets.py\" \\\n  --video /path/to/video.mp4 \\\n  --out /path/to/workspace \\\n  --scene-threshold 0.3"},{"language":"markdown","snippet":"# <Video title or topic>\n\n<video src=\"relative/path/to/video.mp4\" controls ></video>\n\n- Original video: <URL>\n- Video file: [<video-title>](relative/path/to/video.mp4)\n- Generated date: YYYY-MM-DD\n\n## Core Summary\n\n<Summarize the most important ideas in 5–10 bullet points.>\n\n## Learning Objectives\n\n<List the concepts or operations the learner should understand after watching.>\n\n## Section-by-Section Notes\n\n### 00:00:00–00:03:20 <Section title>\n\n<Convert the subtitles into readable learning notes. Do not dump the raw transcript.>\n\n![00:01:30 key screenshot](selected_frames/frame_000003__00-01-30.jpg)\n\n## Key Concepts / Terms\n\n| Term | Explanation | Timestamp |\n|---|---|---|\n\n## Steps or Methodology\n\n<If the video is a tutorial, organize the steps. If it is a course, organize the conceptual framework.>\n\n## Review Checklist\n\n- [ ] <Question or checkpoint>"},{"language":"bash","snippet":"which yt-dlp || echo \"yt-dlp not installed. Install with: pip install yt-dlp\"\nwhich ffmpeg || echo \"ffmpeg not installed. Install with: brew install ffmpeg\""}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: video-learning-notes\ndescription: Use this skill when the user provides a video URL and wants a complete Markdown learning note. It downloads the original video, transcribes audio with qwen-audio/STT, extracts timestamped frames with ffmpeg, reads and filters key screenshots one by one in combination with subtitles, and finally generates an illustrated learning note.\n---\n\n# Video Learning Notes\n\n## Overview\n\nConvert a video URL or local video file into a complete Markdown learning note. The note should be structured from the STT subtitle content and include selected key screenshots. Use this skill for requests such as “turn this video into learning notes”, “download this video and transcribe/analyze it”, or similar video-to-learning-note tasks.\n\n## Required Output\n\nCreate a self-contained output directory containing:\n\n- The downloaded original video file.\n- `transcript.srt` generated by qwen-audio/STT.\n- Timestamped frames extracted by ffmpeg under `frames/`.\n- Manually selected key screenshots under `selected_frames/`.\n- The final Markdown file, usually named `video_learning_notes.md`, using relative paths for the source video and images, and citing screenshot timestamps.\n\n## Workflow\n\n### 1. Create a workspace\n\nCreate a dedicated output directory for each video note. Prefer the current task directory or a stable path such as `./<note-title>/`. Keep all generated files inside this directory; do not scatter outputs into shared default folders.\n\n### 2. Download the original video\n\nIf the source is an online video, use the `yt-dlp-downloader` skill/workflow to download the user-provided video URL. Preserve the original or best available quality when possible, and write the video into the current workspace.\n\nCheck dependencies when needed before downloading:\n\n```bash\nwhich yt-dlp || echo \"yt-dlp not installed. Install with: pip install yt-dlp\"\nwhich ffmpeg || echo \"ffmpeg not installed. Install with: brew install ffmpeg\"\n```\n\nRecommended commands:\n\n```bash\n# Generic: download best quality into the workspace\nyt-dlp -P \"/path/to/workspace\" -o \"%(title)s.%(ext)s\" \"VIDEO_URL\"\n\n# YouTube: use browser cookies by default to reduce 403 errors\nyt-dlp -P \"/path/to/workspace\" --cookies-from-browser chrome -o \"%(title)s.%(ext)s\" \"YOUTUBE_URL\"\n\n# Download subtitles when available; still run qwen-audio/STT unless the user only wants official subtitles\nyt-dlp -P \"/path/to/workspace\" --write-subs --sub-langs all -o \"%(title)s.%(ext)s\" \"VIDEO_URL\"\n```\n\nPlatform handling principles:\n\n- YouTube / YouTube Music: use `--cookies-from-browser chrome` by default. Supported browser cookie sources include `chrome`, `firefox`, `safari`, `edge`, `brave`, and `opera`.\n- Bilibili, Twitter/X, TikTok, Douyin, Vimeo, Twitch, and most other platforms: try direct download first.\n- Playlist URLs: ask the user whether to process the entire playlist, one specific video, or a specific range.\n- Quality selection: default to the best available quality. If the user specifies a qua"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7acsrxymzj0gf0zr5t820f9x82644f\",\n  \"slug\": \"video-learning-notes\",\n  \"version\": \"1.0.2\",\n  \"publishedAt\": 1779206102074\n}"},{"path":"references/api_reference.md","content":"# Video Learning Notes Reference\n\n## Dependency expectations\n\n- `yd-dlp-downloader` is responsible for authenticated/video-site downloads.\n- `ffmpeg` and `ffprobe` are required for audio extraction, duration probing, scene detection, and screenshots.\n- qwen-audio/STT should produce `transcript.srt` whenever timestamps are available.\n- The Read tool should be used on extracted frame images to decide which screenshots are educationally useful.\n\n## Frame selection heuristics\n\nPrioritize slides, diagrams, formulas, charts, tables, code, UI operation screens, whiteboard content, and any timestamp where the visual content adds information missing from transcript text.\n\nAvoid selecting frames that are blurry, duplicated, purely decorative, ads/intros/outros, or only show a speaker with no relevant visual information."},{"path":"skill-card.md","content":"## Description:\n\nConverts a video URL or local video file into a Chinese Markdown learning note by downloading and transcribing media, extracting timestamped frames, selecting key screenshots, and organizing section-by-section notes.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[darknoah](https://clawhub.ai/user/darknoah)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nExternal users, educators, learners, and developers use this skill to turn authorized online or local videos into illustrated study notes with transcripts, selected screenshots, and review structure.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The workflow downloads online media and writes video, transcript, frame, and note files locally.\n\nMitigation: Run it in a dedicated workspace and process only videos the user is authorized to access.\n\nRisk: Using logged-in browser cookies for video downloads can expose sensitive session material.\n\nMitigation: Require explicit user approval before using browser cookies; prefer public URLs, manual downloads, dedicated browser profiles, or narrowly exported cookies.\n\nRisk: Generated notes and screenshot interpretations can be incomplete or misleading if transcription or visual review misses context.\n\nMitigation: Review the transcript, selected frames, and final Markdown before relying on the notes.\n\n## Reference(s):\n\n- [Video Learning Notes Reference](references/api_reference.md)\n- [ClawHub skill page](https://clawhub.ai/darknoah/skills/video-learning-notes)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, shell commands, configuration]\n\n**Output Format:** [Markdown learning note with local video, SRT transcript, timestamped image files, and JSON frame manifest]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Uses relative paths, timestamped screenshots, and Chinese prose by default unless the user asks otherwise.]\n\n## Skill Version(s):\n\n1.0.2 (source: server release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Use this skill when the user provides a video URL and wants a complete Markdown learning note. It downloads the original video, transcribes audio with qwen-a... Skill: 视频自动笔记制作 Owner: darknoah Summary: Use this skill when the user provides a video URL and wants a complete Markdown learning note. It downloads the original video, transcribes audio with qwen-a... Tags: latest:1.0.2 Version history: v1.0.2 | 2026-05-19T15:55:02.074Z | user No user-facing or workflow changes in this version. - No files were changed. - Documentation, functionality, and output remain the same as th","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1241,"uniquenessScore":48,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-10T14:00:57.329Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-10T14:00:57.329Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T17:33:31.353Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}