{"id":"7e5ac5f6-bc84-4cfe-a21e-a78cbd2736ec","entityType":"agent","slug":"clawhub-forhonourlx-video-subtitle-extractor","name":"Video Subtitle Extractor","canonicalUrl":"https://www.xpersona.co/agent/clawhub-forhonourlx-video-subtitle-extractor","canonicalPath":"/agent/clawhub-forhonourlx-video-subtitle-extractor","generatedAt":"2026-10-11T01:47:10.712Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-10T22:42:44.197Z","emptyReason":null},"description":"Cross-platform video subtitle extraction using multi-engine ASR (speech-to-text). Downloads audio from video URLs via yt-dlp, transcribes with SenseVoice / w... Skill: Video Subtitle Extractor Owner: forhonourlx Summary: Cross-platform video subtitle extraction using multi-engine ASR (speech-to-text). Downloads audio from video URLs via yt-dlp, transcribes with SenseVoice / w... Tags: ai:1.0.0, asr:1.0.0, chinese:1.0.0, latest:2.0.0, major:2.0.0, subtitle:1.0.0, video:1.0.0, whisper:1.0.0 Version history: v2.0.0 | 2026-05-27T08:24:13.771Z | user Multi-engine ASR: SenseVoice","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.2K downloads reported by the source. Last updated 10/10/2026.","installCommand":"clawhub skill install s171scmakmb14cmkc061v2rkvh875fc9:video-subtitle-extractor","sourceUrl":"https://clawhub.ai/forhonourlx/video-subtitle-extractor","homepage":"https://clawhub.ai/forhonourlx/skills/video-subtitle-extractor","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/forhonourlx/video-subtitle-extractor","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/forhonourlx/skills/video-subtitle-extractor","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":62,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Cross-platform video subtitle extraction using multi-engine ASR (speech-to-text). Downloads audio from video URLs via yt-dlp, transcribes with SenseVoice / w..."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-10T22:42:44.197Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T22:42:44.197Z","emptyReason":null},"stars":null,"forks":null,"downloads":1239,"packageName":null,"latestVersion":"2.0.0","tractionLabel":"1.2K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T22:42:44.138Z","emptyReason":null},"lastUpdatedAt":"2026-10-10T22:42:44.197Z","lastCrawledAt":"2026-10-10T22:42:44.138Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-11T22:42:44.138Z","lastVerifiedAt":null,"highlights":[{"version":"2.0.0","createdAt":"2026-05-27T08:24:13.771Z","changelog":"Multi-engine ASR: SenseVoice Small default + whisper.cpp GGML + openai-whisper. Auto-backend detection. ~1.5GB RAM, 5x faster, ~96% Chinese accuracy.","fileCount":11,"zipByteSize":33519},{"version":"1.0.8","createdAt":"2026-05-26T03:12:19.002Z","changelog":"Add semiconductor/hardware domain calibration (30+ patterns); fix Douyin/TikTok cookies note","fileCount":11,"zipByteSize":29109},{"version":"1.0.7","createdAt":"2026-05-25T23:32:12.753Z","changelog":"--video-quality param + find_artifacts_in_dir","fileCount":10,"zipByteSize":27066},{"version":"1.0.6","createdAt":"2026-05-25T09:34:47.074Z","changelog":"New --save-video flag: download & persist full video (.mp4) as middleware. run.py and standalone download_audio.py both support it. Pipeline metadata now tracks video_path.","fileCount":10,"zipByteSize":25637},{"version":"1.0.5","createdAt":"2026-05-25T07:03:37.506Z","changelog":"New calibrate.py (rule-based, 80+ patterns), run.py --skip-download/--skip-transcribe/--calibrate flags, SKILL.md intermediate artifacts & reuse docs","fileCount":10,"zipByteSize":24857},{"version":"1.0.4","createdAt":"2026-05-24T14:24:17.978Z","changelog":"Remove small model (too poor accuracy), lock default to medium","fileCount":9,"zipByteSize":17568},{"version":"1.0.3","createdAt":"2026-05-22T06:23:21.421Z","changelog":"Improve: Performance benchmarks now show both backends separately","fileCount":9,"zipByteSize":19550},{"version":"1.0.2","createdAt":"2026-05-22T05:00:40.647Z","changelog":"v1.0.2: Xiaohongshu (小红书) platform support. New: 小红书 platform with xhslink.com short link auto-resolution. Improve: Platform table, benchmarks, Quick Start examples.","fileCount":9,"zipByteSize":17590}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s171scmakmb14cmkc061v2rkvh875fc9:video-subtitle-extractor","setupComplexity":"low","setupSteps":["Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-forhonourlx-video-subtitle-extractor/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-forhonourlx-video-subtitle-extractor/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-forhonourlx-video-subtitle-extractor/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-forhonourlx-video-subtitle-extractor/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-forhonourlx-video-subtitle-extractor/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-forhonourlx-video-subtitle-extractor/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T01:47:10.708Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-forhonourlx-video-subtitle-extractor/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-forhonourlx-video-subtitle-extractor/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-forhonourlx-video-subtitle-extractor/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-forhonourlx-video-subtitle-extractor/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-10T22:42:44.197Z","emptyReason":null},"readme":"Skill: Video Subtitle Extractor\n\nOwner: forhonourlx\n\nSummary: Cross-platform video subtitle extraction using multi-engine ASR (speech-to-text). Downloads audio from video URLs via yt-dlp, transcribes with SenseVoice / w...\n\nTags: ai:1.0.0, asr:1.0.0, chinese:1.0.0, latest:2.0.0, major:2.0.0, subtitle:1.0.0, video:1.0.0, whisper:1.0.0\n\nVersion history:\n\nv2.0.0 | 2026-05-27T08:24:13.771Z | user\n\nMulti-engine ASR: SenseVoice Small default + whisper.cpp GGML + openai-whisper. Auto-backend detection. ~1.5GB RAM, 5x faster, ~96% Chinese accuracy.\n\nv1.0.8 | 2026-05-26T03:12:19.002Z | user\n\nAdd semiconductor/hardware domain calibration (30+ patterns); fix Douyin/TikTok cookies note\n\nv1.0.7 | 2026-05-25T23:32:12.753Z | user\n\n--video-quality param + find_artifacts_in_dir\n\nv1.0.6 | 2026-05-25T09:34:47.074Z | user\n\nNew --save-video flag: download & persist full video (.mp4) as middleware. run.py and standalone download_audio.py both support it. Pipeline metadata now tracks video_path.\n\nv1.0.5 | 2026-05-25T07:03:37.506Z | user\n\nNew calibrate.py (rule-based, 80+ patterns), run.py --skip-download/--skip-transcribe/--calibrate flags, SKILL.md intermediate artifacts & reuse docs\n\nv1.0.4 | 2026-05-24T14:24:17.978Z | user\n\nRemove small model (too poor accuracy), lock default to medium\n\nv1.0.3 | 2026-05-22T06:23:21.421Z | user\n\nImprove: Performance benchmarks now show both backends separately\n\nv1.0.2 | 2026-05-22T05:00:40.647Z | user\n\nv1.0.2: Xiaohongshu (小红书) platform support. New: 小红书 platform with xhslink.com short link auto-resolution. Improve: Platform table, benchmarks, Quick Start examples.\n\nv1.0.1 | 2026-05-21T20:02:15.156Z | user\n\nFix ffmpeg PATH injection (root cause of SIGKILL-like failures on Windows), expand ffmpeg search paths 3→7 (winget/scoop/chocolatey/ProgramFiles(x86)), auto-detect GPU for fp16 support, verbose progress bar, better error messages\n\nv1.0.0 | 2026-05-21T11:06:57.883Z | user\n\nCross-platform ASR subtitle extraction pipeline. Auto-installs ffmpeg, yt-dlp, openai-whisper. Configurable models. Multi-format output.\n\nArchive index:\n\nArchive v2.0.0: 11 files, 33519 bytes\n\nFiles: LICENSE (917b), references/asr_models.md (1918b), references/calibration_guide.md (4619b), scripts/calibrate.py (12842b), scripts/download_audio.py (11069b), scripts/install_deps.py (4890b), scripts/run.py (13603b), scripts/transcribe.py (21167b), skill-card.md (2538b), SKILL.md (16895b), _meta.json (143b)\n\nFile v2.0.0:SKILL.md\n\n---\r\nname: video-subtitle-extractor\r\ndescription: |\r\n  Cross-platform video subtitle extraction using multi-engine ASR (speech-to-text). Downloads audio from video URLs via yt-dlp, transcribes with SenseVoice / whisper.cpp / openai-whisper (default: SenseVoice Small for Chinese), and applies LLM-based text calibration for Chinese financial/technical content. Use when: (1) extracting subtitles from Bilibili, Xiaohongshu, YouTube, or any yt-dlp-supported platform, (2) the video has no built-in subtitles, (3) users say \"下载字幕\", \"提取字幕\", \"语音转文字\", \"视频转文字\", \"字幕提取\", \"ASR转写\", (4) needing to transcribe audio files to text, (5) working with Chinese-language video content requiring high-accuracy transcription. Automatically handles dependency installation (ffmpeg, yt-dlp, ASR backends) and model downloads.\r\n---\r\n\r\n# Video Subtitle Extractor 🎬→📝\r\n\r\nCross-platform multi-engine ASR subtitle extraction pipeline. Downloads audio from any yt-dlp-compatible video platform, transcribes with **SenseVoice / whisper.cpp / openai-whisper**, and applies LLM-based text calibration for Chinese content.\r\n\r\n**Default engine**: SenseVoice Small (Alibaba FunASR) — ~1.5GB RAM, 234MB disk, 20× realtime speed, ~96% Chinese accuracy.\r\n\r\n**Tested & verified** on Windows 11 with real Bilibili & Xiaohongshu videos.\r\n\r\n## Quick Start\r\n\r\n```bash\r\n# One-command full pipeline (SenseVoice Small — default, blazing fast for Chinese)\r\npython scripts/run.py <video_url> --output-dir ./output\r\n\r\n# Use whisper.cpp GGML (even lighter, 2GB RAM)\r\npython scripts/run.py <video_url> --backend whispercpp --model medium-q5_1\r\n\r\n# Use openai-whisper (standard, 5GB RAM)\r\npython scripts/run.py <video_url> --backend openai --model medium\r\n\r\n# Download audio only\r\npython scripts/download_audio.py <video_url> <output_dir>\r\n\r\n# Download audio + video (keep both as middleware)\r\npython scripts/download_audio.py <video_url> <output_dir> --save-video --video-quality 1080\r\n\r\n# Transcribe existing audio with auto backend selection\r\npython scripts/transcribe.py <audio_file> --backend auto --language zh\r\n\r\n# Transcribe with specific backend\r\npython scripts/transcribe.py <audio_file> --backend sensevoice --language zh\r\npython scripts/transcribe.py <audio_file> --backend whispercpp --model medium-q5_1\r\npython scripts/transcribe.py <audio_file> --backend openai --model medium\r\n```\r\n\r\n## When to Use This Skill\r\n\r\nUse this skill when:\r\n1. The video has **no built-in subtitles** (Bilibili, Xiaohongshu, YouTube, etc.)\r\n2. You need **high-accuracy Chinese transcription** (~95% with medium model)\r\n3. You want **multiple output formats** (TXT, SRT, VTT, JSON)\r\n4. You need **LLM-assisted text calibration** for financial/technical terms\r\n5. The user says: \"下载字幕\", \"提取字幕\", \"语音转文字\", \"视频转文字\", \"字幕提取\", \"ASR转写\"\r\n\r\n## Workflow\r\n\r\n### Step 0: Install Dependencies (once)\r\n\r\n```bash\r\npython scripts/install_deps.py\r\n```\r\n\r\nAuto-detects OS and installs: ffmpeg (winget/brew/apt), yt-dlp (pip), openai-whisper (pip). Handles Windows ffmpeg path detection even when not in PATH.\r\n\r\n### Step 1: Download Audio\r\n\r\nRun `scripts/download_audio.py <url> [output_dir]`.\r\n\r\nUses yt-dlp to extract the best available audio format (m4a preferred). Supports Bilibili, YouTube, and 1800+ yt-dlp-compatible platforms. The script automatically detects ffmpeg even when not in system PATH.\r\n\r\n**Optional: Download video as middleware**\r\n\r\nAdd `--save-video` to persist the full video alongside audio:\r\n\r\n```bash\r\npython scripts/download_audio.py <url> --save-video --video-quality 1080\r\npython scripts/run.py <url> --save-video --video-quality 720 --calibrate\r\n```\r\n\r\n`--video-quality` preset table:\r\n\r\n| Preset | yt-dlp behaviour | Typical result |\r\n|--------|------------------|----------------|\r\n| `best` (default) | Highest available | 4K on YouTube, 480p on B站 (no login) |\r\n| `1080` | ≤1080p, fallback gracefully | 1080p where available |\r\n| `720` | ≤720p | Good balance for local storage |\r\n| `480` | ≤480p | Minimum acceptable for reference |\r\n| `360` | ≤360p | Extremely small files |\r\n| *raw string* | Direct yt-dlp format selector | Full flexibility |\r\n\r\n> **⚠️ B站 note**: Without login cookies, B站 caps at 480p. 720p+ requires `--cookies-from-browser`.\r\n\r\n**If download fails**: the video may require cookies. Try:\r\n```bash\r\nyt-dlp --cookies-from-browser chrome <url>\r\n```\r\n\r\n### Step 2: ASR Transcription (Multi-Backend)\r\n\r\nRun `scripts/transcribe.py <audio> --backend <engine> --model <size> --language <lang>`.\r\n\r\nThree backends, auto-selected by default (priority: SenseVoice → whisper.cpp → openai-whisper):\r\n\r\n### 🥇 SenseVoice Small (default for Chinese)\r\n\r\n| Property | Value |\r\n|----------|-------|\r\n| RAM | ~1.5GB |\r\n| Disk | ~234MB |\r\n| Speed | 20× realtime (CPU) |\r\n| Chinese accuracy | ~96% 🏆 |\r\n| Model source | ModelScope (auto-download, no VPN needed) |\r\n| Install | `pip install funasr modelscope` |\r\n\r\n> **Why SenseVoice?** Alibaba's FunASR engine, Chinese-optimized from the ground up. No HuggingFace download needed (uses ModelScope mirror in China). ~4× faster than openai-whisper medium on CPU, with comparable or better Chinese accuracy.\r\n\r\n### 🥈 whisper.cpp (GGML quantized, CPU-optimized)\r\n\r\n| Model | RAM | Disk | Speed | Quality | Best For |\r\n|-------|-----|------|-------|---------|----------|\r\n| `tiny-q5_1` | ~0.5GB | 32MB | fastest | low | Testing |\r\n| `small-q5_1` | ~1GB | 466MB | fast | decent | Quick preview |\r\n| `medium-q5_1` | ~2GB | 1.1GB | 3-5× faster than openai | **~95%** ⭐ | **Lightweight quality** |\r\n\r\n> **Setup**: `pip install pywhispercpp` + download GGML model from huggingface.co/ggerganov/whisper.cpp → place in `~/.cache/whispercpp/`. See `--show-backends` for instructions.\r\n\r\n### 🥉 openai-whisper (standard)\r\n\r\n| Model | RAM | Disk | Speed | Quality | Best For |\r\n|-------|-----|------|-------|---------|----------|\r\n| `medium` | ~5GB | 1.42GB | ~165 fps | **~95%** | Standard |\r\n| `large-v3` | ~10GB | 2.88GB | ~80 fps | ~97% | Best accuracy |\r\n| `large-v3-turbo` | ~6GB | 1.6GB | ~120 fps | ~96% | Good balance |\r\n\r\n> **Note**: `small` model removed (88-90% accuracy, superseded by SenseVoice). Use SenseVoice for light weight.\r\n\r\n### Step 3: Rule-Based Calibration\r\n\r\nAfter transcription, apply `calibrate.py` for mechanical corrections:\r\n\r\n```bash\r\n# Calibrate a raw .txt transcript\r\npython scripts/calibrate.py <output_dir>/<video_title>.txt\r\n# Output: <video_title>_calibrated.txt\r\n\r\n# Skip traditional→simplified conversion (already simplified input)\r\npython scripts/calibrate.py raw.txt --no-tradsimp\r\n```\r\n\r\n**Calibrate categories:** homophone fixes, financial terms, AI/tech company names, semiconductor/hardware terms, traditional→simplified (600+ chars).\r\nFor context-aware fixes (semantic errors, ambiguous names), use LLM review on top of rule-based output.\r\nSee `references/calibration_guide.md` for the full 80+ pattern library.\r\n\r\n### Step 4: Deliver Results\r\n\r\nPresent the calibrated text. Always include:\r\n- Model used (medium/large) and quality notes\r\n- Any sections with low confidence or unclear audio\r\n- Summary of corrections applied (counts by category)\r\n\r\n## Intermediate Artifacts & Step-by-Step Reuse\r\n\r\nEvery pipeline stage saves its output to disk. All artifacts persist in `output_dir/` after the run — no data is lost between stages.\r\n\r\n### Artifact Map\r\n\r\n| Stage | Artifact | Format | Filename pattern | Reusable alone? |\r\n|-------|----------|--------|------------------|-----------------|\r\n| Step 1 | Audio | `.m4a` | `<video_title>.m4a` | ✅ `download_audio.py` |\r\n| Step 1 | Video (optional) | `.mp4` | `<video_title>.mp4` | ✅ `download_audio.py --save-video` |\r\n| Step 2 | Transcript (raw) | `.txt` `.srt` `.vtt` `.json` | `<video_title>.txt` | ✅ `transcribe.py` |\r\n| Step 2 | Pipeline metadata | `.json` | `_pipeline_meta.json` | ✅ (reference only) |\r\n| Step 3 | Transcript (calibrated) | `_calibrated.txt` | `<title>_calibrated.txt` | ✅ `calibrate.py` |\r\n\r\n### Selective Reuse via `run.py` Flags\r\n\r\n```bash\r\n# Full pipeline (download + transcribe + calibrate)\r\npython scripts/run.py <url> --calibrate --output-dir ./out\r\n\r\n# Save video as middleware (downloads .m4a + .mp4)\r\npython scripts/run.py <url> --save-video --calibrate --output-dir ./out\r\n\r\n# Audio already exists → skip download, re-transcribe\r\npython scripts/run.py <url> --skip-download --output-dir ./out\r\n\r\n# Audio + transcript already exist → skip both, re-run calibration only\r\npython scripts/run.py <url> --skip-download --skip-transcribe --calibrate --output-dir ./out\r\n```\r\n\r\n`run.py` auto-detects existing artifacts by matching the audio filename base. It will warn and fall back to downloading/transcribing if no match is found.\r\n\r\n### Standalone Scripts\r\n\r\nEach stage has an independent entry point:\r\n\r\n```bash\r\n# Stage 1: Download audio only\r\npython scripts/download_audio.py <url> [output_dir] [filename]\r\n\r\n# Stage 1: Download audio + video at 720p (standalone)\r\npython scripts/download_audio.py <url> [output_dir] --save-video --video-quality 720\r\n\r\n# Stage 2: Transcribe existing audio\r\npython scripts/transcribe.py <audio.m4a> --model medium --language zh --output-dir ./out\r\n\r\n# Stage 3: Calibrate raw transcript (rule-based only)\r\npython scripts/calibrate.py <raw.txt> [--output <path>] [--no-tradsimp]\r\n```\r\n\r\n### Typical Reuse Scenarios\r\n\r\n**Scenario A — Change model, keep audio**\r\n```bash\r\n# Already have .m4a from previous run\r\npython scripts/run.py <url> --skip-download --model large-v3 --output-dir ./out\r\n```\r\n\r\n**Scenario B — Change language, keep audio + transcript**\r\n```bash\r\n# Have both .m4a and .txt; just recalibrate\r\npython scripts/run.py <url> --skip-download --skip-transcribe --calibrate --language en --output-dir ./out\r\n```\r\n\r\n**Scenario C — Batch calibrate multiple transcripts**\r\n```bash\r\n# Apply calibration to all raw .txt files in a directory\r\nGet-ChildItem .\\out\\*.txt | Where-Object { $_.Name -notmatch '_calibrated' } | ForEach-Object {\r\n    python scripts/calibrate.py $_.FullName\r\n}\r\n```\r\n\r\n## Platform Support\r\n\r\n| Platform | Status | Notes |\r\n|----------|--------|-------|\r\n| Bilibili | ✅ | Audio-only streams available without login. 720P+ video needs cookies. |\r\n| Xiaohongshu | ✅ | Full support via `XiaoHongShu` extractor. Short links (xhslink.com) auto-resolved. No cookies needed. |\r\n| YouTube | ✅ | Full support. Cookies may improve format selection. |\r\n| Douyin/TikTok | ⚠️ | Requires login cookies (`--cookies-from-browser` or `--cookies cookies.txt`). No cookies = download fails. |\r\n| All yt-dlp sites | ✅ | 1800+ supported platforms |\r\n\r\n## Extending with New ASR Models\r\n\r\n`scripts/transcribe.py` is designed for backend extensibility:\r\n\r\n1. Add model info to `ALL_MODELS` dict\r\n2. Implement `transcribe_<backend>()` function\r\n3. Add CLI flag in argparse\r\n4. Add backend to `detect_backends()`\r\n\r\n**Available backends**: sensevoice (✅ production), whispercpp (✅ code ready, GGML model manual), openai (✅ production)\r\n\r\n**Backend auto-selection**: When `--backend auto` (default), the engine picks the best available backend in priority order:\r\n1. **SenseVoice** — Chinese-optimized, fastest, lightest\r\n2. **whisper.cpp** — CPU-optimized, quantized models\r\n3. **openai-whisper** — general purpose, most compatible\r\n\r\n## Troubleshooting\r\n\r\n| Problem | Solution |\r\n|---------|----------|\r\n| SIGKILL / ffmpeg FileNotFoundError | ffmpeg not in PATH. Script auto-detects 7 common install locations (winget, scoop, chocolatey, manual). If ffmpeg is elsewhere, add its directory to system PATH. |\r\n| yt-dlp download fails | Update yt-dlp: `pip install -U yt-dlp`. Try with cookies. |\r\n| \"No subtitles found\" | Expected. This skill uses ASR, not built-in captions. |\r\n| ffmpeg not found | Run `install_deps.py` (handles Windows non-PATH detection). |\r\n| GPU not utilized | openai-whisper CPU-only by default. SenseVoice also runs on CPU. whisper.cpp uses AVX/SSE SIMD on CPU. |\r\n| `funasr` import error | `pip install funasr modelscope` — SenseVoice backend dependency. |\r\n| `pywhispercpp` import error | `pip install pywhispercpp` — whisper.cpp backend dependency. |\r\n| GGML model not found | Download from huggingface.co/ggerganov/whisper.cpp → place in `~/.cache/whispercpp/`. Or use `--backend sensevoice` instead. |\r\n| WDAC/AppLocker blocks torch DLL | Use `pip install torch==2.5.1 --index-url https://download.pytorch.org/whl/cpu` (CPU-only, signed). Also: `pip install numpy==2.0.2`. |\r\n| HuggingFace blocked (China) | SenseVoice uses ModelScope (no VPN needed). For whisper.cpp GGML, use a VPN to download models once. |\r\n\r\n## Performance Benchmarks (Tested)\r\n\r\n| Video Duration | Model | Backend | Time | RAM Peak | Accuracy |\r\n|---------------|-------|---------|------|----------|----------|\r\n| 7m 49s (Bilibili) | SenseVoice Small | sensevoice | ~20s | ~1.5GB | **~96%** 🏆 |\r\n| 9m 30s (Bilibili) | medium-q5_1 | whispercpp | ~2m | ~2GB | ~95% |\r\n| 9m 30s (Bilibili) | medium | openai | ~4m | ~5GB | ~95% |\r\n| 23m (Bilibili) | SenseVoice Small | sensevoice | ~60s | ~1.5GB | **~96%** 🏆 |\r\n| 23m (Bilibili) | medium | openai | ~12m | ~5GB | ~95% |\r\n\r\nTested on Windows 11, Intel i7, 16GB RAM. Performance may vary by CPU speed.\r\n\r\n## Changelog\r\n\r\n### v2.0.0 — Multi-Engine ASR\r\n- **New**: SenseVoice Small backend (Alibaba FunASR) — default engine, ~1.5GB RAM, 20× realtime, ~96% Chinese accuracy\r\n- **New**: whisper.cpp GGML backend via pywhispercpp — CPU-optimized quantized models (0.5-2GB RAM)\r\n- **New**: `--backend` parameter (`auto`/`sensevoice`/`whispercpp`/`openai`)\r\n- **New**: `transcribe.py --show-backends` diagnostic command\r\n- **New**: Auto-backend detection and fallback chain (SenseVoice → whisper.cpp → openai-whisper)\r\n- **Remove**: `small` model from openai-whisper (superseded by SenseVoice)\r\n- **Remove**: faster-whisper stub (CTranslate2 — WDAC incompatibility, never actually worked)\r\n- **Improve**: transcriber architecture — clean backend dispatch, shared output writer\r\n- **Improve**: Chinese-optimized default path (SenseVoice Small via ModelScope, no VPN needed)\r\n\r\n### v1.0.8\r\n- **New**: Semiconductor / hardware domain calibration rules (30+ patterns) — 韬定律→道定律, 全站协同→全栈协同, 量子碎穿→量子隧穿, 吸片→芯片, 奈米→纳米, etc.\r\n- **Improve**: Huawei 道定律 video now correctly calibrated (0→26 pattern corrections, ~98% accuracy)\r\n- **Fix**: Douyin/TikTok platform note — clearly states cookies requirement\r\n\r\n### v1.0.7\r\n- **New**: `--video-quality` parameter — presets best/2160/1440/1080/720/480/360 + raw yt-dlp format support\r\n- **New**: `find_artifacts_in_dir()` replaces `find_audio_in_dir()` — caches both .m4a AND .mp4 on `--skip-download --save-video`\r\n- **Change**: Video download format selector now uses `height<=` filters (graceful fallback) instead of hardcoded mp4-only\r\n- **Improve**: `download_video()` now reports which quality preset is in use\r\n- **Improve**: Artifact map includes video (.mp4) with quality column\r\n- **Improve**: SKILL.md adds `--video-quality` usage table and B站 login caveats\r\n\r\n### v1.0.6\r\n- **New**: `--save-video` flag — download and persist full video (.mp4) alongside audio\r\n- **New**: `download_video()` function in download_audio.py (standalone: `--save-video`)\r\n- **Improve**: Artifact map now includes video (.mp4) as first-class middleware\r\n- **Improve**: `_pipeline_meta.json` includes `video_path` for full traceability\r\n\r\n### v1.0.5\r\n- **Remove**: `small` model from all backends (88-90% accuracy, too poor for production)\r\n- **Change**: Default model locked to `medium` (was medium in code, but docs still promoted small)\r\n- **Improve**: Model table and benchmarks now medium-only baseline\r\n\r\n### v1.0.2\r\n- **New**: Xiaohongshu (小红书) platform support — yt-dlp `XiaoHongShu` extractor\r\n- **New**: Short link auto-resolution (xhslink.com → full URL via redirect)\r\n- **Improve**: Platform support table now lists 小红书 explicitly\r\n- **Improve**: Quick Start examples include xhslink.com usage\r\n\r\n### v1.0.1\r\n- **Fix**: Expanded ffmpeg search paths from 3→7 (winget/scoop/chocolatey/ProgramFiles(x86))\r\n- **Fix**: `ensure_deps()` now injects ffmpeg into `os.environ['PATH']` on success\r\n- **Fix**: SIGKILL troubleshooting updated — root cause is ffmpeg PATH, not OOM\r\n- **Improve**: Auto-detect GPU (`torch.cuda.is_available()`) for fp16 support\r\n- **Improve**: `verbose=True` for real-time transcription progress visibility\r\n- **Improve**: More accurate error messages in dependency checks\r\n\r\n### v1.0.0\r\n- Initial release: download (yt-dlp) + transcribe (whisper) + calibrate (LLM) pipeline\r\n- 7 ffmpeg install path auto-detection\r\n- Multi-format output (TXT, SRT, VTT, JSON)\r\n- Platform support: Bilibili, YouTube, all yt-dlp sites\n\nFile v2.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn71y79mxxwmzynnpj6e284kch82gwd2\",\n  \"slug\": \"video-subtitle-extractor\",\n  \"version\": \"2.0.0\",\n  \"publishedAt\": 1779870253771\n}\n\nFile v2.0.0:references/asr_models.md\n\n# ASR Model Selection Guide\r\n\r\n## Available Models\r\n\r\n| Model | RAM | Disk | Speed | Quality | Best For |\r\n|-------|-----|------|-------|---------|----------|\r\n| `small` | ~2GB | 461MB | Fast (~1min/6min audio) | Good | Quick tests, low-resource systems |\r\n| `medium` | ~5GB | 1.42GB | Medium (~3-5min) | High | **Recommended default** - best quality/speed ratio |\r\n| `large-v3` | ~10GB | 2.88GB | Slow (~10-20min) | Best | Production quality, needs high RAM |\r\n| `large-v3-turbo` | ~6GB | 1.6GB | Medium-fast | High | Good compromise, smaller than large-v3 |\r\n\r\n## Language-Specific Notes\r\n\r\n### Chinese (zh)\r\n- `medium`: Good for general Chinese content. Some errors on homophones and financial terms.\r\n- `large-v3`: Best Chinese accuracy, handles accents and domain terminology better.\r\n- Common errors: 同音字混淆 (硬扛→硬钢), 金融术语 (抛压→抛押, 交筹→焦愁), K线术语 (十字星→14星)\r\n\r\n### English (en)\r\n- `small`: Sufficient for clear English speech.\r\n- `medium`: Excellent accuracy for most content.\r\n\r\n## Memory Constraints\r\n\r\nOn Windows, `large-v3` may be killed (SIGKILL) on systems with <16GB RAM due to FP32 fallback.\r\nIf killed, fall back to `medium` or use `larger-v3-turbo`.\r\n\r\n## Future Model Compatibility\r\n\r\nThe `transcribe.py` script is designed for easy backend extension:\r\n- `faster-whisper`: CTranslate2 backend, more memory efficient\r\n- `whisper.cpp`: Native C++ implementation\r\n- `mlx-whisper`: Apple Silicon optimized\r\n- Cloud APIs: AssemblyAI, iFlytek, Whisper API\r\n\r\nTo add a new backend, implement a `transcribe_<backend>()` function in transcribe.py\r\nfollowing the same interface (audio_path, model_name, language, output_dir).\r\n\r\n## Model Auto-Download\r\n\r\nModels are downloaded automatically by openai-whisper on first use.\r\nCache location:\r\n- Windows: `C:\\Users\\<user>\\.cache\\whisper\\`\r\n- macOS: `~/Library/Caches/whisper/`\r\n- Linux: `~/.cache/whisper/`\n\nFile v2.0.0:references/calibration_guide.md\n\n# Text Calibration Guide for Chinese ASR Output\r\n\r\nCommon transcription errors and their corrections. Apply these patterns when calibrating whisper output for Chinese financial/technical content.\r\n\r\n## 1. Homophone Replacements (同音字混淆)\r\n\r\n| Raw (Wrong) | Correct | Example |\r\n|-------------|---------|---------|\r\n| 硬钢 | 硬扛 | 硬钢→硬扛 |\r\n| 抛押 | 抛压 | 消化抛押→消化抛压 |\r\n| 模两个月 | 磨两个月 | 横盘调整模两个月→磨两个月 |\r\n| 膜光短线 | 磨光短线 | 膜光短线→磨光短线 |\r\n| 流通骨 | 流通股 | 流通骨的换手→流通股的换手 |\r\n| 金接盘 | 新接盘 | 金接盘的成本→新接盘的成本 |\r\n| 拉伸 | 拉升 | 拉伸成本→拉升成本 |\r\n| 跟锋 | 跟风 | 跟锋买入→跟风买入 |\r\n| 微转 | 微赚 | 微转就抛售→微赚就抛售 |\r\n| 落带为安 | 落袋为安 | 落带为安→落袋为安 |\r\n| 互盘 | 护盘 | 主力互盘明显→主力护盘明显 |\r\n| 逼散互买 | 逼散户卖 | 洗盘本质是逼散互买→逼散户卖 |\r\n| 仅 | 有 | 仅有资金拖住→有资金托住 |\r\n| 军线 | 均线 | 关键军线→关键均线 |\r\n| 快有动作 | 快有动作 | Already correct, but watch for 快会→快会 |\r\n\r\n## 2. Financial Term Corrections\r\n\r\n| Raw | Correct | Context |\r\n|-----|---------|---------|\r\n| 交筹 | 交筹 | 慢慢就焦愁→慢慢就交筹 |\r\n| 再计 | 在即 | 拉升再计→拉升在即 |\r\n| 没装 | 没仓 | 根本没装→根本没仓 |\r\n| 利空 | 利空 | Already correct, verify |\r\n| K线收14星 | K线收十字星 | 14→十 |\r\n| 14星 | 十字星 | K线收十字星 |\r\n| 洗崩 | 洗崩 | Already correct (跌太多) |\r\n| 割肉 | 割肉 | Already correct |\r\n\r\n## 3. Domain Term Patterns\r\n\r\nWhisper often confuses financial jargon:\r\n- 洗盘 (xǐ pán) vs 洗盘 (same pronunciation but context-dependent)\r\n- 筹码 (chóu mǎ) - usually correct\r\n- 建仓 (jiàn cāng) - usually correct\r\n- 杠杆 (gàng gǎn) - usually correct\r\n- 信托 (xìn tuō) - usually correct\r\n\r\n## 4. Structural Cleanup\r\n\r\n- Add proper punctuation (periods, commas) where ASR output lacks them\r\n- Split long run-on sentences at natural topic breaks\r\n- Format as flowing paragraphs, not timestamp-ordered fragments\r\n- Add section headings for topic shifts: \"洗盘核心目的\", \"三个核心指标\", \"三个信号\", etc.\r\n- Keep timestamps if user wants time-coded output (from .srt/.vtt)\r\n\r\n## 5. Quality Indicators\r\n\r\nAfter calibration, flag low-confidence sections:\r\n- Unclear audio sections (background noise, overlapping speech)\r\n- Rapid technical jargon sequences\r\n- Sections where multiple interpretations are plausible\r\n\r\n## 6. Multi-language Content\r\n\r\nFor bilingual content (Chinese + English):\r\n- Preserve English terms: PE ratio, MA, MACD, KDJ, Bollinger Bands\r\n- Mixed language phrases: \"比如 PE 20倍\", \"MACD 金叉\" are correct\r\n- Don't translate technical terms to Chinese\r\n\r\n## 7. Calibration Output Format\r\n\r\nAfter applying corrections, present as:\r\n- Clean prose with proper Chinese punctuation\r\n- Optional: show what was changed vs raw output\r\n- Optional: timestamp references from original SRT/VTT\r\n\r\n## 8. AI / Tech Domain Corrections (AI 及科技领域)\r\n\r\nFor videos about AI, semiconductors, and tech investment topics, watch for:\r\n\r\n### Chinese Company Names (whisper frequent errors)\r\n| Raw (Wrong) | Correct | Context |\r\n|-------------|---------|---------|\r\n| 中繼續創 | 中际旭创 | A股光模块龙头 |\r\n| 新益勝 | 新易盛 | A股光模块 |\r\n| 天賦通信 | 天孚通信 | A股通信 |\r\n| 元傑科技 | 源杰科技 | A股芯片 |\r\n| 阿力 | 阿里 | 阿里巴巴 |\r\n| Alley | 阿里 | Context: 腾讯、阿里、字节 |\r\n\r\n### AI Product & Term Names\r\n| Raw (Wrong) | Correct |\r\n|-------------|---------|\r\n| Deepseat | DeepSeek |\r\n| ChadGPT | ChatGPT |\r\n| Cloud | Claude (if Anthropic context) |\r\n| Moe價構 / 莫架构 | MoE架构 (Mixture of Experts) |\r\n| HPM | HBM (High Bandwidth Memory) |\r\n| 光膜块 | 光模块 (Optical Module) |\r\n| 夜冷 | 液冷 (Liquid Cooling) |\r\n| 巨深智能 | 具身智能 (Embodied AI) |\r\n| 端側推理 | 端侧推理 (On-device Inference) |\r\n| 推測解碼 | 推测解码 (Speculative Decoding) |\r\n| 主能板块 | 储能板块 (Energy Storage) |\r\n\r\n### Traditional → Simplified Chinese\r\nWhisper medium/large models commonly output traditional Chinese (繁體) for simplified content. Always convert:\r\n- 發→发, 來→来, 時→时, 會→会, 體→体, 報→报, 機→机\r\n- 線→线, 構→构, 購→购, 業→业, 電→电, 纜→缆, 銅→铜\r\n- 壹→一, 貳→二, 等等\n\nFile v2.0.0:skill-card.md\n\n## Description:\n\nCross-platform video subtitle extraction using multi-engine ASR: it downloads audio from video URLs with yt-dlp, transcribes with SenseVoice, whisper.cpp, or openai-whisper, and can calibrate Chinese financial and technical transcripts.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[forhonourlx](https://clawhub.ai/user/forhonourlx)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers, analysts, and content teams use this skill to extract subtitles or transcript files from videos that lack built-in captions, especially Chinese-language financial or technical videos. It guides agents through dependency setup, media download, ASR transcription, optional video retention, and transcript calibration.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The security review flags broad install, download, browser-cookie, and persistent storage behaviors.\n\nMitigation: Review the skill before use, run it only in a trusted environment, and approve dependency installation and media downloads deliberately.\n\nRisk: Browser-cookie workflows can expose account session material to the local toolchain.\n\nMitigation: Avoid browser-cookie options unless the machine and toolchain are trusted; prefer videos that do not require cookies.\n\nRisk: Downloaded media and generated transcripts can remain on disk after execution.\n\nMitigation: Use a dedicated output directory and delete media, transcripts, and metadata after they are no longer needed.\n\n## Reference(s):\n\n- [ASR Model Selection Guide](artifact/references/asr_models.md)\n- [Text Calibration Guide for Chinese ASR Output](artifact/references/calibration_guide.md)\n- [ClawHub Skill Page](https://clawhub.ai/forhonourlx/skills/video-subtitle-extractor)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, code, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown guidance with shell commands and generated transcript files such as TXT, SRT, VTT, and JSON]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May create persistent media, transcript, subtitle, metadata, and calibrated transcript files in the selected output directory.]\n\n## Skill Version(s):\n\n2.0.0 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v2.0.0:LICENSE\n\nMIT No Attribution License\n\nCopyright (c) 2025 forhonourlx\n\nPermission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the \"Software\"), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.\n\nArchive v1.0.8: 11 files, 29109 bytes\n\nFiles: LICENSE (917b), references/asr_models.md (1918b), references/calibration_guide.md (4619b), scripts/calibrate.py (12842b), scripts/download_audio.py (11069b), scripts/install_deps.py (4890b), scripts/run.py (12908b), scripts/transcribe.py (9569b), skill-card.md (2474b), SKILL.md (13199b), _meta.json (143b)\n\nFile v1.0.8:SKILL.md\n\n---\r\nname: video-subtitle-extractor\r\ndescription: |\r\n  Cross-platform video subtitle extraction using ASR (speech-to-text). Downloads audio from video URLs via yt-dlp, transcribes with openai-whisper (medium/large-v3, default: medium), and applies LLM-based text calibration for Chinese financial/technical content. Use when: (1) extracting subtitles from Bilibili, Xiaohongshu, YouTube, or any yt-dlp-supported platform, (2) the video has no built-in subtitles, (3) users say \"下载字幕\", \"提取字幕\", \"语音转文字\", \"视频转文字\", \"字幕提取\", \"ASR转写\", (4) needing to transcribe audio files to text, (5) working with Chinese-language video content requiring high-accuracy transcription. Automatically handles dependency installation (ffmpeg, yt-dlp, openai-whisper) and model downloads.\r\n---\r\n\r\n# Video Subtitle Extractor 🎬→📝\r\n\r\nCross-platform ASR subtitle extraction pipeline. Downloads audio from any yt-dlp-compatible video platform, transcribes with openai-whisper, and applies LLM-based text calibration for Chinese content.\r\n\r\n**Tested & verified** on Windows 11 with real Bilibili & Xiaohongshu videos (medium model, ~95% accuracy for Chinese).\r\n\r\n## Quick Start\r\n\r\n```bash\r\n# One-command full pipeline (Bilibili, Xiaohongshu, YouTube, etc.)\r\npython scripts/run.py <video_url> --model medium --language zh --output-dir ./output\r\n\r\n# Example: Xiaohongshu short link (auto-resolves redirect)\r\npython scripts/run.py https://xhslink.com/xxxxx --model medium --language zh\r\n\r\n# Download audio only\r\npython scripts/download_audio.py <video_url> <output_dir>\r\n\r\n# Download audio + video (keep both as middleware)\r\npython scripts/download_audio.py <video_url> <output_dir> --save-video --video-quality 1080\r\n\r\n# Transcribe existing audio\r\npython scripts/transcribe.py <audio_file> --model medium --language zh\r\n```\r\n\r\n## When to Use This Skill\r\n\r\nUse this skill when:\r\n1. The video has **no built-in subtitles** (Bilibili, Xiaohongshu, YouTube, etc.)\r\n2. You need **high-accuracy Chinese transcription** (~95% with medium model)\r\n3. You want **multiple output formats** (TXT, SRT, VTT, JSON)\r\n4. You need **LLM-assisted text calibration** for financial/technical terms\r\n5. The user says: \"下载字幕\", \"提取字幕\", \"语音转文字\", \"视频转文字\", \"字幕提取\", \"ASR转写\"\r\n\r\n## Workflow\r\n\r\n### Step 0: Install Dependencies (once)\r\n\r\n```bash\r\npython scripts/install_deps.py\r\n```\r\n\r\nAuto-detects OS and installs: ffmpeg (winget/brew/apt), yt-dlp (pip), openai-whisper (pip). Handles Windows ffmpeg path detection even when not in PATH.\r\n\r\n### Step 1: Download Audio\r\n\r\nRun `scripts/download_audio.py <url> [output_dir]`.\r\n\r\nUses yt-dlp to extract the best available audio format (m4a preferred). Supports Bilibili, YouTube, and 1800+ yt-dlp-compatible platforms. The script automatically detects ffmpeg even when not in system PATH.\r\n\r\n**Optional: Download video as middleware**\r\n\r\nAdd `--save-video` to persist the full video alongside audio:\r\n\r\n```bash\r\npython scripts/download_audio.py <url> --save-video --video-quality 1080\r\npython scripts/run.py <url> --save-video --video-quality 720 --calibrate\r\n```\r\n\r\n`--video-quality` preset table:\r\n\r\n| Preset | yt-dlp behaviour | Typical result |\r\n|--------|------------------|----------------|\r\n| `best` (default) | Highest available | 4K on YouTube, 480p on B站 (no login) |\r\n| `1080` | ≤1080p, fallback gracefully | 1080p where available |\r\n| `720` | ≤720p | Good balance for local storage |\r\n| `480` | ≤480p | Minimum acceptable for reference |\r\n| `360` | ≤360p | Extremely small files |\r\n| *raw string* | Direct yt-dlp format selector | Full flexibility |\r\n\r\n> **⚠️ B站 note**: Without login cookies, B站 caps at 480p. 720p+ requires `--cookies-from-browser`.\r\n\r\n**If download fails**: the video may require cookies. Try:\r\n```bash\r\nyt-dlp --cookies-from-browser chrome <url>\r\n```\r\n\r\n### Step 2: ASR Transcription\r\n\r\nRun `scripts/transcribe.py <audio> --model <size> --language <lang>`.\r\n\r\nModels are auto-downloaded on first use (disk space required):\r\n\r\n| Model | RAM | Disk | Speed | Quality | Best For |\r\n|-------|-----|------|-------|---------|----------|\r\n| `medium` | ~5GB | 1.42GB | ~165 fps | **~95%** ✅ | **Recommended (default)** |\r\n| `large-v3` | ~10GB | 2.88GB | ~80 fps | ~97% | Best accuracy |\r\n| `large-v3-turbo` | ~6GB | 1.6GB | ~120 fps | ~96% | Good balance |\r\n\r\n> **⚠️ Windows note**: `large-v3` needs >10GB RAM. If transcription fails, always check ffmpeg PATH first (see Troubleshooting).\r\n\r\nOutput formats: `txt`, `srt`, `vtt`, `json` (default: all).\r\n\r\nSee `references/asr_models.md` for full model comparison.\r\n\r\n### Step 3: Rule-Based Calibration\r\n\r\nAfter transcription, apply `calibrate.py` for mechanical corrections:\r\n\r\n```bash\r\n# Calibrate a raw .txt transcript\r\npython scripts/calibrate.py <output_dir>/<video_title>.txt\r\n# Output: <video_title>_calibrated.txt\r\n\r\n# Skip traditional→simplified conversion (already simplified input)\r\npython scripts/calibrate.py raw.txt --no-tradsimp\r\n```\r\n\r\n**Calibrate categories:** homophone fixes, financial terms, AI/tech company names, semiconductor/hardware terms, traditional→simplified (600+ chars).\r\nFor context-aware fixes (semantic errors, ambiguous names), use LLM review on top of rule-based output.\r\nSee `references/calibration_guide.md` for the full 80+ pattern library.\r\n\r\n### Step 4: Deliver Results\r\n\r\nPresent the calibrated text. Always include:\r\n- Model used (medium/large) and quality notes\r\n- Any sections with low confidence or unclear audio\r\n- Summary of corrections applied (counts by category)\r\n\r\n## Intermediate Artifacts & Step-by-Step Reuse\r\n\r\nEvery pipeline stage saves its output to disk. All artifacts persist in `output_dir/` after the run — no data is lost between stages.\r\n\r\n### Artifact Map\r\n\r\n| Stage | Artifact | Format | Filename pattern | Reusable alone? |\r\n|-------|----------|--------|------------------|-----------------|\r\n| Step 1 | Audio | `.m4a` | `<video_title>.m4a` | ✅ `download_audio.py` |\r\n| Step 1 | Video (optional) | `.mp4` | `<video_title>.mp4` | ✅ `download_audio.py --save-video` |\r\n| Step 2 | Transcript (raw) | `.txt` `.srt` `.vtt` `.json` | `<video_title>.txt` | ✅ `transcribe.py` |\r\n| Step 2 | Pipeline metadata | `.json` | `_pipeline_meta.json` | ✅ (reference only) |\r\n| Step 3 | Transcript (calibrated) | `_calibrated.txt` | `<title>_calibrated.txt` | ✅ `calibrate.py` |\r\n\r\n### Selective Reuse via `run.py` Flags\r\n\r\n```bash\r\n# Full pipeline (download + transcribe + calibrate)\r\npython scripts/run.py <url> --calibrate --output-dir ./out\r\n\r\n# Save video as middleware (downloads .m4a + .mp4)\r\npython scripts/run.py <url> --save-video --calibrate --output-dir ./out\r\n\r\n# Audio already exists → skip download, re-transcribe\r\npython scripts/run.py <url> --skip-download --output-dir ./out\r\n\r\n# Audio + transcript already exist → skip both, re-run calibration only\r\npython scripts/run.py <url> --skip-download --skip-transcribe --calibrate --output-dir ./out\r\n```\r\n\r\n`run.py` auto-detects existing artifacts by matching the audio filename base. It will warn and fall back to downloading/transcribing if no match is found.\r\n\r\n### Standalone Scripts\r\n\r\nEach stage has an independent entry point:\r\n\r\n```bash\r\n# Stage 1: Download audio only\r\npython scripts/download_audio.py <url> [output_dir] [filename]\r\n\r\n# Stage 1: Download audio + video at 720p (standalone)\r\npython scripts/download_audio.py <url> [output_dir] --save-video --video-quality 720\r\n\r\n# Stage 2: Transcribe existing audio\r\npython scripts/transcribe.py <audio.m4a> --model medium --language zh --output-dir ./out\r\n\r\n# Stage 3: Calibrate raw transcript (rule-based only)\r\npython scripts/calibrate.py <raw.txt> [--output <path>] [--no-tradsimp]\r\n```\r\n\r\n### Typical Reuse Scenarios\r\n\r\n**Scenario A — Change model, keep audio**\r\n```bash\r\n# Already have .m4a from previous run\r\npython scripts/run.py <url> --skip-download --model large-v3 --output-dir ./out\r\n```\r\n\r\n**Scenario B — Change language, keep audio + transcript**\r\n```bash\r\n# Have both .m4a and .txt; just recalibrate\r\npython scripts/run.py <url> --skip-download --skip-transcribe --calibrate --language en --output-dir ./out\r\n```\r\n\r\n**Scenario C — Batch calibrate multiple transcripts**\r\n```bash\r\n# Apply calibration to all raw .txt files in a directory\r\nGet-ChildItem .\\out\\*.txt | Where-Object { $_.Name -notmatch '_calibrated' } | ForEach-Object {\r\n    python scripts/calibrate.py $_.FullName\r\n}\r\n```\r\n\r\n## Platform Support\r\n\r\n| Platform | Status | Notes |\r\n|----------|--------|-------|\r\n| Bilibili | ✅ | Audio-only streams available without login. 720P+ video needs cookies. |\r\n| Xiaohongshu | ✅ | Full support via `XiaoHongShu` extractor. Short links (xhslink.com) auto-resolved. No cookies needed. |\r\n| YouTube | ✅ | Full support. Cookies may improve format selection. |\r\n| Douyin/TikTok | ⚠️ | Requires login cookies (`--cookies-from-browser` or `--cookies cookies.txt`). No cookies = download fails. |\r\n| All yt-dlp sites | ✅ | 1800+ supported platforms |\r\n\r\n## Extending with New ASR Models\r\n\r\n`scripts/transcribe.py` is designed for backend extensibility:\r\n\r\n1. Add model info to `MODEL_SIZES` dict\r\n2. Implement `transcribe_<backend>()` function\r\n3. Add CLI flag in argparse\r\n\r\n**Planned backends**: faster-whisper (CTranslate2), whisper.cpp (native C++), Cloud APIs (AssemblyAI, iFlytek).\r\n\r\n## Troubleshooting\r\n\r\n| Problem | Solution |\r\n|---------|----------|\r\n| SIGKILL / ffmpeg FileNotFoundError | ffmpeg not in PATH. Script auto-detects 7 common install locations (winget, scoop, chocolatey, manual). If ffmpeg is elsewhere, add its directory to system PATH. |\r\n| yt-dlp download fails | Update yt-dlp: `pip install -U yt-dlp`. Try with cookies. |\r\n| \"No subtitles found\" | Expected. This skill uses ASR, not built-in captions. |\r\n| ffmpeg not found | Run `install_deps.py` (handles Windows non-PATH detection). |\r\n| GPU not utilized | openai-whisper CPU-only by default. Install `faster-whisper` for GPU. |\r\n\r\n## Performance Benchmarks (Tested)\r\n\r\n| Video Duration | Model | Time | RAM Peak | Accuracy |\r\n|---------------|-------|------|----------|----------|\r\n| 6 min (Bilibili) | medium | ~4m 30s | ~6GB | ~95% |\r\n| 13 min (Bilibili) | medium | ~8m | ~6.5GB | ~95% |\r\n| 15 min (Bilibili) | medium | ~10m | ~5GB | **~95%** ✅ |\r\n| 9 min (Bilibili) | medium | ~4m | ~5GB | **~95%** ✅ |\r\n\r\nTested on Windows 11, Intel i7, 16GB RAM. Performance may vary by CPU speed.\r\n\r\n## Changelog\r\n\r\n### v1.0.8\r\n- **New**: Semiconductor / hardware domain calibration rules (30+ patterns) — 韬定律→道定律, 全站协同→全栈协同, 量子碎穿→量子隧穿, 吸片→芯片, 奈米→纳米, etc.\r\n- **Improve**: Huawei 道定律 video now correctly calibrated (0→26 pattern corrections, ~98% accuracy)\r\n- **Fix**: Douyin/TikTok platform note — clearly states cookies requirement\r\n\r\n### v1.0.7\r\n- **New**: `--video-quality` parameter — presets best/2160/1440/1080/720/480/360 + raw yt-dlp format support\r\n- **New**: `find_artifacts_in_dir()` replaces `find_audio_in_dir()` — caches both .m4a AND .mp4 on `--skip-download --save-video`\r\n- **Change**: Video download format selector now uses `height<=` filters (graceful fallback) instead of hardcoded mp4-only\r\n- **Improve**: `download_video()` now reports which quality preset is in use\r\n- **Improve**: Artifact map includes video (.mp4) with quality column\r\n- **Improve**: SKILL.md adds `--video-quality` usage table and B站 login caveats\r\n\r\n### v1.0.6\r\n- **New**: `--save-video` flag — download and persist full video (.mp4) alongside audio\r\n- **New**: `download_video()` function in download_audio.py (standalone: `--save-video`)\r\n- **Improve**: Artifact map now includes video (.mp4) as first-class middleware\r\n- **Improve**: `_pipeline_meta.json` includes `video_path` for full traceability\r\n\r\n### v1.0.5\r\n- **Remove**: `small` model from all backends (88-90% accuracy, too poor for production)\r\n- **Change**: Default model locked to `medium` (was medium in code, but docs still promoted small)\r\n- **Improve**: Model table and benchmarks now medium-only baseline\r\n\r\n### v1.0.2\r\n- **New**: Xiaohongshu (小红书) platform support — yt-dlp `XiaoHongShu` extractor\r\n- **New**: Short link auto-resolution (xhslink.com → full URL via redirect)\r\n- **Improve**: Platform support table now lists 小红书 explicitly\r\n- **Improve**: Quick Start examples include xhslink.com usage\r\n\r\n### v1.0.1\r\n- **Fix**: Expanded ffmpeg search paths from 3→7 (winget/scoop/chocolatey/ProgramFiles(x86))\r\n- **Fix**: `ensure_deps()` now injects ffmpeg into `os.environ['PATH']` on success\r\n- **Fix**: SIGKILL troubleshooting updated — root cause is ffmpeg PATH, not OOM\r\n- **Improve**: Auto-detect GPU (`torch.cuda.is_available()`) for fp16 support\r\n- **Improve**: `verbose=True` for real-time transcription progress visibility\r\n- **Improve**: More accurate error messages in dependency checks\r\n\r\n### v1.0.0\r\n- Initial release: download (yt-dlp) + transcribe (whisper) + calibrate (LLM) pipeline\r\n- 7 ffmpeg install path auto-detection\r\n- Multi-format output (TXT, SRT, VTT, JSON)\r\n- Platform support: Bilibili, YouTube, all yt-dlp sites\n\nFile v1.0.8:_meta.json\n\n{\n  \"ownerId\": \"kn71y79mxxwmzynnpj6e284kch82gwd2\",\n  \"slug\": \"video-subtitle-extractor\",\n  \"version\": \"1.0.8\",\n  \"publishedAt\": 1779765139002\n}\n\nFile v1.0.8:references/asr_models.md\n\n# ASR Model Selection Guide\r\n\r\n## Available Models\r\n\r\n| Model | RAM | Disk | Speed | Quality | Best For |\r\n|-------|-----|------|-------|---------|----------|\r\n| `small` | ~2GB | 461MB | Fast (~1min/6min audio) | Good | Quick tests, low-resource systems |\r\n| `medium` | ~5GB | 1.42GB | Medium (~3-5min) | High | **Recommended default** - best quality/speed ratio |\r\n| `large-v3` | ~10GB | 2.88GB | Slow (~10-20min) | Best | Production quality, needs high RAM |\r\n| `large-v3-turbo` | ~6GB | 1.6GB | Medium-fast | High | Good compromise, smaller than large-v3 |\r\n\r\n## Language-Specific Notes\r\n\r\n### Chinese (zh)\r\n- `medium`: Good for general Chinese content. Some errors on homophones and financial terms.\r\n- `large-v3`: Best Chinese accuracy, handles accents and domain terminology better.\r\n- Common errors: 同音字混淆 (硬扛→硬钢), 金融术语 (抛压→抛押, 交筹→焦愁), K线术语 (十字星→14星)\r\n\r\n### English (en)\r\n- `small`: Sufficient for clear English speech.\r\n- `medium`: Excellent accuracy for most content.\r\n\r\n## Memory Constraints\r\n\r\nOn Windows, `large-v3` may be killed (SIGKILL) on systems with <16GB RAM due to FP32 fallback.\r\nIf killed, fall back to `medium` or use `larger-v3-turbo`.\r\n\r\n## Future Model Compatibility\r\n\r\nThe `transcribe.py` script is designed for easy backend extension:\r\n- `faster-whisper`: CTranslate2 backend, more memory efficient\r\n- `whisper.cpp`: Native C++ implementation\r\n- `mlx-whisper`: Apple Silicon optimized\r\n- Cloud APIs: AssemblyAI, iFlytek, Whisper API\r\n\r\nTo add a new backend, implement a `transcribe_<backend>()` function in transcribe.py\r\nfollowing the same interface (audio_path, model_name, language, output_dir).\r\n\r\n## Model Auto-Download\r\n\r\nModels are downloaded automatically by openai-whisper on first use.\r\nCache location:\r\n- Windows: `C:\\Users\\<user>\\.cache\\whisper\\`\r\n- macOS: `~/Library/Caches/whisper/`\r\n- Linux: `~/.cache/whisper/`\n\nFile v1.0.8:references/calibration_guide.md\n\n# Text Calibration Guide for Chinese ASR Output\r\n\r\nCommon transcription errors and their corrections. Apply these patterns when calibrating whisper output for Chinese financial/technical content.\r\n\r\n## 1. Homophone Replacements (同音字混淆)\r\n\r\n| Raw (Wrong) | Correct | Example |\r\n|-------------|---------|---------|\r\n| 硬钢 | 硬扛 | 硬钢→硬扛 |\r\n| 抛押 | 抛压 | 消化抛押→消化抛压 |\r\n| 模两个月 | 磨两个月 | 横盘调整模两个月→磨两个月 |\r\n| 膜光短线 | 磨光短线 | 膜光短线→磨光短线 |\r\n| 流通骨 | 流通股 | 流通骨的换手→流通股的换手 |\r\n| 金接盘 | 新接盘 | 金接盘的成本→新接盘的成本 |\r\n| 拉伸 | 拉升 | 拉伸成本→拉升成本 |\r\n| 跟锋 | 跟风 | 跟锋买入→跟风买入 |\r\n| 微转 | 微赚 | 微转就抛售→微赚就抛售 |\r\n| 落带为安 | 落袋为安 | 落带为安→落袋为安 |\r\n| 互盘 | 护盘 | 主力互盘明显→主力护盘明显 |\r\n| 逼散互买 | 逼散户卖 | 洗盘本质是逼散互买→逼散户卖 |\r\n| 仅 | 有 | 仅有资金拖住→有资金托住 |\r\n| 军线 | 均线 | 关键军线→关键均线 |\r\n| 快有动作 | 快有动作 | Already correct, but watch for 快会→快会 |\r\n\r\n## 2. Financial Term Corrections\r\n\r\n| Raw | Correct | Context |\r\n|-----|---------|---------|\r\n| 交筹 | 交筹 | 慢慢就焦愁→慢慢就交筹 |\r\n| 再计 | 在即 | 拉升再计→拉升在即 |\r\n| 没装 | 没仓 | 根本没装→根本没仓 |\r\n| 利空 | 利空 | Already correct, verify |\r\n| K线收14星 | K线收十字星 | 14→十 |\r\n| 14星 | 十字星 | K线收十字星 |\r\n| 洗崩 | 洗崩 | Already correct (跌太多) |\r\n| 割肉 | 割肉 | Already correct |\r\n\r\n## 3. Domain Term Patterns\r\n\r\nWhisper often confuses financial jargon:\r\n- 洗盘 (xǐ pán) vs 洗盘 (same pronunciation but context-dependent)\r\n- 筹码 (chóu mǎ) - usually correct\r\n- 建仓 (jiàn cāng) - usually correct\r\n- 杠杆 (gàng gǎn) - usually correct\r\n- 信托 (xìn tuō) - usually correct\r\n\r\n## 4. Structural Cleanup\r\n\r\n- Add proper punctuation (periods, commas) where ASR output lacks them\r\n- Split long run-on sentences at natural topic breaks\r\n- Format as flowing paragraphs, not timestamp-ordered fragments\r\n- Add section headings for topic shifts: \"洗盘核心目的\", \"三个核心指标\", \"三个信号\", etc.\r\n- Keep timestamps if user wants time-coded output (from .srt/.vtt)\r\n\r\n## 5. Quality Indicators\r\n\r\nAfter calibration, flag low-confidence sections:\r\n- Unclear audio sections (background noise, overlapping speech)\r\n- Rapid technical jargon sequences\r\n- Sections where multiple interpretations are plausible\r\n\r\n## 6. Multi-language Content\r\n\r\nFor bilingual content (Chinese + English):\r\n- Preserve English terms: PE ratio, MA, MACD, KDJ, Bollinger Bands\r\n- Mixed language phrases: \"比如 PE 20倍\", \"MACD 金叉\" are correct\r\n- Don't translate technical terms to Chinese\r\n\r\n## 7. Calibration Output Format\r\n\r\nAfter applying corrections, present as:\r\n- Clean prose with proper Chinese punctuation\r\n- Optional: show what was changed vs raw output\r\n- Optional: timestamp references from original SRT/VTT\r\n\r\n## 8. AI / Tech Domain Corrections (AI 及科技领域)\r\n\r\nFor videos about AI, semiconductors, and tech investment topics, watch for:\r\n\r\n### Chinese Company Names (whisper frequent errors)\r\n| Raw (Wrong) | Correct | Context |\r\n|-------------|---------|---------|\r\n| 中繼續創 | 中际旭创 | A股光模块龙头 |\r\n| 新益勝 | 新易盛 | A股光模块 |\r\n| 天賦通信 | 天孚通信 | A股通信 |\r\n| 元傑科技 | 源杰科技 | A股芯片 |\r\n| 阿力 | 阿里 | 阿里巴巴 |\r\n| Alley | 阿里 | Context: 腾讯、阿里、字节 |\r\n\r\n### AI Product & Term Names\r\n| Raw (Wrong) | Correct |\r\n|-------------|---------|\r\n| Deepseat | DeepSeek |\r\n| ChadGPT | ChatGPT |\r\n| Cloud | Claude (if Anthropic context) |\r\n| Moe價構 / 莫架构 | MoE架构 (Mixture of Experts) |\r\n| HPM | HBM (High Bandwidth Memory) |\r\n| 光膜块 | 光模块 (Optical Module) |\r\n| 夜冷 | 液冷 (Liquid Cooling) |\r\n| 巨深智能 | 具身智能 (Embodied AI) |\r\n| 端側推理 | 端侧推理 (On-device Inference) |\r\n| 推測解碼 | 推测解码 (Speculative Decoding) |\r\n| 主能板块 | 储能板块 (Energy Storage) |\r\n\r\n### Traditional → Simplified Chinese\r\nWhisper medium/large models commonly output traditional Chinese (繁體) for simplified content. Always convert:\r\n- 發→发, 來→来, 時→时, 會→会, 體→体, 報→报, 機→机\r\n- 線→线, 構→构, 購→购, 業→业, 電→电, 纜→缆, 銅→铜\r\n- 壹→一, 貳→二, 等等\n\nFile v1.0.8:skill-card.md\n\n## Description: <br>\nCross-platform video subtitle extraction using ASR that downloads audio from video URLs, transcribes speech with Whisper models, and can apply rule-based text calibration for Chinese financial and technical content. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[forhonourlx](https://clawhub.ai/user/forhonourlx) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and agents use this skill to extract subtitles or transcripts from video URLs and local audio when built-in captions are unavailable. It is especially oriented toward Chinese-language video transcription and domain-term cleanup. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill may install local dependencies and download Whisper models. <br>\nMitigation: Review dependency installation before running it, prefer an isolated environment, and confirm disk and memory requirements for the selected model. <br>\nRisk: The skill fetches media from video platforms and saves audio, video, transcript, and metadata files locally. <br>\nMitigation: Use a dedicated output directory, review platform permissions, and avoid processing media that should not be downloaded or retained. <br>\nRisk: Browser-cookie download workflows can use a logged-in browser session. <br>\nMitigation: Avoid browser cookie access unless necessary; prefer a dedicated account or narrowly scoped cookie file. <br>\n\n\n## Reference(s): <br>\n- [ASR Model Selection Guide](references/asr_models.md) <br>\n- [Text Calibration Guide for Chinese ASR Output](references/calibration_guide.md) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, shell commands, configuration, guidance] <br>\n**Output Format:** [Markdown guidance with shell commands and transcript files in TXT, SRT, VTT, and JSON formats] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [May save downloaded audio or video, generated transcripts, calibrated text, and pipeline metadata to a local output directory.] <br>\n\n## Skill Version(s): <br>\n1.0.8 (source: server release metadata and artifact changelog) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.0.8:LICENSE\n\nMIT No Attribution License\n\nCopyright (c) 2025 forhonourlx\n\nPermission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the \"Software\"), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.\n\nArchive v1.0.7: 10 files, 27066 bytes\n\nFiles: LICENSE (917b), references/asr_models.md (1918b), references/calibration_guide.md (4619b), scripts/calibrate.py (11701b), scripts/download_audio.py (11069b), scripts/install_deps.py (4890b), scripts/run.py (12908b), scripts/transcribe.py (9569b), SKILL.md (12670b), _meta.json (143b)\n\nFile v1.0.7:SKILL.md\n\n---\r\nname: video-subtitle-extractor\r\ndescription: |\r\n  Cross-platform video subtitle extraction using ASR (speech-to-text). Downloads audio from video URLs via yt-dlp, transcribes with openai-whisper (medium/large-v3, default: medium), and applies LLM-based text calibration for Chinese financial/technical content. Use when: (1) extracting subtitles from Bilibili, Xiaohongshu, YouTube, or any yt-dlp-supported platform, (2) the video has no built-in subtitles, (3) users say \"下载字幕\", \"提取字幕\", \"语音转文字\", \"视频转文字\", \"字幕提取\", \"ASR转写\", (4) needing to transcribe audio files to text, (5) working with Chinese-language video content requiring high-accuracy transcription. Automatically handles dependency installation (ffmpeg, yt-dlp, openai-whisper) and model downloads.\r\n---\r\n\r\n# Video Subtitle Extractor 🎬→📝\r\n\r\nCross-platform ASR subtitle extraction pipeline. Downloads audio from any yt-dlp-compatible video platform, transcribes with openai-whisper, and applies LLM-based text calibration for Chinese content.\r\n\r\n**Tested & verified** on Windows 11 with real Bilibili & Xiaohongshu videos (medium model, ~95% accuracy for Chinese).\r\n\r\n## Quick Start\r\n\r\n```bash\r\n# One-command full pipeline (Bilibili, Xiaohongshu, YouTube, etc.)\r\npython scripts/run.py <video_url> --model medium --language zh --output-dir ./output\r\n\r\n# Example: Xiaohongshu short link (auto-resolves redirect)\r\npython scripts/run.py https://xhslink.com/xxxxx --model medium --language zh\r\n\r\n# Download audio only\r\npython scripts/download_audio.py <video_url> <output_dir>\r\n\r\n# Download audio + video (keep both as middleware)\r\npython scripts/download_audio.py <video_url> <output_dir> --save-video --video-quality 1080\r\n\r\n# Transcribe existing audio\r\npython scripts/transcribe.py <audio_file> --model medium --language zh\r\n```\r\n\r\n## When to Use This Skill\r\n\r\nUse this skill when:\r\n1. The video has **no built-in subtitles** (Bilibili, Xiaohongshu, YouTube, etc.)\r\n2. You need **high-accuracy Chinese transcription** (~95% with medium model)\r\n3. You want **multiple output formats** (TXT, SRT, VTT, JSON)\r\n4. You need **LLM-assisted text calibration** for financial/technical terms\r\n5. The user says: \"下载字幕\", \"提取字幕\", \"语音转文字\", \"视频转文字\", \"字幕提取\", \"ASR转写\"\r\n\r\n## Workflow\r\n\r\n### Step 0: Install Dependencies (once)\r\n\r\n```bash\r\npython scripts/install_deps.py\r\n```\r\n\r\nAuto-detects OS and installs: ffmpeg (winget/brew/apt), yt-dlp (pip), openai-whisper (pip). Handles Windows ffmpeg path detection even when not in PATH.\r\n\r\n### Step 1: Download Audio\r\n\r\nRun `scripts/download_audio.py <url> [output_dir]`.\r\n\r\nUses yt-dlp to extract the best available audio format (m4a preferred). Supports Bilibili, YouTube, and 1800+ yt-dlp-compatible platforms. The script automatically detects ffmpeg even when not in system PATH.\r\n\r\n**Optional: Download video as middleware**\r\n\r\nAdd `--save-video` to persist the full video alongside audio:\r\n\r\n```bash\r\npython scripts/download_audio.py <url> --save-video --video-quality 1080\r\npython scripts/run.py <url> --save-video --video-quality 720 --calibrate\r\n```\r\n\r\n`--video-quality` preset table:\r\n\r\n| Preset | yt-dlp behaviour | Typical result |\r\n|--------|------------------|----------------|\r\n| `best` (default) | Highest available | 4K on YouTube, 480p on B站 (no login) |\r\n| `1080` | ≤1080p, fallback gracefully | 1080p where available |\r\n| `720` | ≤720p | Good balance for local storage |\r\n| `480` | ≤480p | Minimum acceptable for reference |\r\n| `360` | ≤360p | Extremely small files |\r\n| *raw string* | Direct yt-dlp format selector | Full flexibility |\r\n\r\n> **⚠️ B站 note**: Without login cookies, B站 caps at 480p. 720p+ requires `--cookies-from-browser`.\r\n\r\n**If download fails**: the video may require cookies. Try:\r\n```bash\r\nyt-dlp --cookies-from-browser chrome <url>\r\n```\r\n\r\n### Step 2: ASR Transcription\r\n\r\nRun `scripts/transcribe.py <audio> --model <size> --language <lang>`.\r\n\r\nModels are auto-downloaded on first use (disk space required):\r\n\r\n| Model | RAM | Disk | Speed | Quality | Best For |\r\n|-------|-----|------|-------|---------|----------|\r\n| `medium` | ~5GB | 1.42GB | ~165 fps | **~95%** ✅ | **Recommended (default)** |\r\n| `large-v3` | ~10GB | 2.88GB | ~80 fps | ~97% | Best accuracy |\r\n| `large-v3-turbo` | ~6GB | 1.6GB | ~120 fps | ~96% | Good balance |\r\n\r\n> **⚠️ Windows note**: `large-v3` needs >10GB RAM. If transcription fails, always check ffmpeg PATH first (see Troubleshooting).\r\n\r\nOutput formats: `txt`, `srt`, `vtt`, `json` (default: all).\r\n\r\nSee `references/asr_models.md` for full model comparison.\r\n\r\n### Step 3: Rule-Based Calibration\r\n\r\nAfter transcription, apply `calibrate.py` for mechanical corrections:\r\n\r\n```bash\r\n# Calibrate a raw .txt transcript\r\npython scripts/calibrate.py <output_dir>/<video_title>.txt\r\n# Output: <video_title>_calibrated.txt\r\n\r\n# Skip traditional→simplified conversion (already simplified input)\r\npython scripts/calibrate.py raw.txt --no-tradsimp\r\n```\r\n\r\n**Calibrate categories:** homophone fixes, financial/AI domain terms, company names, traditional→simplified (600+ chars).\r\nFor context-aware fixes (semantic errors, ambiguous names), use LLM review on top of rule-based output.\r\nSee `references/calibration_guide.md` for the full 80+ pattern library.\r\n\r\n### Step 4: Deliver Results\r\n\r\nPresent the calibrated text. Always include:\r\n- Model used (medium/large) and quality notes\r\n- Any sections with low confidence or unclear audio\r\n- Summary of corrections applied (counts by category)\r\n\r\n## Intermediate Artifacts & Step-by-Step Reuse\r\n\r\nEvery pipeline stage saves its output to disk. All artifacts persist in `output_dir/` after the run — no data is lost between stages.\r\n\r\n### Artifact Map\r\n\r\n| Stage | Artifact | Format | Filename pattern | Reusable alone? |\r\n|-------|----------|--------|------------------|-----------------|\r\n| Step 1 | Audio | `.m4a` | `<video_title>.m4a` | ✅ `download_audio.py` |\r\n| Step 1 | Video (optional) | `.mp4` | `<video_title>.mp4` | ✅ `download_audio.py --save-video` |\r\n| Step 2 | Transcript (raw) | `.txt` `.srt` `.vtt` `.json` | `<video_title>.txt` | ✅ `transcribe.py` |\r\n| Step 2 | Pipeline metadata | `.json` | `_pipeline_meta.json` | ✅ (reference only) |\r\n| Step 3 | Transcript (calibrated) | `_calibrated.txt` | `<title>_calibrated.txt` | ✅ `calibrate.py` |\r\n\r\n### Selective Reuse via `run.py` Flags\r\n\r\n```bash\r\n# Full pipeline (download + transcribe + calibrate)\r\npython scripts/run.py <url> --calibrate --output-dir ./out\r\n\r\n# Save video as middleware (downloads .m4a + .mp4)\r\npython scripts/run.py <url> --save-video --calibrate --output-dir ./out\r\n\r\n# Audio already exists → skip download, re-transcribe\r\npython scripts/run.py <url> --skip-download --output-dir ./out\r\n\r\n# Audio + transcript already exist → skip both, re-run calibration only\r\npython scripts/run.py <url> --skip-download --skip-transcribe --calibrate --output-dir ./out\r\n```\r\n\r\n`run.py` auto-detects existing artifacts by matching the audio filename base. It will warn and fall back to downloading/transcribing if no match is found.\r\n\r\n### Standalone Scripts\r\n\r\nEach stage has an independent entry point:\r\n\r\n```bash\r\n# Stage 1: Download audio only\r\npython scripts/download_audio.py <url> [output_dir] [filename]\r\n\r\n# Stage 1: Download audio + video at 720p (standalone)\r\npython scripts/download_audio.py <url> [output_dir] --save-video --video-quality 720\r\n\r\n# Stage 2: Transcribe existing audio\r\npython scripts/transcribe.py <audio.m4a> --model medium --language zh --output-dir ./out\r\n\r\n# Stage 3: Calibrate raw transcript (rule-based only)\r\npython scripts/calibrate.py <raw.txt> [--output <path>] [--no-tradsimp]\r\n```\r\n\r\n### Typical Reuse Scenarios\r\n\r\n**Scenario A — Change model, keep audio**\r\n```bash\r\n# Already have .m4a from previous run\r\npython scripts/run.py <url> --skip-download --model large-v3 --output-dir ./out\r\n```\r\n\r\n**Scenario B — Change language, keep audio + transcript**\r\n```bash\r\n# Have both .m4a and .txt; just recalibrate\r\npython scripts/run.py <url> --skip-download --skip-transcribe --calibrate --language en --output-dir ./out\r\n```\r\n\r\n**Scenario C — Batch calibrate multiple transcripts**\r\n```bash\r\n# Apply calibration to all raw .txt files in a directory\r\nGet-ChildItem .\\out\\*.txt | Where-Object { $_.Name -notmatch '_calibrated' } | ForEach-Object {\r\n    python scripts/calibrate.py $_.FullName\r\n}\r\n```\r\n\r\n## Platform Support\r\n\r\n| Platform | Status | Notes |\r\n|----------|--------|-------|\r\n| Bilibili | ✅ | Audio-only streams available without login. 720P+ video needs cookies. |\r\n| Xiaohongshu | ✅ | Full support via `XiaoHongShu` extractor. Short links (xhslink.com) auto-resolved. No cookies needed. |\r\n| YouTube | ✅ | Full support. Cookies may improve format selection. |\r\n| Douyin/TikTok | ✅ | Via yt-dlp |\r\n| All yt-dlp sites | ✅ | 1800+ supported platforms |\r\n\r\n## Extending with New ASR Models\r\n\r\n`scripts/transcribe.py` is designed for backend extensibility:\r\n\r\n1. Add model info to `MODEL_SIZES` dict\r\n2. Implement `transcribe_<backend>()` function\r\n3. Add CLI flag in argparse\r\n\r\n**Planned backends**: faster-whisper (CTranslate2), whisper.cpp (native C++), Cloud APIs (AssemblyAI, iFlytek).\r\n\r\n## Troubleshooting\r\n\r\n| Problem | Solution |\r\n|---------|----------|\r\n| SIGKILL / ffmpeg FileNotFoundError | ffmpeg not in PATH. Script auto-detects 7 common install locations (winget, scoop, chocolatey, manual). If ffmpeg is elsewhere, add its directory to system PATH. |\r\n| yt-dlp download fails | Update yt-dlp: `pip install -U yt-dlp`. Try with cookies. |\r\n| \"No subtitles found\" | Expected. This skill uses ASR, not built-in captions. |\r\n| ffmpeg not found | Run `install_deps.py` (handles Windows non-PATH detection). |\r\n| GPU not utilized | openai-whisper CPU-only by default. Install `faster-whisper` for GPU. |\r\n\r\n## Performance Benchmarks (Tested)\r\n\r\n| Video Duration | Model | Time | RAM Peak | Accuracy |\r\n|---------------|-------|------|----------|----------|\r\n| 6 min (Bilibili) | medium | ~4m 30s | ~6GB | ~95% |\r\n| 13 min (Bilibili) | medium | ~8m | ~6.5GB | ~95% |\r\n| 15 min (Bilibili) | medium | ~10m | ~5GB | **~95%** ✅ |\r\n| 9 min (Bilibili) | medium | ~4m | ~5GB | **~95%** ✅ |\r\n\r\nTested on Windows 11, Intel i7, 16GB RAM. Performance may vary by CPU speed.\r\n\r\n## Changelog\r\n\r\n### v1.0.7\r\n- **New**: `--video-quality` parameter — presets best/2160/1440/1080/720/480/360 + raw yt-dlp format support\r\n- **New**: `find_artifacts_in_dir()` replaces `find_audio_in_dir()` — caches both .m4a AND .mp4 on `--skip-download --save-video`\r\n- **Change**: Video download format selector now uses `height<=` filters (graceful fallback) instead of hardcoded mp4-only\r\n- **Improve**: `download_video()` now reports which quality preset is in use\r\n- **Improve**: Artifact map includes video (.mp4) with quality column\r\n- **Improve**: SKILL.md adds `--video-quality` usage table and B站 login caveats\r\n\r\n### v1.0.6\r\n- **New**: `--save-video` flag — download and persist full video (.mp4) alongside audio\r\n- **New**: `download_video()` function in download_audio.py (standalone: `--save-video`)\r\n- **Improve**: Artifact map now includes video (.mp4) as first-class middleware\r\n- **Improve**: `_pipeline_meta.json` includes `video_path` for full traceability\r\n\r\n### v1.0.5\r\n- **Remove**: `small` model from all backends (88-90% accuracy, too poor for production)\r\n- **Change**: Default model locked to `medium` (was medium in code, but docs still promoted small)\r\n- **Improve**: Model table and benchmarks now medium-only baseline\r\n\r\n### v1.0.2\r\n- **New**: Xiaohongshu (小红书) platform support — yt-dlp `XiaoHongShu` extractor\r\n- **New**: Short link auto-resolution (xhslink.com → full URL via redirect)\r\n- **Improve**: Platform support table now lists 小红书 explicitly\r\n- **Improve**: Quick Start examples include xhslink.com usage\r\n\r\n### v1.0.1\r\n- **Fix**: Expanded ffmpeg search paths from 3→7 (winget/scoop/chocolatey/ProgramFiles(x86))\r\n- **Fix**: `ensure_deps()` now injects ffmpeg into `os.environ['PATH']` on success\r\n- **Fix**: SIGKILL troubleshooting updated — root cause is ffmpeg PATH, not OOM\r\n- **Improve**: Auto-detect GPU (`torch.cuda.is_available()`) for fp16 support\r\n- **Improve**: `verbose=True` for real-time transcription progress visibility\r\n- **Improve**: More accurate error messages in dependency checks\r\n\r\n### v1.0.0\r\n- Initial release: download (yt-dlp) + transcribe (whisper) + calibrate (LLM) pipeline\r\n- 7 ffmpeg install path auto-detection\r\n- Multi-format output (TXT, SRT, VTT, JSON)\r\n- Platform support: Bilibili, YouTube, all yt-dlp sites\n\nFile v1.0.7:_meta.json\n\n{\n  \"ownerId\": \"kn71y79mxxwmzynnpj6e284kch82gwd2\",\n  \"slug\": \"video-subtitle-extractor\",\n  \"version\": \"1.0.7\",\n  \"publishedAt\": 1779751932753\n}\n\nFile v1.0.7:references/asr_models.md\n\n# ASR Model Selection Guide\r\n\r\n## Available Models\r\n\r\n| Model | RAM | Disk | Speed | Quality | Best For |\r\n|-------|-----|------|-------|---------|----------|\r\n| `small` | ~2GB | 461MB | Fast (~1min/6min audio) | Good | Quick tests, low-resource systems |\r\n| `medium` | ~5GB | 1.42GB | Medium (~3-5min) | High | **Recommended default** - best quality/speed ratio |\r\n| `large-v3` | ~10GB | 2.88GB | Slow (~10-20min) | Best | Production quality, needs high RAM |\r\n| `large-v3-turbo` | ~6GB | 1.6GB | Medium-fast | High | Good compromise, smaller than large-v3 |\r\n\r\n## Language-Specific Notes\r\n\r\n### Chinese (zh)\r\n- `medium`: Good for general Chinese content. Some errors on homophones and financial terms.\r\n- `large-v3`: Best Chinese accuracy, handles accents and domain terminology better.\r\n- Common errors: 同音字混淆 (硬扛→硬钢), 金融术语 (抛压→抛押, 交筹→焦愁), K线术语 (十字星→14星)\r\n\r\n### English (en)\r\n- `small`: Sufficient for clear English speech.\r\n- `medium`: Excellent accuracy for most content.\r\n\r\n## Memory Constraints\r\n\r\nOn Windows, `large-v3` may be killed (SIGKILL) on systems with <16GB RAM due to FP32 fallback.\r\nIf killed, fall back to `medium` or use `larger-v3-turbo`.\r\n\r\n## Future Model Compatibility\r\n\r\nThe `transcribe.py` script is designed for easy backend extension:\r\n- `faster-whisper`: CTranslate2 backend, more memory efficient\r\n- `whisper.cpp`: Native C++ implementation\r\n- `mlx-whisper`: Apple Silicon optimized\r\n- Cloud APIs: AssemblyAI, iFlytek, Whisper API\r\n\r\nTo add a new backend, implement a `transcribe_<backend>()` function in transcribe.py\r\nfollowing the same interface (audio_path, model_name, language, output_dir).\r\n\r\n## Model Auto-Download\r\n\r\nModels are downloaded automatically by openai-whisper on first use.\r\nCache location:\r\n- Windows: `C:\\Users\\<user>\\.cache\\whisper\\`\r\n- macOS: `~/Library/Caches/whisper/`\r\n- Linux: `~/.cache/whisper/`\n\nFile v1.0.7:references/calibration_guide.md\n\n# Text Calibration Guide for Chinese ASR Output\r\n\r\nCommon transcription errors and their corrections. Apply these patterns when calibrating whisper output for Chinese financial/technical content.\r\n\r\n## 1. Homophone Replacements (同音字混淆)\r\n\r\n| Raw (Wrong) | Correct | Example |\r\n|-------------|---------|---------|\r\n| 硬钢 | 硬扛 | 硬钢→硬扛 |\r\n| 抛押 | 抛压 | 消化抛押→消化抛压 |\r\n| 模两个月 | 磨两个月 | 横盘调整模两个月→磨两个月 |\r\n| 膜光短线 | 磨光短线 | 膜光短线→磨光短线 |\r\n| 流通骨 | 流通股 | 流通骨的换手→流通股的换手 |\r\n| 金接盘 | 新接盘 | 金接盘的成本→新接盘的成本 |\r\n| 拉伸 | 拉升 | 拉伸成本→拉升成本 |\r\n| 跟锋 | 跟风 | 跟锋买入→跟风买入 |\r\n| 微转 | 微赚 | 微转就抛售→微赚就抛售 |\r\n| 落带为安 | 落袋为安 | 落带为安→落袋为安 |\r\n| 互盘 | 护盘 | 主力互盘明显→主力护盘明显 |\r\n| 逼散互买 | 逼散户卖 | 洗盘本质是逼散互买→逼散户卖 |\r\n| 仅 | 有 | 仅有资金拖住→有资金托住 |\r\n| 军线 | 均线 | 关键军线→关键均线 |\r\n| 快有动作 | 快有动作 | Already correct, but watch for 快会→快会 |\r\n\r\n## 2. Financial Term Corrections\r\n\r\n| Raw | Correct | Context |\r\n|-----|---------|---------|\r\n| 交筹 | 交筹 | 慢慢就焦愁→慢慢就交筹 |\r\n| 再计 | 在即 | 拉升再计→拉升在即 |\r\n| 没装 | 没仓 | 根本没装→根本没仓 |\r\n| 利空 | 利空 | Already correct, verify |\r\n| K线收14星 | K线收十字星 | 14→十 |\r\n| 14星 | 十字星 | K线收十字星 |\r\n| 洗崩 | 洗崩 | Already correct (跌太多) |\r\n| 割肉 | 割肉 | Already correct |\r\n\r\n## 3. Domain Term Patterns\r\n\r\nWhisper often confuses financial jargon:\r\n- 洗盘 (xǐ pán) vs 洗盘 (same pronunciation but context-dependent)\r\n- 筹码 (chóu mǎ) - usually correct\r\n- 建仓 (jiàn cāng) - usually correct\r\n- 杠杆 (gàng gǎn) - usually correct\r\n- 信托 (xìn tuō) - usually correct\r\n\r\n## 4. Structural Cleanup\r\n\r\n- Add proper punctuation (periods, commas) where ASR output lacks them\r\n- Split long run-on sentences at natural topic breaks\r\n- Format as flowing paragraphs, not timestamp-ordered fragments\r\n- Add section headings for topic shifts: \"洗盘核心目的\", \"三个核心指标\", \"三个信号\", etc.\r\n- Keep timestamps if user wants time-coded output (from .srt/.vtt)\r\n\r\n## 5. Quality Indicators\r\n\r\nAfter calibration, flag low-confidence sections:\r\n- Unclear audio sections (background noise, overlapping speech)\r\n- Rapid technical jargon sequences\r\n- Sections where multiple interpretations are plausible\r\n\r\n## 6. Multi-language Content\r\n\r\nFor bilingual content (Chinese + English):\r\n- Preserve English terms: PE ratio, MA, MACD, KDJ, Bollinger Bands\r\n- Mixed language phrases: \"比如 PE 20倍\", \"MACD 金叉\" are correct\r\n- Don't translate technical terms to Chinese\r\n\r\n## 7. Calibration Output Format\r\n\r\nAfter applying corrections, present as:\r\n- Clean prose with proper Chinese punctuation\r\n- Optional: show what was changed vs raw output\r\n- Optional: timestamp references from original SRT/VTT\r\n\r\n## 8. AI / Tech Domain Corrections (AI 及科技领域)\r\n\r\nFor videos about AI, semiconductors, and tech investment topics, watch for:\r\n\r\n### Chinese Company Names (whisper frequent errors)\r\n| Raw (Wrong) | Correct | Context |\r\n|-------------|---------|---------|\r\n| 中繼續創 | 中际旭创 | A股光模块龙头 |\r\n| 新益勝 | 新易盛 | A股光模块 |\r\n| 天賦通信 | 天孚通信 | A股通信 |\r\n| 元傑科技 | 源杰科技 | A股芯片 |\r\n| 阿力 | 阿里 | 阿里巴巴 |\r\n| Alley | 阿里 | Context: 腾讯、阿里、字节 |\r\n\r\n### AI Product & Term Names\r\n| Raw (Wrong) | Correct |\r\n|-------------|---------|\r\n| Deepseat | DeepSeek |\r\n| ChadGPT | ChatGPT |\r\n| Cloud | Claude (if Anthropic context) |\r\n| Moe價構 / 莫架构 | MoE架构 (Mixture of Experts) |\r\n| HPM | HBM (High Bandwidth Memory) |\r\n| 光膜块 | 光模块 (Optical Module) |\r\n| 夜冷 | 液冷 (Liquid Cooling) |\r\n| 巨深智能 | 具身智能 (Embodied AI) |\r\n| 端側推理 | 端侧推理 (On-device Inference) |\r\n| 推測解碼 | 推测解码 (Speculative Decoding) |\r\n| 主能板块 | 储能板块 (Energy Storage) |\r\n\r\n### Traditional → Simplified Chinese\r\nWhisper medium/large models commonly output traditional Chinese (繁體) for simplified content. Always convert:\r\n- 發→发, 來→来, 時→时, 會→会, 體→体, 報→报, 機→机\r\n- 線→线, 構→构, 購→购, 業→业, 電→电, 纜→缆, 銅→铜\r\n- 壹→一, 貳→二, 等等\n\nFile v1.0.7:LICENSE\n\nMIT No Attribution License\n\nCopyright (c) 2025 forhonourlx\n\nPermission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the \"Software\"), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.\n\nArchive v1.0.6: 10 files, 25637 bytes\n\nFiles: LICENSE (917b), references/asr_models.md (1918b), references/calibration_guide.md (4619b), scripts/calibrate.py (11701b), scripts/download_audio.py (9351b), scripts/install_deps.py (4890b), scripts/run.py (11716b), scripts/transcribe.py (9569b), SKILL.md (10886b), _meta.json (143b)\n\nFile v1.0.6:SKILL.md\n\n---\r\nname: video-subtitle-extractor\r\ndescription: |\r\n  Cross-platform video subtitle extraction using ASR (speech-to-text). Downloads audio from video URLs via yt-dlp, transcribes with openai-whisper (medium/large-v3, default: medium), and applies LLM-based text calibration for Chinese financial/technical content. Use when: (1) extracting subtitles from Bilibili, Xiaohongshu, YouTube, or any yt-dlp-supported platform, (2) the video has no built-in subtitles, (3) users say \"下载字幕\", \"提取字幕\", \"语音转文字\", \"视频转文字\", \"字幕提取\", \"ASR转写\", (4) needing to transcribe audio files to text, (5) working with Chinese-language video content requiring high-accuracy transcription. Automatically handles dependency installation (ffmpeg, yt-dlp, openai-whisper) and model downloads.\r\n---\r\n\r\n# Video Subtitle Extractor 🎬→📝\r\n\r\nCross-platform ASR subtitle extraction pipeline. Downloads audio from any yt-dlp-compatible video platform, transcribes with openai-whisper, and applies LLM-based text calibration for Chinese content.\r\n\r\n**Tested & verified** on Windows 11 with real Bilibili & Xiaohongshu videos (medium model, ~95% accuracy for Chinese).\r\n\r\n## Quick Start\r\n\r\n```bash\r\n# One-command full pipeline (Bilibili, Xiaohongshu, YouTube, etc.)\r\npython scripts/run.py <video_url> --model medium --language zh --output-dir ./output\r\n\r\n# Example: Xiaohongshu short link (auto-resolves redirect)\r\npython scripts/run.py https://xhslink.com/xxxxx --model medium --language zh\r\n\r\n# Download audio only\r\npython scripts/download_audio.py <video_url> <output_dir>\r\n\r\n# Transcribe existing audio\r\npython scripts/transcribe.py <audio_file> --model medium --language zh\r\n```\r\n\r\n## When to Use This Skill\r\n\r\nUse this skill when:\r\n1. The video has **no built-in subtitles** (Bilibili, Xiaohongshu, YouTube, etc.)\r\n2. You need **high-accuracy Chinese transcription** (~95% with medium model)\r\n3. You want **multiple output formats** (TXT, SRT, VTT, JSON)\r\n4. You need **LLM-assisted text calibration** for financial/technical terms\r\n5. The user says: \"下载字幕\", \"提取字幕\", \"语音转文字\", \"视频转文字\", \"字幕提取\", \"ASR转写\"\r\n\r\n## Workflow\r\n\r\n### Step 0: Install Dependencies (once)\r\n\r\n```bash\r\npython scripts/install_deps.py\r\n```\r\n\r\nAuto-detects OS and installs: ffmpeg (winget/brew/apt), yt-dlp (pip), openai-whisper (pip). Handles Windows ffmpeg path detection even when not in PATH.\r\n\r\n### Step 1: Download Audio\r\n\r\nRun `scripts/download_audio.py <url> [output_dir]`.\r\n\r\nUses yt-dlp to extract the best available audio format (m4a preferred). Supports Bilibili, YouTube, and 1800+ yt-dlp-compatible platforms. The script automatically detects ffmpeg even when not in system PATH.\r\n\r\n**If download fails**: the video may require cookies. Try:\r\n```bash\r\nyt-dlp --cookies-from-browser chrome <url>\r\n```\r\n\r\n### Step 2: ASR Transcription\r\n\r\nRun `scripts/transcribe.py <audio> --model <size> --language <lang>`.\r\n\r\nModels are auto-downloaded on first use (disk space required):\r\n\r\n| Model | RAM | Disk | Speed | Quality | Best For |\r\n|-------|-----|------|-------|---------|----------|\r\n| `medium` | ~5GB | 1.42GB | ~165 fps | **~95%** ✅ | **Recommended (default)** |\r\n| `large-v3` | ~10GB | 2.88GB | ~80 fps | ~97% | Best accuracy |\r\n| `large-v3-turbo` | ~6GB | 1.6GB | ~120 fps | ~96% | Good balance |\r\n\r\n> **⚠️ Windows note**: `large-v3` needs >10GB RAM. If transcription fails, always check ffmpeg PATH first (see Troubleshooting).\r\n\r\nOutput formats: `txt`, `srt`, `vtt`, `json` (default: all).\r\n\r\nSee `references/asr_models.md` for full model comparison.\r\n\r\n### Step 3: Rule-Based Calibration\r\n\r\nAfter transcription, apply `calibrate.py` for mechanical corrections:\r\n\r\n```bash\r\n# Calibrate a raw .txt transcript\r\npython scripts/calibrate.py <output_dir>/<video_title>.txt\r\n# Output: <video_title>_calibrated.txt\r\n\r\n# Skip traditional→simplified conversion (already simplified input)\r\npython scripts/calibrate.py raw.txt --no-tradsimp\r\n```\r\n\r\n**Calibrate categories:** homophone fixes, financial/AI domain terms, company names, traditional→simplified (600+ chars).\r\nFor context-aware fixes (semantic errors, ambiguous names), use LLM review on top of rule-based output.\r\nSee `references/calibration_guide.md` for the full 80+ pattern library.\r\n\r\n### Step 4: Deliver Results\r\n\r\nPresent the calibrated text. Always include:\r\n- Model used (medium/large) and quality notes\r\n- Any sections with low confidence or unclear audio\r\n- Summary of corrections applied (counts by category)\r\n\r\n## Intermediate Artifacts & Step-by-Step Reuse\r\n\r\nEvery pipeline stage saves its output to disk. All artifacts persist in `output_dir/` after the run — no data is lost between stages.\r\n\r\n### Artifact Map\r\n\r\n| Stage | Artifact | Format | Filename pattern | Reusable alone? |\r\n|-------|----------|--------|------------------|-----------------|\r\n| Step 1 | Audio | `.m4a` | `<video_title>.m4a` | ✅ `download_audio.py` |\r\n| Step 2 | Transcript (raw) | `.txt` `.srt` `.vtt` `.json` | `<video_title>.txt` | ✅ `transcribe.py` |\r\n| Step 2 | Pipeline metadata | `.json` | `_pipeline_meta.json` | ✅ (reference only) |\r\n| Step 3 | Transcript (calibrated) | `_calibrated.txt` | `<title>_calibrated.txt` | ✅ `calibrate.py` |\r\n\r\n### Selective Reuse via `run.py` Flags\r\n\r\n```bash\r\n# Full pipeline (download + transcribe + calibrate)\r\npython scripts/run.py <url> --calibrate --output-dir ./out\r\n\r\n# Save video as middleware (downloads .m4a + .mp4)\r\npython scripts/run.py <url> --save-video --calibrate --output-dir ./out\r\n\r\n# Audio already exists → skip download, re-transcribe\r\npython scripts/run.py <url> --skip-download --output-dir ./out\r\n\r\n# Audio + transcript already exist → skip both, re-run calibration only\r\npython scripts/run.py <url> --skip-download --skip-transcribe --calibrate --output-dir ./out\r\n```\r\n\r\n`run.py` auto-detects existing artifacts by matching the audio filename base. It will warn and fall back to downloading/transcribing if no match is found.\r\n\r\n### Standalone Scripts\r\n\r\nEach stage has an independent entry point:\r\n\r\n```bash\r\n# Stage 1: Download audio only\r\npython scripts/download_audio.py <url> [output_dir] [filename]\r\n\r\n# Stage 1: Download audio + video (standalone)\r\npython scripts/download_audio.py <url> [output_dir] --save-video\r\n\r\n# Stage 2: Transcribe existing audio\r\npython scripts/transcribe.py <audio.m4a> --model medium --language zh --output-dir ./out\r\n\r\n# Stage 3: Calibrate raw transcript (rule-based only)\r\npython scripts/calibrate.py <raw.txt> [--output <path>] [--no-tradsimp]\r\n```\r\n\r\n### Typical Reuse Scenarios\r\n\r\n**Scenario A — Change model, keep audio**\r\n```bash\r\n# Already have .m4a from previous run\r\npython scripts/run.py <url> --skip-download --model large-v3 --output-dir ./out\r\n```\r\n\r\n**Scenario B — Change language, keep audio + transcript**\r\n```bash\r\n# Have both .m4a and .txt; just recalibrate\r\npython scripts/run.py <url> --skip-download --skip-transcribe --calibrate --language en --output-dir ./out\r\n```\r\n\r\n**Scenario C — Batch calibrate multiple transcripts**\r\n```bash\r\n# Apply calibration to all raw .txt files in a directory\r\nGet-ChildItem .\\out\\*.txt | Where-Object { $_.Name -notmatch '_calibrated' } | ForEach-Object {\r\n    python scripts/calibrate.py $_.FullName\r\n}\r\n```\r\n\r\n## Platform Support\r\n\r\n| Platform | Status | Notes |\r\n|----------|--------|-------|\r\n| Bilibili | ✅ | Audio-only streams available without login. 720P+ video needs cookies. |\r\n| Xiaohongshu | ✅ | Full support via `XiaoHongShu` extractor. Short links (xhslink.com) auto-resolved. No cookies needed. |\r\n| YouTube | ✅ | Full support. Cookies may improve format selection. |\r\n| Douyin/TikTok | ✅ | Via yt-dlp |\r\n| All yt-dlp sites | ✅ | 1800+ supported platforms |\r\n\r\n## Extending with New ASR Models\r\n\r\n`scripts/transcribe.py` is designed for backend extensibility:\r\n\r\n1. Add model info to `MODEL_SIZES` dict\r\n2. Implement `transcribe_<backend>()` function\r\n3. Add CLI flag in argparse\r\n\r\n**Planned backends**: faster-whisper (CTranslate2), whisper.cpp (native C++), Cloud APIs (AssemblyAI, iFlytek).\r\n\r\n## Troubleshooting\r\n\r\n| Problem | Solution |\r\n|---------|----------|\r\n| SIGKILL / ffmpeg FileNotFoundError | ffmpeg not in PATH. Script auto-detects 7 common install locations (winget, scoop, chocolatey, manual). If ffmpeg is elsewhere, add its directory to system PATH. |\r\n| yt-dlp download fails | Update yt-dlp: `pip install -U yt-dlp`. Try with cookies. |\r\n| \"No subtitles found\" | Expected. This skill uses ASR, not built-in captions. |\r\n| ffmpeg not found | Run `install_deps.py` (handles Windows non-PATH detection). |\r\n| GPU not utilized | openai-whisper CPU-only by default. Install `faster-whisper` for GPU. |\r\n\r\n## Performance Benchmarks (Tested)\r\n\r\n| Video Duration | Model | Time | RAM Peak | Accuracy |\r\n|---------------|-------|------|----------|----------|\r\n| 6 min (Bilibili) | medium | ~4m 30s | ~6GB | ~95% |\r\n| 13 min (Bilibili) | medium | ~8m | ~6.5GB | ~95% |\r\n| 15 min (Bilibili) | medium | ~10m | ~5GB | **~95%** ✅ |\r\n| 9 min (Bilibili) | medium | ~4m | ~5GB | **~95%** ✅ |\r\n\r\nTested on Windows 11, Intel i7, 16GB RAM. Performance may vary by CPU speed.\r\n\r\n## Changelog\r\n\r\n### v1.0.6\r\n- **New**: `--save-video` flag — download and persist full video (.mp4) alongside audio\r\n- **New**: `download_video()` function in download_audio.py (standalone: `--save-video`)\r\n- **Improve**: Artifact map now includes video (.mp4) as first-class middleware\r\n- **Improve**: `_pipeline_meta.json` includes `video_path` for full traceability\r\n\r\n### v1.0.5\r\n- **Remove**: `small` model from all backends (88-90% accuracy, too poor for production)\r\n- **Change**: Default model locked to `medium` (was medium in code, but docs still promoted small)\r\n- **Improve**: Model table and benchmarks now medium-only baseline\r\n\r\n### v1.0.2\r\n- **New**: Xiaohongshu (小红书) platform support — yt-dlp `XiaoHongShu` extractor\r\n- **New**: Short link auto-resolution (xhslink.com → full URL via redirect)\r\n- **Improve**: Platform support table now lists 小红书 explicitly\r\n- **Improve**: Quick Start examples include xhslink.com usage\r\n\r\n### v1.0.1\r\n- **Fix**: Expanded ffmpeg search paths from 3→7 (winget/scoop/chocolatey/ProgramFiles(x86))\r\n- **Fix**: `ensure_deps()` now injects ffmpeg into `os.environ['PATH']` on success\r\n- **Fix**: SIGKILL troubleshooting updated — root cause is ffmpeg PATH, not OOM\r\n- **Improve**: Auto-detect GPU (`torch.cuda.is_available()`) for fp16 support\r\n- **Improve**: `verbose=True` for real-time transcription progress visibility\r\n- **Improve**: More accurate error messages in dependency checks\r\n\r\n### v1.0.0\r\n- Initial release: download (yt-dlp) + transcribe (whisper) + calibrate (LLM) pipeline\r\n- 7 ffmpeg install path auto-detection\r\n- Multi-format output (TXT, SRT, VTT, JSON)\r\n- Platform support: Bilibili, YouTube, all yt-dlp sites\n\nFile v1.0.6:_meta.json\n\n{\n  \"ownerId\": \"kn71y79mxxwmzynnpj6e284kch82gwd2\",\n  \"slug\": \"video-subtitle-extractor\",\n  \"version\": \"1.0.6\",\n  \"publishedAt\": 1779701687074\n}\n\nFile v1.0.6:references/asr_models.md\n\n# ASR Model Selection Guide\r\n\r\n## Available Models\r\n\r\n| Model | RAM | Disk | Speed | Quality | Best For |\r\n|-------|-----|------|-------|---------|----------|\r\n| `small` | ~2GB | 461MB | Fast (~1min/6min audio) | Good | Quick tests, low-resource systems |\r\n| `medium` | ~5GB | 1.42GB | Medium (~3-5min) | High | **Recommended default** - best quality/speed ratio |\r\n| `large-v3` | ~10GB | 2.88GB | Slow (~10-20min) | Best | Production quality, needs high RAM |\r\n| `large-v3-turbo` | ~6GB | 1.6GB | Medium-fast | High | Good compromise, smaller than large-v3 |\r\n\r\n## Language-Specific Notes\r\n\r\n### Chinese (zh)\r\n- `medium`: Good for general Chinese content. Some errors on homophones and financial terms.\r\n- `large-v3`: Best Chinese accuracy, handles accents and domain terminology better.\r\n- Common errors: 同音字混淆 (硬扛→硬钢), 金融术语 (抛压→抛押, 交筹→焦愁), K线术语 (十字星→14星)\r\n\r\n### English (en)\r\n- `small`: Sufficient for clear English speech.\r\n- `medium`: Excellent accuracy for most content.\r\n\r\n## Memory Constraints\r\n\r\nOn Windows, `large-v3` may be killed (SIGKILL) on systems with <16GB RAM due to FP32 fallback.\r\nIf killed, fall back to `medium` or use `larger-v3-turbo`.\r\n\r\n## Future Model Compatibility\r\n\r\nThe `transcribe.py` script is designed for easy backend extension:\r\n- `faster-whisper`: CTranslate2 backend, more memory efficient\r\n- `whisper.cpp`: Native C++ implementation\r\n- `mlx-whisper`: Apple Silicon optimized\r\n- Cloud APIs: AssemblyAI, iFlytek, Whisper API\r\n\r\nTo add a new backend, implement a `transcribe_<backend>()` function in transcribe.py\r\nfollowing the same interface (audio_path, model_name, language, output_dir).\r\n\r\n## Model Auto-Download\r\n\r\nModels are downloaded automatically by openai-whisper on first use.\r\nCache location:\r\n- Windows: `C:\\Users\\<user>\\.cache\\whisper\\`\r\n- macOS: `~/Library/Caches/whisper/`\r\n- Linux: `~/.cache/whisper/`\n\nFile v1.0.6:references/calibration_guide.md\n\n# Text Calibration Guide for Chinese ASR Output\r\n\r\nCommon transcription errors and their corrections. Apply these patterns when calibrating whisper output for Chinese financial/technical content.\r\n\r\n## 1. Homophone Replacements (同音字混淆)\r\n\r\n| Raw (Wrong) | Correct | Example |\r\n|-------------|---------|---------|\r\n| 硬钢 | 硬扛 | 硬钢→硬扛 |\r\n| 抛押 | 抛压 | 消化抛押→消化抛压 |\r\n| 模两个月 | 磨两个月 | 横盘调整模两个月→磨两个月 |\r\n| 膜光短线 | 磨光短线 | 膜光短线→磨光短线 |\r\n| 流通骨 | 流通股 | 流通骨的换手→流通股的换手 |\r\n| 金接盘 | 新接盘 | 金接盘的成本→新接盘的成本 |\r\n| 拉伸 | 拉升 | 拉伸成本→拉升成本 |\r\n| 跟锋 | 跟风 | 跟锋买入→跟风买入 |\r\n| 微转 | 微赚 | 微转就抛售→微赚就抛售 |\r\n| 落带为安 | 落袋为安 | 落带为安→落袋为安 |\r\n| 互盘 | 护盘 | 主力互盘明显→主力护盘明显 |\r\n| 逼散互买 | 逼散户卖 | 洗盘本质是逼散互买→逼散户卖 |\r\n| 仅 | 有 | 仅有资金拖住→有资金托住 |\r\n| 军线 | 均线 | 关键军线→关键均线 |\r\n| 快有动作 | 快有动作 | Already correct, but watch for 快会→快会 |\r\n\r\n## 2. Financial Term Corrections\r\n\r\n| Raw | Correct | Context |\r\n|-----|---------|---------|\r\n| 交筹 | 交筹 | 慢慢就焦愁→慢慢就交筹 |\r\n| 再计 | 在即 | 拉升再计→拉升在即 |\r\n| 没装 | 没仓 | 根本没装→根本没仓 |\r\n| 利空 | 利空 | Already correct, verify |\r\n| K线收14星 | K线收十字星 | 14→十 |\r\n| 14星 | 十字星 | K线收十字星 |\r\n| 洗崩 | 洗崩 | Already correct (跌太多) |\r\n| 割肉 | 割肉 | Already correct |\r\n\r\n## 3. Domain Term Patterns\r\n\r\nWhisper often confuses financial jargon:\r\n- 洗盘 (xǐ pán) vs 洗盘 (same pronunciation but context-dependent)\r\n- 筹码 (chóu mǎ) - usually correct\r\n- 建仓 (jiàn cāng) - usually correct\r\n- 杠杆 (gàng gǎn) - usually correct\r\n- 信托 (xìn tuō) - usually correct\r\n\r\n## 4. Structural Cleanup\r\n\r\n- Add proper punctuation (periods, commas) where ASR output lacks them\r\n- Split long run-on sentences at natural topic breaks\r\n- Format as flowing paragraphs, not timestamp-ordered fragments\r\n- Add section headings for topic shifts: \"洗盘核心目的\", \"三个核心指标\", \"三个信号\", etc.\r\n- Keep timestamps if user wants time-coded output (from .srt/.vtt)\r\n\r\n## 5. Quality Indicators\r\n\r\nAfter calibration, flag low-confidence sections:\r\n- Unclear audio sections (background noise, overlapping speech)\r\n- Rapid technical jargon sequences\r\n- Sections where multiple interpretations are plausible\r\n\r\n## 6. Multi-language Content\r\n\r\nFor bilingual content (Chinese + English):\r\n- Preserve English terms: PE ratio, MA, MACD, KDJ, Bollinger Bands\r\n- Mixed language phrases: \"比如 PE 20倍\", \"MACD 金叉\" are correct\r\n- Don't translate technical terms to Chinese\r\n\r\n## 7. Calibration Output Format\r\n\r\nAfter applying corrections, present as:\r\n- Clean prose with proper Chinese punctuation\r\n- Optional: show what was changed vs raw output\r\n- Optional: timestamp references from original SRT/VTT\r\n\r\n## 8. AI / Tech Domain Corrections (AI 及科技领域)\r\n\r\nFor videos about AI, semiconductors, and tech investment topics, watch for:\r\n\r\n### Chinese Company Names (whisper frequent errors)\r\n| Raw (Wrong) | Correct | Context |\r\n|-------------|---------|---------|\r\n| 中繼續創 | 中际旭创 | A股光模块龙头 |\r\n| 新益勝 | 新易盛 | A股光模块 |\r\n| 天賦通信 | 天孚通信 | A股通信 |\r\n| 元傑科技 | 源杰科技 | A股芯片 |\r\n| 阿力 | 阿里 | 阿里巴巴 |\r\n| Alley | 阿里 | Context: 腾讯、阿里、字节 |\r\n\r\n### AI Product & Term Names\r\n| Raw (Wrong) | Correct |\r\n|-------------|---------|\r\n| Deepseat | DeepSeek |\r\n| ChadGPT | ChatGPT |\r\n| Cloud | Claude (if Anthropic context) |\r\n| Moe價構 / 莫架构 | MoE架构 (Mixture of Experts) |\r\n| HPM | HBM (High Bandwidth Memory) |\r\n| 光膜块 | 光模块 (Optical Module) |\r\n| 夜冷 | 液冷 (Liquid Cooling) |\r\n| 巨深智能 | 具身智能 (Embodied AI) |\r\n| 端側推理 | 端侧推理 (On-device Inference) |\r\n| 推測解碼 | 推测解码 (Speculative Decoding) |\r\n| 主能板块 | 储能板块 (Energy Storage) |\r\n\r\n### Traditional → Simplified Chinese\r\nWhisper medium/large models commonly output traditional Chinese (繁體) for simplified content. Always convert:\r\n- 發→发, 來→来, 時→时, 會→会, 體→体, 報→报, 機→机\r\n- 線→线, 構→构, 購→购, 業→业, 電→电, 纜→缆, 銅→铜\r\n- 壹→一, 貳→二, 等等\n\nFile v1.0.6:LICENSE\n\nMIT No Attribution License\n\nCopyright (c) 2025 forhonourlx\n\nPermission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the \"Software\"), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.\n\nArchive v1.0.5: 10 files, 24857 bytes\n\nFiles: LICENSE (917b), references/asr_models.md (1918b), references/calibration_guide.md (4619b), scripts/calibrate.py (11701b), scripts/download_audio.py (5515b), scripts/install_deps.py (4890b), scripts/run.py (10836b), scripts/transcribe.py (9569b), SKILL.md (10285b), _meta.json (143b)\n\nFile v1.0.5:SKILL.md\n\n---\r\nname: video-subtitle-extractor\r\ndescription: |\r\n  Cross-platform video subtitle extraction using ASR (speech-to-text). Downloads audio from video URLs via yt-dlp, transcribes with openai-whisper (medium/large-v3, default: medium), and applies LLM-based text calibration for Chinese financial/technical content. Use when: (1) extracting subtitles from Bilibili, Xiaohongshu, YouTube, or any yt-dlp-supported platform, (2) the video has no built-in subtitles, (3) users say \"下载字幕\", \"提取字幕\", \"语音转文字\", \"视频转文字\", \"字幕提取\", \"ASR转写\", (4) needing to transcribe audio files to text, (5) working with Chinese-language video content requiring high-accuracy transcription. Automatically handles dependency installation (ffmpeg, yt-dlp, openai-whisper) and model downloads.\r\n---\r\n\r\n# Video Subtitle Extractor 🎬→📝\r\n\r\nCross-platform ASR subtitle extraction pipeline. Downloads audio from any yt-dlp-compatible video platform, transcribes with openai-whisper, and applies LLM-based text calibration for Chinese content.\r\n\r\n**Tested & verified** on Windows 11 with real Bilibili & Xiaohongshu videos (medium model, ~95% accuracy for Chinese).\r\n\r\n## Quick Start\r\n\r\n```bash\r\n# One-command full pipeline (Bilibili, Xiaohongshu, YouTube, etc.)\r\npython scripts/run.py <video_url> --model medium --language zh --output-dir ./output\r\n\r\n# Example: Xiaohongshu short link (auto-resolves redirect)\r\npython scripts/run.py https://xhslink.com/xxxxx --model medium --language zh\r\n\r\n# Download audio only\r\npython scripts/download_audio.py <video_url> <output_dir>\r\n\r\n# Transcribe existing audio\r\npython scripts/transcribe.py <audio_file> --model medium --language zh\r\n```\r\n\r\n## When to Use This Skill\r\n\r\nUse this skill when:\r\n1. The video has **no built-in subtitles** (Bilibili, Xiaohongshu, YouTube, etc.)\r\n2. You need **high-accuracy Chinese transcription** (~95% with medium model)\r\n3. You want **multiple output formats** (TXT, SRT, VTT, JSON)\r\n4. You need **LLM-assisted text calibration** for financial/technical terms\r\n5. The user says: \"下载字幕\", \"提取字幕\", \"语音转文字\", \"视频转文字\", \"字幕提取\", \"ASR转写\"\r\n\r\n## Workflow\r\n\r\n### Step 0: Install Dependencies (once)\r\n\r\n```bash\r\npython scripts/install_deps.py\r\n```\r\n\r\nAuto-detects OS and installs: ffmpeg (winget/brew/apt), yt-dlp (pip), openai-whisper (pip). Handles Windows ffmpeg path detection even when not in PATH.\r\n\r\n### Step 1: Download Audio\r\n\r\nRun `scripts/download_audio.py <url> [output_dir]`.\r\n\r\nUses yt-dlp to extract the best available audio format (m4a preferred). Supports Bilibili, YouTube, and 1800+ yt-dlp-compatible platforms. The script automatically detects ffmpeg even when not in system PATH.\r\n\r\n**If download fails**: the video may require cookies. Try:\r\n```bash\r\nyt-dlp --cookies-from-browser chrome <url>\r\n```\r\n\r\n### Step 2: ASR Transcription\r\n\r\nRun `scripts/transcribe.py <audio> --model <size> --language <lang>`.\r\n\r\nModels are auto-downloaded on first use (disk space required):\r\n\r\n| Model | RAM | Disk | Speed | Quality | Best For |\r\n|-------|-----|------|-------|---------|----------|\r\n| `medium` | ~5GB | 1.42GB | ~165 fps | **~95%** ✅ | **Recommended (default)** |\r\n| `large-v3` | ~10GB | 2.88GB | ~80 fps | ~97% | Best accuracy |\r\n| `large-v3-turbo` | ~6GB | 1.6GB | ~120 fps | ~96% | Good balance |\r\n\r\n> **⚠️ Windows note**: `large-v3` needs >10GB RAM. If transcription fails, always check ffmpeg PATH first (see Troubleshooting).\r\n\r\nOutput formats: `txt`, `srt`, `vtt`, `json` (default: all).\r\n\r\nSee `references/asr_models.md` for full model comparison.\r\n\r\n### Step 3: Rule-Based Calibration\r\n\r\nAfter transcription, apply `calibrate.py` for mechanical corrections:\r\n\r\n```bash\r\n# Calibrate a raw .txt transcript\r\npython scripts/calibrate.py <output_dir>/<video_title>.txt\r\n# Output: <video_title>_calibrated.txt\r\n\r\n# Skip traditional→simplified conversion (already simplified input)\r\npython scripts/calibrate.py raw.txt --no-tradsimp\r\n```\r\n\r\n**Calibrate categories:** homophone fixes, financial/AI domain terms, company names, traditional→simplified (600+ chars).\r\nFor context-aware fixes (semantic errors, ambiguous names), use LLM review on top of rule-based output.\r\nSee `references/calibration_guide.md` for the full 80+ pattern library.\r\n\r\n### Step 4: Deliver Results\r\n\r\nPresent the calibrated text. Always include:\r\n- Model used (medium/large) and quality notes\r\n- Any sections with low confidence or unclear audio\r\n- Summary of corrections applied (counts by category)\r\n\r\n## Intermediate Artifacts & Step-by-Step Reuse\r\n\r\nEvery pipeline stage saves its output to disk. All artifacts persist in `output_dir/` after the run — no data is lost between stages.\r\n\r\n### Artifact Map\r\n\r\n| Stage | Artifact | Format | Filename pattern | Reusable alone? |\r\n|-------|----------|--------|------------------|-----------------|\r\n| Step 1 | Audio | `.m4a` | `<video_title>.m4a` | ✅ `download_audio.py` |\r\n| Step 2 | Transcript (raw) | `.txt` `.srt` `.vtt` `.json` | `<video_title>.txt` | ✅ `transcribe.py` |\r\n| Step 2 | Pipeline metadata | `.json` | `_pipeline_meta.json` | ✅ (reference only) |\r\n| Step 3 | Transcript (calibrated) | `_calibrated.txt` | `<title>_calibrated.txt` | ✅ `calibrate.py` |\r\n\r\n### Selective Reuse via `run.py` Flags\r\n\r\n```bash\r\n# Full pipeline (download + transcribe + calibrate)\r\npython scripts/run.py <url> --calibrate --output-dir ./out\r\n\r\n# Audio already exists → skip download, re-transcribe\r\npython scripts/run.py <url> --skip-download --output-dir ./out\r\n\r\n# Audio + transcript already exist → skip both, re-run calibration only\r\npython scripts/run.py <url> --skip-download --skip-transcribe --calibrate --output-dir ./out\r\n```\r\n\r\n`run.py` auto-detects existing artifacts by matching the audio filename base. It will warn and fall back to downloading/transcribing if no match is found.\r\n\r\n### Standalone Scripts\r\n\r\nEach stage has an independent entry point:\r\n\r\n```bash\r\n# Stage 1: Download audio only\r\npython scripts/download_audio.py <url> [output_dir] [filename]\r\n\r\n# Stage 2: Transcribe existing audio\r\npython scripts/transcribe.py <audio.m4a> --model medium --language zh --output-dir ./out\r\n\r\n# Stage 3: Calibrate raw transcript (rule-based only)\r\npython scripts/calibrate.py <raw.txt> [--output <path>] [--no-tradsimp]\r\n```\r\n\r\n### Typical Reuse Scenarios\r\n\r\n**Scenario A — Change model, keep audio**\r\n```bash\r\n# Already have .m4a from previous run\r\npython scripts/run.py <url> --skip-download --model large-v3 --output-dir ./out\r\n```\r\n\r\n**Scenario B — Change language, keep audio + transcript**\r\n```bash\r\n# Have both .m4a and .txt; just recalibrate\r\npython scripts/run.py <url> --skip-download --skip-transcribe --calibrate --language en --output-dir ./out\r\n```\r\n\r\n**Scenario C — Batch calibrate multiple transcripts**\r\n```bash\r\n# Apply calibration to all raw .txt files in a directory\r\nGet-ChildItem .\\out\\*.txt | Where-Object { $_.Name -notmatch '_calibrated' } | ForEach-Object {\r\n    python scripts/calibrate.py $_.FullName\r\n}\r\n```\r\n\r\n## Platform Support\r\n\r\n| Platform | Status | Notes |\r\n|----------|--------|-------|\r\n| Bilibili | ✅ | Audio-only streams available without login. 720P+ video needs cookies. |\r\n| Xiaohongshu | ✅ | Full support via `XiaoHongShu` extractor. Short links (xhslink.com) auto-resolved. No cookies needed. |\r\n| YouTube | ✅ | Full support. Cookies may improve format selection. |\r\n| Douyin/TikTok | ✅ | Via yt-dlp |\r\n| All yt-dlp sites | ✅ | 1800+ supported platforms |\r\n\r\n## Extending with New ASR Models\r\n\r\n`scripts/transcribe.py` is designed for backend extensibility:\r\n\r\n1. Add model info to `MODEL_SIZES` dict\r\n2. Implement `transcribe_<backend>()` function\r\n3. Add CLI flag in argparse\r\n\r\n**Planned backends**: faster-whisper (CTranslate2), whisper.cpp (native C++), Cloud APIs (AssemblyAI, iFlytek).\r\n\r\n## Troubleshooting\r\n\r\n| Problem | Solution |\r\n|---------|----------|\r\n| SIGKILL / ffmpeg FileNotFoundError | ffmpeg not in PATH. Script auto-detects 7 common install locations (winget, scoop, chocolatey, manual). If ffmpeg is elsewhere, add its directory to system PATH. |\r\n| yt-dlp download fails | Update yt-dlp: `pip install -U yt-dlp`. Try with cookies. |\r\n| \"No subtitles found\" | Expected. This skill uses ASR, not built-in captions. |\r\n| ffmpeg not found | Run `install_deps.py` (handles Windows non-PATH detection). |\r\n| GPU not utilized | openai-whisper CPU-only by default. Install `faster-whisper` for GPU. |\r\n\r\n## Performance Benchmarks (Tested)\r\n\r\n| Video Duration | Model | Time | RAM Peak | Accuracy |\r\n|---------------|-------|------|----------|----------|\r\n| 6 min (Bilibili) | medium | ~4m 30s | ~6GB | ~95% |\r\n| 13 min (Bilibili) | medium | ~8m | ~6.5GB | ~95% |\r\n| 15 min (Bilibili) | medium | ~10m | ~5GB | **~95%** ✅ |\r\n| 9 min (Bilibili) | medium | ~4m | ~5GB | **~95%** ✅ |\r\n\r\nTested on Windows 11, Intel i7, 16GB RAM. Performance may vary by CPU speed.\r\n\r\n## Changelog\r\n\r\n### v1.0.4\r\n- **Remove**: `small` model from all backends (88-90% accuracy, too poor for production)\r\n- **Change**: Default model locked to `medium` (was medium in code, but docs still promoted small)\r\n- **Improve**: Model table and benchmarks now medium-only baseline\r\n\r\n### v1.0.2\r\n- **New**: Xiaohongshu (小红书) platform support — yt-dlp `XiaoHongShu` extractor\r\n- **New**: Short link auto-resolution (xhslink.com → full URL via redirect)\r\n- **Improve**: Platform support table now lists 小红书 explicitly\r\n- **Improve**: Quick Start examples include xhslink.com usage\r\n\r\n### v1.0.1\r\n- **Fix**: Expanded ffmpeg search paths from 3→7 (winget/scoop/chocolatey/ProgramFiles(x86))\r\n- **Fix**: `ensure_deps()` now injects ffmpeg into `os.environ['PATH']` on success\r\n- **Fix**: SIGKILL troubleshooting updated — root cause is ffmpeg PATH, not OOM\r\n- **Improve**: Auto-detect GPU (`torch.cuda.is_available()`) for fp16 support\r\n- **Improve**: `verbose=True` for real-time transcription progress visibility\r\n- **Improve**: More accurate error messages in dependency checks\r\n\r\n### v1.0.0\r\n- Initial release: download (yt-dlp) + transcribe (whisper) + calibrate (LLM) pipeline\r\n- 7 ffmpeg install path auto-detection\r\n- Multi-format output (TXT, SRT, VTT, JSON)\r\n- Platform support: Bilibili, YouTube, all yt-dlp sites\n\nFile v1.0.5:_meta.json\n\n{\n  \"ownerId\": \"kn71y79mxxwmzynnpj6e284kch82gwd2\",\n  \"slug\": \"video-subtitle-extractor\",\n  \"version\": \"1.0.5\",\n  \"publishedAt\": 1779692617506\n}\n\nFile v1.0.5:references/asr_models.md\n\n# ASR Model Selection Guide\r\n\r\n## Available Models\r\n\r\n| Model | RAM | Disk | Speed | Quality | Best For |\r\n|-------|-----|------|-------|---------|----------|\r\n| `small` | ~2GB | 461MB | Fast (~1min/6min audio) | Good | Quick tests, low-resource systems |\r\n| `medium` | ~5GB | 1.42GB | Medium (~3-5min) | High | **Recommended default** - best quality/speed ratio |\r\n| `large-v3` | ~10GB | 2.88GB | Slow (~10-20min) | Best | Production quality, needs high RAM |\r\n| `large-v3-turbo` | ~6GB | 1.6GB | Medium-fast | High | Good compromise, smaller than large-v3 |\r\n\r\n## Language-Specific Notes\r\n\r\n### Chinese (zh)\r\n- `medium`: Good for general Chinese content. Some errors on homophones and financial terms.\r\n- `large-v3`: Best Chinese accuracy, handles accents and domain terminology better.\r\n- Common errors: 同音字混淆 (硬扛→硬钢), 金融术语 (抛压→抛押, 交筹→焦愁), K线术语 (十字星→14星)\r\n\r\n### English (en)\r\n- `small`: Sufficient for clear English speech.\r\n- `medium`: Excellent accuracy for most content.\r\n\r\n## Memory Constraints\r\n\r\nOn Windows, `large-v3` may be killed (SIGKILL) on systems with <16GB RAM due to FP32 fallback.\r\nIf killed, fall back to `medium` or use `larger-v3-turbo`.\r\n\r\n## Future Model Compatibility\r\n\r\nThe `transcribe.py` script is designed for easy backend extension:\r\n- `faster-whisper`: CTranslate2 backend, more memory efficient\r\n- `whisper.cpp`: Native C++ implementation\r\n- `mlx-whisper`: Apple Silicon optimized\r\n- Cloud APIs: AssemblyAI, iFlytek, Whisper API\r\n\r\nTo add a new backend, implement a `transcribe_<backend>()` function in transcribe.py\r\nfollowing the same interface (audio_path, model_name, language, output_dir).\r\n\r\n## Model Auto-Download\r\n\r\nModels are downloaded automatically by openai-whisper on first use.\r\nCache location:\r\n- Windows: `C:\\Users\\<user>\\.cache\\whisper\\`\r\n- macOS: `~/Library/Caches/whisper/`\r\n- Linux: `~/.cache/whisper/`\n\nFile v1.0.5:references/calibration_guide.md\n\n# Text Calibration Guide for Chinese ASR Output\r\n\r\nCommon transcription errors and their corrections. Apply these patterns when calibrating whisper output for Chinese financial/technical content.\r\n\r\n## 1. Homophone Replacements (同音字混淆)\r\n\r\n| Raw (Wrong) | Correct | Example |\r\n|-------------|---------|---------|\r\n| 硬钢 | 硬扛 | 硬钢→硬扛 |\r\n| 抛押 | 抛压 | 消化抛押→消化抛压 |\r\n| 模两个月 | 磨两个月 | 横盘调整模两个月→磨两个月 |\r\n| 膜光短线 | 磨光短线 | 膜光短线→磨光短线 |\r\n| 流通骨 | 流通股 | 流通骨的换手→流通股的换手 |\r\n| 金接盘 | 新接盘 | 金接盘的成本→新接盘的成本 |\r\n| 拉伸 | 拉升 | 拉伸成本→拉升成本 |\r\n| 跟锋 | 跟风 | 跟锋买入→跟风买入 |\r\n| 微转 | 微赚 | 微转就抛售→微赚就抛售 |\r\n| 落带为安 | 落袋为安 | 落带为安→落袋为安 |\r\n| 互盘 | 护盘 | 主力互盘明显→主力护盘明显 |\r\n| 逼散互买 | 逼散户卖 | 洗盘本质是逼散互买→逼散户卖 |\r\n| 仅 | 有 | 仅有资金拖住→有资金托住 |\r\n| 军线 | 均线 | 关键军线→关键均线 |\r\n| 快有动作 | 快有动作 | Already correct, but watch for 快会→快会 |\r\n\r\n## 2. Financial Term Corrections\r\n\r\n| Raw | Correct | Context |\r\n|-----|---------|---------|\r\n| 交筹 | 交筹 | 慢慢就焦愁→慢慢就交筹 |\r\n| 再计 | 在即 | 拉升再计→拉升在即 |\r\n| 没装 | 没仓 | 根本没装→根本没仓 |\r\n| 利空 | 利空 | Already correct, verify |\r\n| K线收14星 | K线收十字星 | 14→十 |\r\n| 14星 | 十字星 | K线收十字星 |\r\n| 洗崩 | 洗崩 | Already correct (跌太多) |\r\n| 割肉 | 割肉 | Already correct |\r\n\r\n## 3. Domain Term Patterns\r\n\r\nWhisper often confuses financial jargon:\r\n- 洗盘 (xǐ pán) vs 洗盘 (same pronunciation but context-dependent)\r\n- 筹码 (chóu mǎ) - usually correct\r\n- 建仓 (jiàn cāng) - usually correct\r\n- 杠杆 (gàng gǎn) - usually correct\r\n- 信托 (xìn tuō) - usually correct\r\n\r\n## 4. Structural Cleanup\r\n\r\n- Add proper punctuation (periods, commas) where ASR output lacks them\r\n- Split long run-on sentences at natural topic breaks\r\n- Format as flowing paragraphs, not timestamp-ordered fragments\r\n- Add section headings for topic shifts: \"洗盘核心目的\", \"三个核心指标\", \"三个信号\", etc.\r\n- Keep timestamps if user wants time-coded output (from .srt/.vtt)\r\n\r\n## 5. Quality Indicators\r\n\r\nAfter calibration, flag low-confidence sections:\r\n- Unclear audio sections (background noise, overlapping speech)\r\n- Rapid technical jargon sequences\r\n- Sections where multiple interpretations are plausible\r\n\r\n## 6. Multi-language Content\r\n\r\nFor bilingual content (Chinese + English):\r\n- Preserve English terms: PE ratio, MA, MACD, KDJ, Bollinger Bands\r\n- Mixed language phrases: \"比如 PE 20倍\", \"MACD 金叉\" are correct\r\n- Don't translate technical terms to Chinese\r\n\r\n## 7. Calibration Output Format\r\n\r\nAfter applying corrections, present as:\r\n- Clean prose with proper Chinese punctuation\r\n- Optional: show what was changed vs raw output\r\n- Optional: timestamp references from original SRT/VTT\r\n\r\n## 8. AI / Tech Domain Corrections (AI 及科技领域)\r\n\r\nFor videos about AI, semiconductors, and tech investment topics, watch for:\r\n\r\n### Chinese Company Names (whisper frequent errors)\r\n| Raw (Wrong) | Correct | Context |\r\n|-------------|---------|---------|\r\n| 中繼續創 | 中际旭创 | A股光模块龙头 |\r\n| 新益勝 | 新易盛 | A股光模块 |\r\n| 天賦通信 | 天孚通信 | A股通信 |\r\n| 元傑科技 | 源杰科技 | A股芯片 |\r\n| 阿力 | 阿里 | 阿里巴巴 |\r\n| Alley | 阿里 | Context: 腾讯、阿里、字节 |\r\n\r\n### AI Product & Term Names\r\n| Raw (Wrong) | Correct |\r\n|-------------|---------|\r\n| Deepseat | DeepSeek |\r\n| ChadGPT | ChatGPT |\r\n| Cloud | Claude (if Anthropic context) |\r\n| Moe價構 / 莫架构 | MoE架构 (Mixture of Experts) |\r\n| HPM | HBM (High Bandwidth Memory) |\r\n| 光膜块 | 光模块 (Optical Module) |\r\n| 夜冷 | 液冷 (Liquid Cooling) |\r\n| 巨深智能 | 具身智能 (Embodied AI) |\r\n| 端側推理 | 端侧推理 (On-device Inference) |\r\n| 推測解碼 | 推测解码 (Speculative Decoding) |\r\n| 主能板块 | 储能板块 (Energy Storage) |\r\n\r\n### Traditional → Simplified Chinese\r\nWhisper medium/large models commonly output traditional Chinese (繁體) for simplified content. Always convert:\r\n- 發→发, 來→来, 時→时, 會→会, 體→体, 報→报, 機→机\r\n- 線→线, 構→构, 購→购, 業→业, 電→电, 纜→缆, 銅→铜\r\n- 壹→一, 貳→二, 等等\n\nFile v1.0.5:LICENSE\n\nMIT No Attribution License\n\nCopyright (c) 2025 forhonourlx\n\nPermission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the \"Software\"), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.\n\nArchive v1.0.4: 9 files, 17568 bytes\n\nFiles: LICENSE (917b), references/asr_models.md (1918b), references/calibration_guide.md (4619b), scripts/download_audio.py (5515b), scripts/install_deps.py (4890b), scripts/run.py (6332b), scripts/transcribe.py (9569b), SKILL.md (7829b), _meta.json (143b)\n\nFile v1.0.4:SKILL.md\n\n---\r\nname: video-subtitle-extractor\r\ndescription: |\r\n  Cross-platform video subtitle extraction using ASR (speech-to-text). Downloads audio from video URLs via yt-dlp, transcribes with openai-whisper (medium/large-v3, default: medium), and applies LLM-based text calibration for Chinese financial/technical content. Use when: (1) extracting subtitles from Bilibili, Xiaohongshu, YouTube, or any yt-dlp-supported platform, (2) the video has no built-in subtitles, (3) users say \"下载字幕\", \"提取字幕\", \"语音转文字\", \"视频转文字\", \"字幕提取\", \"ASR转写\", (4) needing to transcribe audio files to text, (5) working with Chinese-language video content requiring high-accuracy transcription. Automatically handles dependency installation (ffmpeg, yt-dlp, openai-whisper) and model downloads.\r\n---\r\n\r\n# Video Subtitle Extractor 🎬→📝\r\n\r\nCross-platform ASR subtitle extraction pipeline. Downloads audio from any yt-dlp-compatible video platform, transcribes with openai-whisper, and applies LLM-based text calibration for Chinese content.\r\n\r\n**Tested & verified** on Windows 11 with real Bilibili & Xiaohongshu videos (medium model, ~95% accuracy for Chinese).\r\n\r\n## Quick Start\r\n\r\n```bash\r\n# One-command full pipeline (Bilibili, Xiaohongshu, YouTube, etc.)\r\npython scripts/run.py <video_url> --model medium --language zh --output-dir ./output\r\n\r\n# Example: Xiaohongshu short link (auto-resolves redirect)\r\npython scripts/run.py https://xhslink.com/xxxxx --model medium --language zh\r\n\r\n# Download audio only\r\npython scripts/download_audio.py <video_url> <output_dir>\r\n\r\n# Transcribe existing audio\r\npython scripts/transcribe.py <audio_file> --model medium --language zh\r\n```\r\n\r\n## When to Use This Skill\r\n\r\nUse this skill when:\r\n1. The video has **no built-in subtitles** (Bilibili, Xiaohongshu, YouTube, etc.)\r\n2. You need **high-accuracy Chinese transcription** (~95% with medium model)\r\n3. You want **multiple output formats** (TXT, SRT, VTT, JSON)\r\n4. You need **LLM-assisted text calibration** for financial/technical terms\r\n5. The user says: \"下载字幕\", \"提取字幕\", \"语音转文字\", \"视频转文字\", \"字幕提取\", \"ASR转写\"\r\n\r\n## Workflow\r\n\r\n### Step 0: Install Dependencies (once)\r\n\r\n```bash\r\npython scripts/install_deps.py\r\n```\r\n\r\nAuto-detects OS and installs: ffmpeg (winget/brew/apt), yt-dlp (pip), openai-whisper (pip). Handles Windows ffmpeg path detection even when not in PATH.\r\n\r\n### Step 1: Download Audio\r\n\r\nRun `scripts/download_audio.py <url> [output_dir]`.\r\n\r\nUses yt-dlp to extract the best available audio format (m4a preferred). Supports Bilibili, YouTube, and 1800+ yt-dlp-compatible platforms. The script automatically detects ffmpeg even when not in system PATH.\r\n\r\n**If download fails**: the video may require cookies. Try:\r\n```bash\r\nyt-dlp --cookies-from-browser chrome <url>\r\n```\r\n\r\n### Step 2: ASR Transcription\r\n\r\nRun `scripts/transcribe.py <audio> --model <size> --language <lang>`.\r\n\r\nModels are auto-downloaded on first use (disk space required):\r\n\r\n| Model | RAM | Disk | Speed | Quality | Best For |\r\n|-------|-----|------|-------|---------|----------|\r\n| `medium` | ~5GB | 1.42GB | ~165 fps | **~95%** ✅ | **Recommended (default)** |\r\n| `large-v3` | ~10GB | 2.88GB | ~80 fps | ~97% | Best accuracy |\r\n| `large-v3-turbo` | ~6GB | 1.6GB | ~120 fps | ~96% | Good balance |\r\n\r\n> **⚠️ Windows note**: `large-v3` needs >10GB RAM. If transcription fails, always check ffmpeg PATH first (see Troubleshooting).\r\n\r\nOutput formats: `txt`, `srt`, `vtt`, `json` (default: all).\r\n\r\nSee `references/asr_models.md` for full model comparison.\r\n\r\n### Step 3: LLM Text Calibration\r\n\r\nAfter transcription, read the `.txt` output and apply corrections. Key calibration categories:\r\n\r\n1. **Homophone fixes** (同音字): 硬钢→硬扛, 模→磨, 骨→股\r\n2. **Company/product names**: Deepseat→DeepSeek, 中繼續創→中际旭创, HPM→HBM\r\n3. **Financial terms**: 抛押→抛压, 护盘 (not 互盘), 筹码, K线收十字星 (not 14星)\r\n4. **Common substitutions**: 跟锋→跟风, 微转→微赚, 落带为安→落袋为安\r\n5. **Traditional→Simplified**: If model outputs traditional Chinese, convert to simplified\r\n6. **Structural cleanup**: Add paragraph breaks at topic shifts, format as prose\r\n\r\nSee `references/calibration_guide.md` for the full 30+ pattern library.\r\n\r\n### Step 4: Deliver Results\r\n\r\nPresent the calibrated text. Always include:\r\n- Model used (medium/large) and quality notes\r\n- Any sections with low confidence or unclear audio\r\n- Summary of corrections applied (counts by category)\r\n\r\n## Platform Support\r\n\r\n| Platform | Status | Notes |\r\n|----------|--------|-------|\r\n| Bilibili | ✅ | Audio-only streams available without login. 720P+ video needs cookies. |\r\n| Xiaohongshu | ✅ | Full support via `XiaoHongShu` extractor. Short links (xhslink.com) auto-resolved. No cookies needed. |\r\n| YouTube | ✅ | Full support. Cookies may improve format selection. |\r\n| Douyin/TikTok | ✅ | Via yt-dlp |\r\n| All yt-dlp sites | ✅ | 1800+ supported platforms |\r\n\r\n## Extending with New ASR Models\r\n\r\n`scripts/transcribe.py` is designed for backend extensibility:\r\n\r\n1. Add model info to `MODEL_SIZES` dict\r\n2. Implement `transcribe_<backend>()` function\r\n3. Add CLI flag in argparse\r\n\r\n**Planned backends**: faster-whisper (CTranslate2), whisper.cpp (native C++), Cloud APIs (AssemblyAI, iFlytek).\r\n\r\n## Troubleshooting\r\n\r\n| Problem | Solution |\r\n|---------|----------|\r\n| SIGKILL / ffmpeg FileNotFoundError | ffmpeg not in PATH. Script auto-detects 7 common install locations (winget, scoop, chocolatey, manual). If ffmpeg is elsewhere, add its directory to system PATH. |\r\n| yt-dlp download fails | Update yt-dlp: `pip install -U yt-dlp`. Try with cookies. |\r\n| \"No subtitles found\" | Expected. This skill uses ASR, not built-in captions. |\r\n| ffmpeg not found | Run `install_deps.py` (handles Windows non-PATH detection). |\r\n| GPU not utilized | openai-whisper CPU-only by default. Install `faster-whisper` for GPU. |\r\n\r\n## Performance Benchmarks (Tested)\r\n\r\n| Video Duration | Model | Time | RAM Peak | Accuracy |\r\n|---------------|-------|------|----------|----------|\r\n| 6 min (Bilibili) | medium | ~4m 30s | ~6GB | ~95% |\r\n| 13 min (Bilibili) | medium | ~8m | ~6.5GB | ~95% |\r\n| 15 min (Bilibili) | medium | ~10m | ~5GB | **~95%** ✅ |\r\n| 9 min (Bilibili) | medium | ~4m | ~5GB | **~95%** ✅ |\r\n\r\nTested on Windows 11, Intel i7, 16GB RAM. Performance may vary by CPU speed.\r\n\r\n## Changelog\r\n\r\n### v1.0.4\r\n- **Remove**: `small` model from all backends (88-90% accuracy, too poor for production)\r\n- **Change**: Default model locked to `medium` (was medium in code, but docs still promoted small)\r\n- **Improve**: Model table and benchmarks now medium-only baseline\r\n\r\n### v1.0.2\r\n- **New**: Xiaohongshu (小红书) platform support — yt-dlp `XiaoHongShu` extractor\r\n- **New**: Short link auto-resolution (xhslink.com → full URL via redirect)\r\n- **Improve**: Platform support table now lists 小红书 explicitly\r\n- **Improve**: Quick Start examples include xhslink.com usage\r\n\r\n### v1.0.1\r\n- **Fix**: Expanded ffmpeg search paths from 3→7 (winget/scoop/chocolatey/ProgramFiles(x86))\r\n- **Fix**: `ensure_deps()` now injects ffmpeg into `os.environ['PATH']` on success\r\n- **Fix**: SIGKILL troubleshooting updated — root cause is ffmpeg PATH, not OOM\r\n- **Improve**: Auto-detect GPU (`torch.cuda.is_available()`) for fp16 support\r\n- **Improve**: `verbose=True` for real-time transcription progress visibility\r\n- **Improve**: More accurate error messages in dependency checks\r\n\r\n### v1.0.0\r\n- Initial release: download (yt-dlp) + transcribe (whisper) + calibrate (LLM) pipeline\r\n- 7 ffmpeg install path auto-detection\r\n- Multi-format output (TXT, SRT, VTT, JSON)\r\n- Platform support: Bilibili, YouTube, all yt-dlp sites\n\nFile v1.0.4:_meta.json\n\n{\n  \"ownerId\": \"kn71y79mxxwmzynnpj6e284kch82gwd2\",\n  \"slug\": \"video-subtitle-extractor\",\n  \"version\": \"1.0.4\",\n  \"publishedAt\": 1779632657978\n}\n\nFile v1.0.4:references/asr_models.md\n\n# ASR Model Selection Guide\r\n\r\n## Available Models\r\n\r\n| Model | RAM | Disk | Speed | Quality | Best For |\r\n|-------|-----|------|-------|---------|----------|\r\n| `small` | ~2GB | 461MB | Fast (~1min/6min audio) | Good | Quick tests, low-resource systems |\r\n| `medium` | ~5GB | 1.42GB | Medium (~3-5min) | High | **Recommended default** - best quality/speed ratio |\r\n| `large-v3` | ~10GB | 2.88GB | Slow (~10-20min) | Best | Production quality, needs high RAM |\r\n| `large-v3-turbo` | ~6GB | 1.6GB | Medium-fast | High | Good compromise, smaller than large-v3 |\r\n\r\n## Language-Specific Notes\r\n\r\n### Chinese (zh)\r\n- `medium`: Good for general Chinese content. Some errors on homophones and financial terms.\r\n- `large-v3`: Best Chinese accuracy, handles accents and domain terminology better.\r\n- Common errors: 同音字混淆 (硬扛→硬钢), 金融术语 (抛压→抛押, 交筹→焦愁), K线术语 (十字星→14星)\r\n\r\n### English (en)\r\n- `small`: Sufficient for clear English speech.\r\n- `medium`: Excellent accuracy for most content.\r\n\r\n## Memory Constraints\r\n\r\nOn Windows, `large-v3` may be killed (SIGKILL) on systems with <16GB RAM due to FP32 fallback.\r\nIf killed, fall back to `medium` or use `larger-v3-turbo`.\r\n\r\n## Future Model Compatibility\r\n\r\nThe `transcribe.py` script is designed for easy backend extension:\r\n- `faster-whisper`: CTranslate2 backend, more memory efficient\r\n- `whisper.cpp`: Native C++ implementation\r\n- `mlx-whisper`: Apple Silicon optimized\r\n- Cloud APIs: AssemblyAI, iFlytek, Whisper API\r\n\r\nTo add a new backend, implement a `transcribe_<backend>()` function in transcribe.py\r\nfollowing the same interface (audio_path, model_name, language, output_dir).\r\n\r\n## Model Auto-Download\r\n\r\nModels are downloaded automatically by openai-whisper on first use.\r\nCache location:\r\n- Windows: `C:\\Users\\<user>\\.cache\\whisper\\`\r\n- macOS: `~/Library/Caches/whisper/`\r\n- Linux: `~/.cache/whisper/`\n\nFile v1.0.4:references/calibration_guide.md\n\n# Text Calibration Guide for Chinese ASR Output\r\n\r\nCommon transcription errors and their corrections. Apply these patterns when calibrating whisper output for Chinese financial/technical content.\r\n\r\n## 1. Homophone Replacements (同音字混淆)\r\n\r\n| Raw (Wrong) | Correct | Example |\r\n|-------------|---------|---------|\r\n| 硬钢 | 硬扛 | 硬钢→硬扛 |\r\n| 抛押 | 抛压 | 消化抛押→消化抛压 |\r\n| 模两个月 | 磨两个月 | 横盘调整模两个月→磨两个月 |\r\n| 膜光短线 | 磨光短线 | 膜光短线→磨光短线 |\r\n| 流通骨 | 流通股 | 流通骨的换手→流通股的换手 |\r\n| 金接盘 | 新接盘 | 金接盘的成本→新接盘的成本 |\r\n| 拉伸 | 拉升 | 拉伸成本→拉升成本 |\r\n| 跟锋 | 跟风 | 跟锋买入→跟风买入 |\r\n| 微转 | 微赚 | 微转就抛售→微赚就抛售 |\r\n| 落带为安 | 落袋为安 | 落带为安→落袋为安 |\r\n| 互盘 | 护盘 | 主力互盘明显→主力护盘明显 |\r\n| 逼散互买 | 逼散户卖 | 洗盘本质是逼散互买→逼散户卖 |\r\n| 仅 | 有 | 仅有资金拖住→有资金托住 |\r\n| 军线 | 均线 | 关键军线→关键均线 |\r\n| 快有动作 | 快有动作 | Already correct, but watch for 快会→快会 |\r\n\r\n## 2. Financial Term Corrections\r\n\r\n| Raw | Correct | Context |\r\n|-----|---------|---------|\r\n| 交筹 | 交筹 | 慢慢就焦愁→慢慢就交筹 |\r\n| 再计 | 在即 | 拉升再计→拉升在即 |\r\n| 没装 | 没仓 | 根本没装→根本没仓 |\r\n| 利空 | 利空 | Already correct, verify |\r\n| K线收14星 | K线收十字星 | 14→十 |\r\n| 14星 | 十字星 | K线收十字星 |\r\n| 洗崩 | 洗崩 | Already correct (跌太多) |\r\n| 割肉 | 割肉 | Already correct |\r\n\r\n## 3. Domain Term Patterns\r\n\r\nWhisper often confuses financial jargon:\r\n- 洗盘 (xǐ pán) vs 洗盘 (same pronunciation but context-dependent)\r\n- 筹码 (chóu mǎ) - usually correct\r\n- 建仓 (jiàn cāng) - usually correct\r\n- 杠杆 (gàng gǎn) - usually correct\r\n- 信托 (xìn tuō) - usually correct\r\n\r\n## 4. Structural Cleanup\r\n\r\n- Add proper punctuation (periods, commas) where ASR output lacks them\r\n- Split long run-on sentences at natural topic breaks\r\n- Format as flowing paragraphs, not timestamp-ordered fragments\r\n- Add section headings for topic shifts: \"洗盘核心目的\", \"三个核心指标\", \"三个信号\", etc.\r\n- Keep timestamps if user wants time-coded output (from .srt/.vtt)\r\n\r\n## 5. Quality Indicators\r\n\r\nAfter calibration, flag low-confidence sections:\r\n- Unclear audio sections (background noise, overlapping speech)\r\n- Rapid technical jargon sequences\r\n- Sections where multiple interpretations are plausible\r\n\r\n## 6. Multi-language Content\r\n\r\nFor bilingual content (Chinese + English):\r\n- Preserve English terms: PE ratio, MA, MACD, KDJ, Bollinger Bands\r\n- Mixed language phrases: \"比如 PE 20倍\", \"MACD 金叉\" are correct\r\n- Don't translate technical terms to Chinese\r\n\r\n## 7. Calibration Output Format\r\n\r\nAfter applying corrections, present as:\r\n- Clean prose with proper Chinese punctuation\r\n- Optional: show what was changed vs raw output\r\n- Optional: timestamp references from original SRT/VTT\r\n\r\n## 8. AI / Tech Domain Corrections (AI 及科技领域)\r\n\r\nFor videos about AI, semiconductors, and tech investment topics, watch for:\r\n\r\n### Chinese Company Names (whisper frequent errors)\r\n| Raw (Wrong) | Correct | Context |\r\n|-------------|---------|---------|\r\n| 中繼續創 | 中际旭创 | A股光模块龙头 |\r\n| 新益勝 | 新易盛 | A股光模块 |\r\n| 天賦通信 | 天孚通信 | A股通信 |\r\n| 元傑科技 | 源杰科技 | A股芯片 |\r\n| 阿力 | 阿里 | 阿里巴巴 |\r\n| Alley | 阿里 | Context: 腾讯、阿里、字节 |\r\n\r\n### AI Product & Term Names\r\n| Raw (Wrong) | Correct |\r\n|-------------|---------|\r\n| Deepseat | DeepSeek |\r\n| ChadGPT | ChatGPT |\r\n| Cloud | Claude (if Anthropic context) |\r\n| Moe價構 / 莫架构 | MoE架构 (Mixture of Experts) |\r\n| HPM | HBM (High Bandwidth Memory) |\r\n| 光膜块 | 光模块 (Optical Module) |\r\n| 夜冷 | 液冷 (Liquid Cooling) |\r\n| 巨深智能 | 具身智能 (Embodied AI) |\r\n| 端側推理 | 端侧推理 (On-device Inference) |\r\n| 推測解碼 | 推测解码 (Speculative Decoding) |\r\n| 主能板块 | 储能板块 (Energy Storage) |\r\n\r\n### Traditional → Simplified Chinese\r\nWhisper medium/large models commonly output traditional Chinese (繁體) for simplified content. Always convert:\r\n- 發→发, 來→来, 時→时, 會→会, 體→体, 報→报, 機→机\r\n- 線→线, 構→构, 購→购, 業→业, 電→电, 纜→缆, 銅→铜\r\n- 壹→一, 貳→二, 等等\n\nFile v1.0.4:LICENSE\n\nMIT No Attribution License\n\nCopyright (c) 2025 forhonourlx\n\nPermission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the \"Software\"), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.\n\nArchive v1.0.3: 9 files, 19550 bytes\n\nFiles: LICENSE (917b), references/asr_models.md (3157b), references/calibration_guide.md (4619b), scripts/download_audio.py (5515b), scripts/install_deps.py (4890b), scripts/run.py (6624b), scripts/transcribe.py (14376b), SKILL.md (9265b), _meta.json (143b)\n\nFile v1.0.3:SKILL.md\n\n---\r\nname: video-subtitle-extractor\r\ndescription: |\r\n  Cross-platform video subtitle extraction using ASR (speech-to-text). Downloads audio from video URLs via yt-dlp, transcribes with openai-whisper (small/medium/large-v3), and applies LLM-based text calibration for Chinese financial/technical content. Use when: (1) extracting subtitles from Bilibili, Xiaohongshu, YouTube, or any yt-dlp-supported platform, (2) the video has no built-in subtitles, (3) users say \"下载字幕\", \"提取字幕\", \"语音转文字\", \"视频转文字\", \"字幕提取\", \"ASR转写\", (4) needing to transcribe audio files to text, (5) working with Chinese-language video content requiring high-accuracy transcription. Automatically handles dependency installation (ffmpeg, yt-dlp, openai-whisper) and model downloads.\r\n---\r\n\r\n# Video Subtitle Extractor 🎬→📝\r\n\r\nCross-platform ASR subtitle extraction pipeline. Downloads audio from any yt-dlp-compatible video platform, transcribes with openai-whisper, and applies LLM-based text calibration for Chinese content.\r\n\r\n**Tested & verified** on Windows 11 with real Bilibili & Xiaohongshu videos (medium model, ~95% accuracy for Chinese).\r\n\r\n## Quick Start\r\n\r\n```bash\r\n# One-command full pipeline (Bilibili, Xiaohongshu, YouTube, etc.)\r\npython scripts/run.py <video_url> --model medium --language zh --output-dir ./output\r\n\r\n# Example: Xiaohongshu short link (auto-resolves redirect)\r\npython scripts/run.py https://xhslink.com/xxxxx --model medium --language zh\r\n\r\n# Download audio only\r\npython scripts/download_audio.py <video_url> <output_dir>\r\n\r\n# Transcribe existing audio\r\npython scripts/transcribe.py <audio_file> --model medium --language zh\r\n```\r\n\r\n## When to Use This Skill\r\n\r\nUse this skill when:\r\n1. The video has **no built-in subtitles** (Bilibili, Xiaohongshu, YouTube, etc.)\r\n2. You need **high-accuracy Chinese transcription** (~95% with medium model)\r\n3. You want **multiple output formats** (TXT, SRT, VTT, JSON)\r\n4. You need **LLM-assisted text calibration** for financial/technical terms\r\n5. The user says: \"下载字幕\", \"提取字幕\", \"语音转文字\", \"视频转文字\", \"字幕提取\", \"ASR转写\"\r\n\r\n## Workflow\r\n\r\n### Step 0: Install Dependencies (once)\r\n\r\n```bash\r\npython scripts/install_deps.py\r\n```\r\n\r\nAuto-detects OS and installs: ffmpeg (winget/brew/apt), yt-dlp (pip), openai-whisper (pip). Handles Windows ffmpeg path detection even when not in PATH.\r\n\r\n### Step 1: Download Audio\r\n\r\nRun `scripts/download_audio.py <url> [output_dir]`.\r\n\r\nUses yt-dlp to extract the best available audio format (m4a preferred). Supports Bilibili, YouTube, and 1800+ yt-dlp-compatible platforms. The script automatically detects ffmpeg even when not in system PATH.\r\n\r\n**If download fails**: the video may require cookies. Try:\r\n```bash\r\nyt-dlp --cookies-from-browser chrome <url>\r\n```\r\n\r\n### Step 2: ASR Transcription\r\n\r\nRun `scripts/transcribe.py <audio> --model <size> --language <lang>`.\r\n\r\nModels are auto-downloaded on first use (disk space required):\r\n\r\n### openai-whisper backend\r\n\r\n| Model | RAM | Disk | Speed | Quality | Best For |\r\n|-------|-----|------|-------|---------|----------|\r\n| `tiny` | ~1GB | 75MB | fastest | low | Testing |\r\n| `medium` | ~5GB | 1.42GB | ~165 fps | **~95%** ✅ | **Recommended** |\r\n| `large-v3` | ~10GB | 2.88GB | ~80 fps | ~97% | Best accuracy |\r\n| `large-v3-turbo` | ~6GB | 1.6GB | ~120 fps | ~96% | Good balance |\r\n\r\n> **Note**: `small` model removed from openai-whisper (88-90% accuracy, superseded). Use faster-whisper backend for lightweight option.\r\n\r\n### faster-whisper backend (CTranslate2, 4x faster, 50% less RAM)\r\n\r\n| Model | RAM | Disk | Speed | Quality | Best For |\r\n|-------|-----|------|-------|---------|----------|\r\n| `small` | ~1.5GB | 461MB | ~1800 fps | ~90% | **Lightweight** ⭐ |\r\n| `medium` | ~3GB | 1.42GB | ~600 fps | **~95%** ✅ | **Recommended** |\r\n| `large-v3` | ~6GB | 2.88GB | ~300 fps | ~97% | Best accuracy |\r\n\r\n> **⚠️ Windows note**: `large-v3` needs >10GB RAM. If transcription fails, always check ffmpeg PATH first (see Troubleshooting).\r\n> **⚠️ faster-whisper**: Uses CTranslate2 native DLL. Blocked by WDAC/AppLocker on some managed Windows machines. Falls back to openai-whisper automatically.\r\n\r\nOutput formats: `txt`, `srt`, `vtt`, `json` (default: all).\r\n\r\nBackend: `--backend auto` (default, prefer faster-whisper) / `openai` / `faster`.\r\n\r\nSee `references/asr_models.md` for full model comparison.\r\n\r\n### Step 3: LLM Text Calibration\r\n\r\nAfter transcription, read the `.txt` output and apply corrections. Key calibration categories:\r\n\r\n1. **Homophone fixes** (同音字): 硬钢→硬扛, 模→磨, 骨→股\r\n2. **Company/product names**: Deepseat→DeepSeek, 中繼續創→中际旭创, HPM→HBM\r\n3. **Financial terms**: 抛押→抛压, 护盘 (not 互盘), 筹码, K线收十字星 (not 14星)\r\n4. **Common substitutions**: 跟锋→跟风, 微转→微赚, 落带为安→落袋为安\r\n5. **Traditional→Simplified**: If model outputs traditional Chinese, convert to simplified\r\n6. **Structural cleanup**: Add paragraph breaks at topic shifts, format as prose\r\n\r\nSee `references/calibration_guide.md` for the full 30+ pattern library.\r\n\r\n### Step 4: Deliver Results\r\n\r\nPresent the calibrated text. Always include:\r\n- Model used (small/medium/large) and quality notes\r\n- Any sections with low confidence or unclear audio\r\n- Summary of corrections applied (counts by category)\r\n\r\n## Platform Support\r\n\r\n| Platform | Status | Notes |\r\n|----------|--------|-------|\r\n| Bilibili | ✅ | Audio-only streams available without login. 720P+ video needs cookies. |\r\n| Xiaohongshu | ✅ | Full support via `XiaoHongShu` extractor. Short links (xhslink.com) auto-resolved. No cookies needed. |\r\n| YouTube | ✅ | Full support. Cookies may improve format selection. |\r\n| Douyin/TikTok | ✅ | Via yt-dlp |\r\n| All yt-dlp sites | ✅ | 1800+ supported platforms |\r\n\r\n## Extending with New ASR Models\r\n\r\n`scripts/transcribe.py` is designed for backend extensibility:\r\n\r\n1. Add model info to `MODEL_SIZES` dict\r\n2. Implement `transcribe_<backend>()` function\r\n3. Add CLI flag in argparse\r\n\r\n**Planned backends**: whisper.cpp (native C++), Cloud APIs (AssemblyAI, iFlytek).\r\n\r\n## Troubleshooting\r\n\r\n| Problem | Solution |\r\n|---------|----------|\r\n| SIGKILL / ffmpeg FileNotFoundError | ffmpeg not in PATH. Script auto-detects 7 common install locations (winget, scoop, chocolatey, manual). If ffmpeg is elsewhere, add its directory to system PATH. |\r\n| yt-dlp download fails | Update yt-dlp: `pip install -U yt-dlp`. Try with cookies. |\r\n| \"No subtitles found\" | Expected. This skill uses ASR, not built-in captions. |\r\n| ffmpeg not found | Run `install_deps.py` (handles Windows non-PATH detection). |\r\n| GPU not utilized | openai-whisper CPU-only by default. Install `faster-whisper` for GPU. |\r\n\r\n## Performance Benchmarks (Tested)\r\n\r\n| Video Duration | Backend | Model | Time | RAM Peak | Accuracy |\r\n|---------------|---------|-------|------|----------|----------|\r\n| 6 min (Bilibili) | openai | medium | ~4m 30s | ~6GB | ~95% |\r\n| 15 min (Bilibili) | openai | medium | ~10m | ~5GB | **~95%** ✅ |\r\n| 28 min (Xiaohongshu) | openai | small | ~4m 30s | ~2.5GB | ~90%* |\r\n| 15 min (Bilibili) | faster | medium | ~2m** | ~3GB | **~95%** ✅ |\r\n| 15 min (Bilibili) | faster | small | ~1m** | ~1.5GB | ~90% |\r\n\r\n*English content; Chinese accuracy ~88-90% for small model.\r\n**Estimated; faster-whisper benchmarks show 4x speedup vs openai-whisper (see faster-whisper PyPI).\r\nTested on Windows 11, Intel i7, 16GB RAM.\r\n\r\n## Changelog\r\n\r\n### v1.0.3\r\n- **New**: `faster-whisper` backend support via `--backend faster` (CTranslate2, 4x faster, 50% less RAM)\r\n- **New**: `--backend auto` (default) auto-detects faster-whisper, falls back to openai-whisper\r\n- **Removed**: `small` model from openai-whisper backend (88-90% accuracy, superseded by faster-whisper)\r\n- **Improve**: `small` model still available via faster-whisper backend (lightweight option)\r\n- **Improve**: Performance benchmarks now show both backends separately\r\n- **Improve**: `--backend` flag added to `run.py` and `transcribe.py`\r\n\r\n### v1.0.2\r\n- **New**: Xiaohongshu (小红书) platform support — yt-dlp `XiaoHongShu` extractor\r\n- **New**: Short link auto-resolution (xhslink.com → full URL via redirect)\r\n- **Improve**: Platform support table now lists 小红书 explicitly\r\n- **Improve**: Added 28-minute Xiaohongshu benchmark (small model, English)\r\n- **Improve**: Quick Start examples include xhslink.com usage\r\n\r\n### v1.0.1\r\n- **Fix**: Expanded ffmpeg search paths from 3→7 (winget/scoop/chocolatey/ProgramFiles(x86))\r\n- **Fix**: `ensure_deps()` now injects ffmpeg into `os.environ['PATH']` on success\r\n- **Fix**: SIGKILL troubleshooting updated — root cause is ffmpeg PATH, not OOM\r\n- **Improve**: Auto-detect GPU (`torch.cuda.is_available()`) for fp16 support\r\n- **Improve**: `verbose=True` for real-time transcription progress visibility\r\n- **Improve**: More accurate error messages in dependency checks\r\n\r\n### v1.0.0\r\n- Initial release: download (yt-dlp) + transcribe (whisper) + calibrate (LLM) pipeline\r\n- 7 ffmpeg install path auto-detection\r\n- Multi-format output (TXT, SRT, VTT, JSON)\r\n- Platform support: Bilibili, YouTube, all yt-dlp sites\n\nFile v1.0.3:_meta.json\n\n{\n  \"ownerId\": \"kn71y79mxxwmzynnpj6e284kch82gwd2\",\n  \"slug\": \"video-subtitle-extractor\",\n  \"version\": \"1.0.3\",\n  \"publishedAt\": 1779431001421\n}\n\nFile v1.0.3:references/asr_models.md\n\n# ASR Model Selection Guide\r\n\r\n## Available Backends\r\n\r\n### openai-whisper (default fallback)\r\nOriginal OpenAI Whisper via PyTorch. Stable, widely compatible.\r\n\r\n| Model | RAM | Disk | Speed | Quality | Best For |\r\n|-------|-----|------|-------|---------|----------|\r\n| `tiny` | ~1GB | 75MB | ~500 fps | low | Testing only |\r\n| `medium` | ~5GB | 1.42GB | ~165 fps | **~95%** ✅ | **Recommended default** |\r\n| `large-v3` | ~10GB | 2.88GB | ~80 fps | ~97% | Production quality |\r\n| `large-v3-turbo` | ~6GB | 1.6GB | ~120 fps | ~96% | Good compromise |\r\n\r\n> **Note**: `small` model removed from openai-whisper backend (88-90% accuracy, not worth the disk space). Use faster-whisper backend for lightweight option.\r\n\r\n### faster-whisper (recommended)\r\nCTranslate2-optimized Whisper. **4x faster, 50% less RAM** — same model weights, different inference engine. Install: `pip install faster-whisper`.\r\n\r\n| Model | RAM | Disk | Speed | Quality | Best For |\r\n|-------|-----|------|-------|---------|----------|\r\n| `small` | ~1.5GB | 461MB | ~1800 fps | ~90% | Low-resource systems |\r\n| `medium` | ~3GB | 1.42GB | ~600 fps | **~95%** ✅ | **Recommended** |\r\n| `large-v3` | ~6GB | 2.88GB | ~300 fps | ~97% | Best accuracy |\r\n\r\n**Key advantages over openai-whisper:**\r\n- int8 quantization (CPU) or fp16 (GPU) — ~50% less RAM\r\n- CTranslate2 inference engine — 4x faster on CPU\r\n- Batched inference for even more speed\r\n- Same model weights as openai-whisper (no quality tradeoff)\r\n\r\n**Known limitations:**\r\n- CTranslate2 native DLL blocked by WDAC/AppLocker on some Windows machines → auto-falls back to openai-whisper\r\n- Not compatible with whisper.cpp model format (GGML/GGUF)\r\n\r\n### Other Backends (Future)\r\n\r\n- **whisper.cpp**: Native C++ with GGML/GGUF format, smallest footprint, good for edge devices\r\n- **SenseVoice** (FunAudioLLM): Non-autoregressive, excellent Chinese accuracy (~95% small model), ~200MB |\r\n\r\n## Language-Specific Notes\r\n\r\n### Chinese (zh)\r\n- `medium`: Good for general Chinese content. Some errors on homophones and financial terms.\r\n- `large-v3`: Best Chinese accuracy, handles accents and domain terminology better.\r\n- Common errors: 同音字混淆 (硬扛→硬钢), 金融术语 (抛压→抛押, 交筹→焦愁), K线术语 (十字星→14星)\r\n\r\n### English (en)\r\n- `small`: Sufficient for clear English speech.\r\n- `medium`: Excellent accuracy for most content.\r\n\r\n## Memory Constraints\r\n\r\nOn Windows, `large-v3` may be killed (SIGKILL) on systems with <16GB RAM due to FP32 fallback.\r\nIf killed, fall back to `medium` or use `larger-v3-turbo`.\r\n\r\n## Adding New Backends\r\n\r\nTo add a new backend, implement a `transcribe_<backend>()` function in `transcribe.py`\r\nfollowing the same interface: `(audio_path, model_name, language, output_dir)`.\r\n\r\nThen register in:\r\n1. `check_whisper()` — add import detection\r\n2. CLI argparse — add `--backend` choice\r\n3. `transcribe()` — add routing logic\r\n\r\n## Model Auto-Download\r\n\r\nModels are downloaded automatically by openai-whisper on first use.\r\nCache location:\r\n- Windows: `C:\\Users\\<user>\\.cache\\whisper\\`\r\n- macOS: `~/Library/Caches/whisper/`\r\n- Linux: `~/.cache/whisper/`\n\nFile v1.0.3:references/calibration_guide.md\n\n# Text Calibration Guide for Chinese ASR Output\r\n\r\nCommon transcription errors and their corrections. Apply these patterns when calibrating whisper output for Chinese financial/technical content.\r\n\r\n## 1. Homophone Replacements (同音字混淆)\r\n\r\n| Raw (Wrong) | Correct | Example |\r\n|-------------|---------|---------|\r\n| 硬钢 | 硬扛 | 硬钢→硬扛 |\r\n| 抛押 | 抛压 | 消化抛押→消化抛压 |\r\n| 模两个月 | 磨两个月 | 横盘调整模两个月→磨两个月 |\r\n| 膜光短线 | 磨光短线 | 膜光短线→磨光短线 |\r\n| 流通骨 | 流通股 | 流通骨的换手→流通股的换手 |\r\n| 金接盘 | 新接盘 | 金接盘的成本→新接盘的成本 |\r\n| 拉伸 | 拉升 | 拉伸成本→拉升成本 |\r\n| 跟锋 | 跟风 | 跟锋买入→跟风买入 |\r\n| 微转 | 微赚 | 微转就抛售→微赚就抛售 |\r\n| 落带为安 | 落袋为安 | 落带为安→落袋为安 |\r\n| 互盘 | 护盘 | 主力互盘明显→主力护盘明显 |\r\n| 逼散互买 | 逼散户卖 | 洗盘本质是逼散互买→逼散户卖 |\r\n| 仅 | 有 | 仅有资金拖住→有资金托住 |\r\n| 军线 | 均线 | 关键军线→关键均线 |\r\n| 快有动作 | 快有动作 | Already correct, but watch for 快会→快会 |\r\n\r\n## 2. Financial Term Corrections\r\n\r\n| Raw | Correct | Context |\r\n|-----|---------|---------|\r\n| 交筹 | 交筹 | 慢慢就焦愁→慢慢就交筹 |\r\n| 再计 | 在即 | 拉升再计→拉升在即 |\r\n| 没装 | 没仓 | 根本没装→根本没仓 |\r\n| 利空 | 利空 | Already correct, verify |\r\n| K线收14星 | K线收十字星 | 14→十 |\r\n| 14星 | 十字星 | K线收十字星 |\r\n| 洗崩 | 洗崩 | Already correct (跌太多) |\r\n| 割肉 | 割肉 | Already correct |\r\n\r\n## 3. Domain Term Patterns\r\n\r\nWhisper often confuses financial jargon:\r\n- 洗盘 (xǐ pán) vs 洗盘 (same pronunciation but context-dependent)\r\n- 筹码 (chóu mǎ) - usually correct\r\n- 建仓 (jiàn cāng) - usually correct\r\n- 杠杆 (gàng gǎn) - usually correct\r\n- 信托 (xìn tuō) - usually correct\r\n\r\n## 4. Structural Cleanup\r\n\r\n- Add proper punctuation (periods, commas) where ASR output lacks them\r\n- Split long run-on sentences at natural topic breaks\r\n- Format as flowing paragraphs, not timestamp-ordered fragments\r\n- Add section headings for topic shifts: \"洗盘核心目的\", \"三个核心指标\", \"三个信号\", etc.\r\n- Keep timestamps if user wants time-coded output (from .srt/.vtt)\r\n\r\n## 5. Quality Indicators\r\n\r\nAfter calibration, flag low-confidence sections:\r\n- Unclear audio sections (background noise, overlapping speech)\r\n- Rapid technical jargon sequences\r\n- Sections where multiple interpretations are plausible\r\n\r\n## 6. Multi-language Content\r\n\r\nFor bilingual content (Chinese + English):\r\n- Preserve English terms: PE ratio, MA, MACD, KDJ, Bollinger Bands\r\n- Mixed language phrases: \"比如 PE 20倍\", \"MACD 金叉\" are correct\r\n- Don't translate technical terms to Chinese\r\n\r\n## 7. Calibration Output Format\r\n\r\nAfter applying corrections, present as:\r\n- Clean prose with proper Chinese punctuation\r\n- Optional: show what was changed vs raw output\r\n- Optional: timestamp references from original SRT/VTT\r\n\r\n## 8. AI / Tech Domain Corrections (AI 及科技领域)\r\n\r\nFor videos about AI, semiconductors, and tech investment topics, watch for:\r\n\r\n### Chinese Company Names (whisper frequent errors)\r\n| Raw (Wrong) | Correct | Context |\r\n|-------------|---------|---------|\r\n| 中繼續創 | 中际旭创 | A股光模块龙头 |\r\n| 新益勝 | 新易盛 | A股光模块 |\r\n| 天賦通信 | 天孚通信 | A股通信 |\r\n| 元傑科技 | 源杰科技 | A股芯片 |\r\n| 阿力 | 阿里 | 阿里巴巴 |\r\n| Alley | 阿里 | Context: 腾讯、阿里、字节 |\r\n\r\n### AI Product & Term Names\r\n| Raw (Wrong) | Correct |\r\n|-------------|---------|\r\n| Deepseat | DeepSeek |\r\n| ChadGPT | ChatGPT |\r\n| Cloud | Claude (if Anthropic context) |\r\n| Moe價構 / 莫架构 | MoE架构 (Mixture of Experts) |\r\n| HPM | HBM (High Bandwidth Memory) |\r\n| 光膜块 | 光模块 (Optical Module) |\r\n| 夜冷 | 液冷 (Liquid Cooling) |\r\n| 巨深智能 | 具身智能 (Embodied AI) |\r\n| 端側推理 | 端侧推理 (On-device Inference) |\r\n| 推測解碼 | 推测解码 (Speculative Decoding) |\r\n| 主能板块 | 储能板块 (Energy Storage) |\r\n\r\n### Traditional → Simplified Chinese\r\nWhisper medium/large models commonly output traditional Chinese (繁體) for simplified content. Always convert:\r\n- 發→发, 來→来, 時→时, 會→会, 體→体, 報→报, 機→机\r\n- 線→线, 構→构, 購→购, 業→业, 電→电, 纜→缆, 銅→铜\r\n- 壹→一, 貳→二, 等等\n\nFile v1.0.3:LICENSE\n\nMIT No Attribution License\n\nCopyright (c) 2025 forhonourlx\n\nPermission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the \"Software\"), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.\n\nArchive v1.0.2: 9 files, 17590 bytes\n\nFiles: LICENSE (917b), references/asr_models.md (1918b), references/calibration_guide.md (4619b), scripts/download_audio.py (5515b), scripts/install_deps.py (4890b), scripts/run.py (6346b), scripts/transcribe.py (9666b), SKILL.md (7897b), _meta.json (143b)\n\nFile v1.0.2:SKILL.md\n\n---\r\nname: video-subtitle-extractor\r\ndescription: |\r\n  Cross-platform video subtitle extraction using ASR (speech-to-text). Downloads audio from video URLs via yt-dlp, transcribes with openai-whisper (small/medium/large-v3), and applies LLM-based text calibration for Chinese financial/technical content. Use when: (1) extracting subtitles from Bilibili, Xiaohongshu, YouTube, or any yt-dlp-supported platform, (2) the video has no built-in subtitles, (3) users say \"下载字幕\", \"提取字幕\", \"语音转文字\", \"视频转文字\", \"字幕提取\", \"ASR转写\", (4) needing to transcribe audio files to text, (5) working with Chinese-language video content requiring high-accuracy transcription. Automatically handles dependency installation (ffmpeg, yt-dlp, openai-whisper) and model downloads.\r\n---\r\n\r\n# Video Subtitle Extractor 🎬→📝\r\n\r\nCross-platform ASR subtitle extraction pipeline. Downloads audio from any yt-dlp-compatible video platform, transcribes with openai-whisper, and applies LLM-based text calibration for Chinese content.\r\n\r\n**Tested & verified** on Windows 11 with real Bilibili & Xiaohongshu videos (medium model, ~95% accuracy for Chinese).\r\n\r\n## Quick Start\r\n\r\n```bash\r\n# One-command full pipeline (Bilibili, Xiaohongshu, YouTube, etc.)\r\npython scripts/run.py <video_url> --model medium --language zh --output-dir ./output\r\n\r\n# Example: Xiaohongshu short link (auto-resolves redirect)\r\npython scripts/run.py https://xhslink.com/xxxxx --model medium --language zh\r\n\r\n# Download audio only\r\npython scripts/download_audio.py <video_url> <output_dir>\r\n\r\n# Transcribe existing audio\r\npython scripts/transcribe.py <audio_file> --model medium --language zh\r\n```\r\n\r\n## When to Use This Skill\r\n\r\nUse this skill when:\r\n1. The video has **no built-in subtitles** (Bilibili, Xiaohongshu, YouTube, etc.)\r\n2. You need **high-accuracy Chinese transcription** (~95% with medium model)\r\n3. You want **multiple output formats** (TXT, SRT, VTT, JSON)\r\n4. You need **LLM-assisted text calibration** for financial/technical terms\r\n5. The user says: \"下载字幕\", \"提取字幕\", \"语音转文字\", \"视频转文字\", \"字幕提取\", \"ASR转写\"\r\n\r\n## Workflow\r\n\r\n### Step 0: Install Dependencies (once)\r\n\r\n```bash\r\npython scripts/install_deps.py\r\n```\r\n\r\nAuto-detects OS and installs: ffmpeg (winget/brew/apt), yt-dlp (pip), openai-whisper (pip). Handles Windows ffmpeg path detection even when not in PATH.\r\n\r\n### Step 1: Download Audio\r\n\r\nRun `scripts/download_audio.py <url> [output_dir]`.\r\n\r\nUses yt-dlp to extract the best available audio format (m4a preferred). Supports Bilibili, YouTube, and 1800+ yt-dlp-compatible platforms. The script automatically detects ffmpeg even when not in system PATH.\r\n\r\n**If download fails**: the video may require cookies. Try:\r\n```bash\r\nyt-dlp --cookies-from-browser chrome <url>\r\n```\r\n\r\n### Step 2: ASR Transcription\r\n\r\nRun `scripts/transcribe.py <audio> --model <size> --language <lang>`.\r\n\r\nModels are auto-downloaded on first use (disk space required):\r\n\r\n| Model | RAM | Disk | Speed | Quality | Best For |\r\n|-------|-----|------|-------|---------|----------|\r\n| `small` | ~2GB | 461MB | ~475 fps | ~90% | Quick tests |\r\n| `medium` | ~5GB | 1.42GB | ~165 fps | **~95%** ✅ | **Recommended** |\r\n| `large-v3` | ~10GB | 2.88GB | ~80 fps | ~97% | Best accuracy |\r\n| `large-v3-turbo` | ~6GB | 1.6GB | ~120 fps | ~96% | Good balance |\r\n\r\n> **⚠️ Windows note**: `large-v3` needs >10GB RAM. If transcription fails, always check ffmpeg PATH first (see Troubleshooting).\r\n\r\nOutput formats: `txt`, `srt`, `vtt`, `json` (default: all).\r\n\r\nSee `references/asr_models.md` for full model comparison.\r\n\r\n### Step 3: LLM Text Calibration\r\n\r\nAfter transcription, read the `.txt` output and apply corrections. Key calibration categories:\r\n\r\n1. **Homophone fixes** (同音字): 硬钢→硬扛, 模→磨, 骨→股\r\n2. **Company/product names**: Deepseat→DeepSeek, 中繼續創→中际旭创, HPM→HBM\r\n3. **Financial terms**: 抛押→抛压, 护盘 (not 互盘), 筹码, K线收十字星 (not 14星)\r\n4. **Common substitutions**: 跟锋→跟风, 微转→微赚, 落带为安→落袋为安\r\n5. **Traditional→Simplified**: If model outputs traditional Chinese, convert to simplified\r\n6. **Structural cleanup**: Add paragraph breaks at topic shifts, format as prose\r\n\r\nSee `references/calibration_guide.md` for the full 30+ pattern library.\r\n\r\n### Step 4: Deliver Results\r\n\r\nPresent the calibrated text. Always include:\r\n- Model used (small/medium/large) and quality notes\r\n- Any sections with low confidence or unclear audio\r\n- Summary of corrections applied (counts by category)\r\n\r\n## Platform Support\r\n\r\n| Platform | Status | Notes |\r\n|----------|--------|-------|\r\n| Bilibili | ✅ | Audio-only streams available without login. 720P+ video needs cookies. |\r\n| Xiaohongshu | ✅ | Full support via `XiaoHongShu` extractor. Short links (xhslink.com) auto-resolved. No cookies needed. |\r\n| YouTube | ✅ | Full support. Cookies may improve format selection. |\r\n| Douyin/TikTok | ✅ | Via yt-dlp |\r\n| All yt-dlp sites | ✅ | 1800+ supported platforms |\r\n\r\n## Extending with New ASR Models\r\n\r\n`scripts/transcribe.py` is designed for backend extensibility:\r\n\r\n1. Add model info to `MODEL_SIZES` dict\r\n2. Implement `transcribe_<backend>()` function\r\n3. Add CLI flag in argparse\r\n\r\n**Planned backends**: faster-whisper (CTranslate2), whisper.cpp (native C++), Cloud APIs (AssemblyAI, iFlytek).\r\n\r\n## Troubleshooting\r\n\r\n| Problem | Solution |\r\n|---------|----------|\r\n| SIGKILL / ffmpeg FileNotFoundError | ffmpeg not in PATH. Script auto-detects 7 common install locations (winget, scoop, chocolatey, manual). If ffmpeg is elsewhere, add its directory to system PATH. |\r\n| yt-dlp download fails | Update yt-dlp: `pip install -U yt-dlp`. Try with cookies. |\r\n| \"No subtitles found\" | Expected. This skill uses ASR, not built-in captions. |\r\n| ffmpeg not found | Run `install_deps.py` (handles Windows non-PATH detection). |\r\n| GPU not utilized | openai-whisper CPU-only by default. Install `faster-whisper` for GPU. |\r\n\r\n## Performance Benchmarks (Tested)\r\n\r\n| Video Duration | Model | Time | RAM Peak | Accuracy |\r\n|---------------|-------|------|----------|----------|\r\n| 6 min (Bilibili) | small | ~1m 17s | ~2.5GB | ~90% |\r\n| 6 min (Bilibili) | medium | ~4m 30s | ~6GB | ~95% |\r\n| 13 min (Bilibili) | medium | ~8m | ~6.5GB | ~95% |\r\n| 15 min (Bilibili) | small | ~3m | ~2.5GB | ~88% |\r\n| 15 min (Bilibili) | medium | ~10m | ~5GB | **~95%** ✅ |\r\n| 28 min (Xiaohongshu) | small | ~4m 30s | ~2.5GB | ~90%* |\r\n\r\n*English content; Chinese accuracy would be ~88-90% for small model.\r\nTested on Windows 11, Intel i7, 16GB RAM. Medium model recommended as default. Performance may vary by CPU speed.\r\n\r\n## Changelog\r\n\r\n### v1.0.2\r\n- **New**: Xiaohongshu (小红书) platform support — yt-dlp `XiaoHongShu` extractor\r\n- **New**: Short link auto-resolution (xhslink.com → full URL via redirect)\r\n- **Improve**: Platform support table now lists 小红书 explicitly\r\n- **Improve**: Added 28-minute Xiaohongshu benchmark (small model, English)\r\n- **Improve**: Quick Start examples include xhslink.com usage\r\n\r\n### v1.0.1\r\n- **Fix**: Expanded ffmpeg search paths from 3→7 (winget/scoop/chocolatey/ProgramFiles(x86))\r\n- **Fix**: `ensure_deps()` now injects ffmpeg into `os.environ['PATH']` on success\r\n- **Fix**: SIGKILL troubleshooting updated — root cause is ffmpeg PATH, not OOM\r\n- **Improve**: Auto-detect GPU (`torch.cuda.is_available()`) for fp16 support\r\n- **Improve**: `verbose=True` for real-time transcription progress visibility\r\n- **Improve**: More accurate error messages in dependency checks\r\n\r\n### v1.0.0\r\n- Initial release: download (yt-dlp) + transcribe (whisper) + calibrate (LLM) pipeline\r\n- 7 ffmpeg install path auto-detection\r\n- Multi-format output (TXT, SRT, VTT, JSON)\r\n- Platform support: Bilibili, YouTube, all yt-dlp sites\n\nFile v1.0.2:_meta.json\n\n{\n  \"ownerId\": \"kn71y79mxxwmzynnpj6e284kch82gwd2\",\n  \"slug\": \"video-subtitle-extractor\",\n  \"version\": \"1.0.2\",\n  \"publishedAt\": 1779426040647\n}\n\nFile v1.0.2:references/asr_models.md\n\n# ASR Model Selection Guide\r\n\r\n## Available Models\r\n\r\n| Model | RAM | Disk | Speed | Quality | Best For |\r\n|-------|-----|------|-------|---------|----------|\r\n| `small` | ~2GB | 461MB | Fast (~1min/6min audio) | Good | Quick tests, low-resource systems |\r\n| `medium` | ~5GB | 1.42GB | Medium (~3-5min) | High | **Recommended default** - best quality/speed ratio |\r\n| `large-v3` | ~10GB | 2.88GB | Slow (~10-20min) | Best | Production quality, needs high RAM |\r\n| `large-v3-turbo` | ~6GB | 1.6GB | Medium-fast | High | Good compromise, smaller than large-v3 |\r\n\r\n## Language-Specific Notes\r\n\r\n### Chinese (zh)\r\n- `medium`: Good for general Chinese content. Some errors on homophones and financial terms.\r\n- `large-v3`: Best Chinese accuracy, handles accents and domain terminology better.\r\n- Common errors: 同音字混淆 (硬扛→硬钢), 金融术语 (抛压→抛押, 交筹→焦愁), K线术语 (十字星→14星)\r\n\r\n### English (en)\r\n- `small`: Sufficient for clear English speech.\r\n- `medium`: Excellent accuracy for most content.\r\n\r\n## Memory Constraints\r\n\r\nOn Windows, `large-v3` may be killed (SIGKILL) on systems with <16GB RAM due to FP32 fallback.\r\nIf killed, fall back to `medium` or use `larger-v3-turbo`.\r\n\r\n## Future Model Compatibility\r\n\r\nThe `transcribe.py` script is designed for easy \n\nArchive v1.0.1: 9 files, 17270 bytes\n\nFiles: LICENSE (917b), references/asr_models.md (1918b), references/calibration_guide.md (4619b), scripts/download_audio.py (5515b), scripts/install_deps.py (4890b), scripts/run.py (6261b), scripts/transcribe.py (9666b), SKILL.md (7029b), _meta.json (143b)\n\nArchive v1.0.0: 9 files, 16490 bytes\n\nFiles: LICENSE (917b), references/asr_models.md (1918b), references/calibration_guide.md (4619b), scripts/download_audio.py (5102b), scripts/install_deps.py (4890b), scripts/run.py (5891b), scripts/transcribe.py (9029b), SKILL.md (5976b), _meta.json (143b)","readmeExcerpt":"Skill: Video Subtitle Extractor Owner: forhonourlx Summary: Cross-platform video subtitle extraction using multi-engine ASR (speech-to-text). Downloads audio from video URLs via yt-dlp, transcribes with SenseVoice / w... Tags: ai:1.0.0, asr:1.0.0, chinese:1.0.0, latest:2.0.0, major:2.0.0, subtitle:1.0.0, video:1.0.0, whisper:1.0.0 Version history: v2.0.0 | 2026-05-27T08:24:13.771Z | user Multi-engine ASR: SenseVoice ","codeSnippets":[],"executableExamples":[],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\r\nname: video-subtitle-extractor\r\ndescription: |\r\n  Cross-platform video subtitle extraction using multi-engine ASR (speech-to-text). Downloads audio from video URLs via yt-dlp, transcribes with SenseVoice / whisper.cpp / openai-whisper (default: SenseVoice Small for Chinese), and applies LLM-based text calibration for Chinese financial/technical content. Use when: (1) extracting subtitles from Bilibili, Xiaohongshu, YouTube, or any yt-dlp-supported platform, (2) the video has no built-in subtitles, (3) users say \"下载字幕\", \"提取字幕\", \"语音转文字\", \"视频转文字\", \"字幕提取\", \"ASR转写\", (4) needing to transcribe audio files to text, (5) working with Chinese-language video content requiring high-accuracy transcription. Automatically handles dependency installation (ffmpeg, yt-dlp, ASR backends) and model downloads.\r\n---\r\n\r\n# Video Subtitle Extractor 🎬→📝\r\n\r\nCross-platform multi-engine ASR subtitle extraction pipeline. Downloads audio from any yt-dlp-compatible video platform, transcribes with **SenseVoice / whisper.cpp / openai-whisper**, and applies LLM-based text calibration for Chinese content.\r\n\r\n**Default engine**: SenseVoice Small (Alibaba FunASR) — ~1.5GB RAM, 234MB disk, 20× realtime speed, ~96% Chinese accuracy.\r\n\r\n**Tested & verified** on Windows 11 with real Bilibili & Xiaohongshu videos.\r\n\r\n## Quick Start\r\n\r\n```bash\r\n# One-command full pipeline (SenseVoice Small — default, blazing fast for Chinese)\r\npython scripts/run.py <video_url> --output-dir ./output\r\n\r\n# Use whisper.cpp GGML (even lighter, 2GB RAM)\r\npython scripts/run.py <video_url> --backend whispercpp --model medium-q5_1\r\n\r\n# Use openai-whisper (standard, 5GB RAM)\r\npython scripts/run.py <video_url> --backend openai --model medium\r\n\r\n# Download audio only\r\npython scripts/download_audio.py <video_url> <output_dir>\r\n\r\n# Download audio + video (keep both as middleware)\r\npython scripts/download_audio.py <video_url> <output_dir> --save-video --video-quality 1080\r\n\r\n# Transcribe existing audio with auto backend selection\r\npython scripts/transcribe.py <audio_file> --backend auto --language zh\r\n\r\n# Transcribe with specific backend\r\npython scripts/transcribe.py <audio_file> --backend sensevoice --language zh\r\npython scripts/transcribe.py <audio_file> --backend whispercpp --model medium-q5_1\r\npython scripts/transcribe.py <audio_file> --backend openai --model medium\r\n```\r\n\r\n## When to Use This Skill\r\n\r\nUse this skill when:\r\n1. The video has **no built-in subtitles** (Bilibili, Xiaohongshu, YouTube, etc.)\r\n2. You need **high-accuracy Chinese transcription** (~95% with medium model)\r\n3. You want **multiple output formats** (TXT, SRT, VTT, JSON)\r\n4. You need **LLM-assisted text calibration** for financial/technical terms\r\n5. The user says: \"下载字幕\", \"提取字幕\", \"语音转文字\", \"视频转文字\", \"字幕提取\", \"ASR转写\"\r\n\r\n## Workflow\r\n\r\n### Step 0: Install Dependencies (once)\r\n\r\n```bash\r\npython scripts/install_deps.py\r\n```\r\n\r\nAuto-detects OS and installs: ffmpeg (winget/brew/apt), yt-dlp (pip), openai-whisper (pip). Handles Windows ffmpeg path"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn71y79mxxwmzynnpj6e284kch82gwd2\",\n  \"slug\": \"video-subtitle-extractor\",\n  \"version\": \"2.0.0\",\n  \"publishedAt\": 1779870253771\n}"},{"path":"references/asr_models.md","content":"# ASR Model Selection Guide\r\n\r\n## Available Models\r\n\r\n| Model | RAM | Disk | Speed | Quality | Best For |\r\n|-------|-----|------|-------|---------|----------|\r\n| `small` | ~2GB | 461MB | Fast (~1min/6min audio) | Good | Quick tests, low-resource systems |\r\n| `medium` | ~5GB | 1.42GB | Medium (~3-5min) | High | **Recommended default** - best quality/speed ratio |\r\n| `large-v3` | ~10GB | 2.88GB | Slow (~10-20min) | Best | Production quality, needs high RAM |\r\n| `large-v3-turbo` | ~6GB | 1.6GB | Medium-fast | High | Good compromise, smaller than large-v3 |\r\n\r\n## Language-Specific Notes\r\n\r\n### Chinese (zh)\r\n- `medium`: Good for general Chinese content. Some errors on homophones and financial terms.\r\n- `large-v3`: Best Chinese accuracy, handles accents and domain terminology better.\r\n- Common errors: 同音字混淆 (硬扛→硬钢), 金融术语 (抛压→抛押, 交筹→焦愁), K线术语 (十字星→14星)\r\n\r\n### English (en)\r\n- `small`: Sufficient for clear English speech.\r\n- `medium`: Excellent accuracy for most content.\r\n\r\n## Memory Constraints\r\n\r\nOn Windows, `large-v3` may be killed (SIGKILL) on systems with <16GB RAM due to FP32 fallback.\r\nIf killed, fall back to `medium` or use `larger-v3-turbo`.\r\n\r\n## Future Model Compatibility\r\n\r\nThe `transcribe.py` script is designed for easy backend extension:\r\n- `faster-whisper`: CTranslate2 backend, more memory efficient\r\n- `whisper.cpp`: Native C++ implementation\r\n- `mlx-whisper`: Apple Silicon optimized\r\n- Cloud APIs: AssemblyAI, iFlytek, Whisper API\r\n\r\nTo add a new backend, implement a `transcribe_<backend>()` function in transcribe.py\r\nfollowing the same interface (audio_path, model_name, language, output_dir).\r\n\r\n## Model Auto-Download\r\n\r\nModels are downloaded automatically by openai-whisper on first use.\r\nCache location:\r\n- Windows: `C:\\Users\\<user>\\.cache\\whisper\\`\r\n- macOS: `~/Library/Caches/whisper/`\r\n- Linux: `~/.cache/whisper/`"},{"path":"references/calibration_guide.md","content":"# Text Calibration Guide for Chinese ASR Output\r\n\r\nCommon transcription errors and their corrections. Apply these patterns when calibrating whisper output for Chinese financial/technical content.\r\n\r\n## 1. Homophone Replacements (同音字混淆)\r\n\r\n| Raw (Wrong) | Correct | Example |\r\n|-------------|---------|---------|\r\n| 硬钢 | 硬扛 | 硬钢→硬扛 |\r\n| 抛押 | 抛压 | 消化抛押→消化抛压 |\r\n| 模两个月 | 磨两个月 | 横盘调整模两个月→磨两个月 |\r\n| 膜光短线 | 磨光短线 | 膜光短线→磨光短线 |\r\n| 流通骨 | 流通股 | 流通骨的换手→流通股的换手 |\r\n| 金接盘 | 新接盘 | 金接盘的成本→新接盘的成本 |\r\n| 拉伸 | 拉升 | 拉伸成本→拉升成本 |\r\n| 跟锋 | 跟风 | 跟锋买入→跟风买入 |\r\n| 微转 | 微赚 | 微转就抛售→微赚就抛售 |\r\n| 落带为安 | 落袋为安 | 落带为安→落袋为安 |\r\n| 互盘 | 护盘 | 主力互盘明显→主力护盘明显 |\r\n| 逼散互买 | 逼散户卖 | 洗盘本质是逼散互买→逼散户卖 |\r\n| 仅 | 有 | 仅有资金拖住→有资金托住 |\r\n| 军线 | 均线 | 关键军线→关键均线 |\r\n| 快有动作 | 快有动作 | Already correct, but watch for 快会→快会 |\r\n\r\n## 2. Financial Term Corrections\r\n\r\n| Raw | Correct | Context |\r\n|-----|---------|---------|\r\n| 交筹 | 交筹 | 慢慢就焦愁→慢慢就交筹 |\r\n| 再计 | 在即 | 拉升再计→拉升在即 |\r\n| 没装 | 没仓 | 根本没装→根本没仓 |\r\n| 利空 | 利空 | Already correct, verify |\r\n| K线收14星 | K线收十字星 | 14→十 |\r\n| 14星 | 十字星 | K线收十字星 |\r\n| 洗崩 | 洗崩 | Already correct (跌太多) |\r\n| 割肉 | 割肉 | Already correct |\r\n\r\n## 3. Domain Term Patterns\r\n\r\nWhisper often confuses financial jargon:\r\n- 洗盘 (xǐ pán) vs 洗盘 (same pronunciation but context-dependent)\r\n- 筹码 (chóu mǎ) - usually correct\r\n- 建仓 (jiàn cāng) - usually correct\r\n- 杠杆 (gàng gǎn) - usually correct\r\n- 信托 (xìn tuō) - usually correct\r\n\r\n## 4. Structural Cleanup\r\n\r\n- Add proper punctuation (periods, commas) where ASR output lacks them\r\n- Split long run-on sentences at natural topic breaks\r\n- Format as flowing paragraphs, not timestamp-ordered fragments\r\n- Add section headings for topic shifts: \"洗盘核心目的\", \"三个核心指标\", \"三个信号\", etc.\r\n- Keep timestamps if user wants time-coded output (from .srt/.vtt)\r\n\r\n## 5. Quality Indicators\r\n\r\nAfter calibration, flag low-confidence sections:\r\n- Unclear audio sections (background noise, overlapping speech)\r\n- Rapid technical jargon sequences\r\n- Sections where multiple interpretations are plausible\r\n\r\n## 6. Multi-language Content\r\n\r\nFor bilingual content (Chinese + English):\r\n- Preserve English terms: PE ratio, MA, MACD, KDJ, Bollinger Bands\r\n- Mixed language phrases: \"比如 PE 20倍\", \"MACD 金叉\" are correct\r\n- Don't translate technical terms to Chinese\r\n\r\n## 7. Calibration Output Format\r\n\r\nAfter applying corrections, present as:\r\n- Clean prose with proper Chinese punctuation\r\n- Optional: show what was changed vs raw output\r\n- Optional: timestamp references from original SRT/VTT\r\n\r\n## 8. AI / Tech Domain Corrections (AI 及科技领域)\r\n\r\nFor videos about AI, semiconductors, and tech investment topics, watch for:\r\n\r\n### Chinese Company Names (whisper frequent errors)\r\n| Raw (Wrong) | Correct | Context |\r\n|-------------|---------|---------|\r\n| 中繼續創 | 中际旭创 | A股光模块龙头 |\r\n| 新益勝 | 新易盛 | A股光模块 |\r\n| 天賦通信 | 天孚通信 | A股通信 |\r\n| 元傑科技 | 源杰科技 | A股芯片 |\r\n| 阿力 | 阿里 | 阿里巴巴 |\r\n| Alley | 阿里 | Context: 腾讯、阿里、字节 |\r\n\r\n### AI Product & Term Names\r\n| Raw (Wrong) | Correct |\r\n|-------------|---------|\r\n| Deepseat | DeepSeek |\r\n| ChadGPT | ChatGPT |\r\n|"},{"path":"skill-card.md","content":"## Description:\n\nCross-platform video subtitle extraction using multi-engine ASR: it downloads audio from video URLs with yt-dlp, transcribes with SenseVoice, whisper.cpp, or openai-whisper, and can calibrate Chinese financial and technical transcripts.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[forhonourlx](https://clawhub.ai/user/forhonourlx)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers, analysts, and content teams use this skill to extract subtitles or transcript files from videos that lack built-in captions, especially Chinese-language financial or technical videos. It guides agents through dependency setup, media download, ASR transcription, optional video retention, and transcript calibration.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The security review flags broad install, download, browser-cookie, and persistent storage behaviors.\n\nMitigation: Review the skill before use, run it only in a trusted environment, and approve dependency installation and media downloads deliberately.\n\nRisk: Browser-cookie workflows can expose account session material to the local toolchain.\n\nMitigation: Avoid browser-cookie options unless the machine and toolchain are trusted; prefer videos that do not require cookies.\n\nRisk: Downloaded media and generated transcripts can remain on disk after execution.\n\nMitigation: Use a dedicated output directory and delete media, transcripts, and metadata after they are no longer needed.\n\n## Reference(s):\n\n- [ASR Model Selection Guide](artifact/references/asr_models.md)\n- [Text Calibration Guide for Chinese ASR Output](artifact/references/calibration_guide.md)\n- [ClawHub Skill Page](https://clawhub.ai/forhonourlx/skills/video-subtitle-extractor)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, code, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown guidance with shell commands and generated transcript files such as TXT, SRT, VTT, and JSON]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May create persistent media, transcript, subtitle, metadata, and calibrated transcript files in the selected output directory.]\n\n## Skill Version(s):\n\n2.0.0 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Cross-platform video subtitle extraction using multi-engine ASR (speech-to-text). Downloads audio from video URLs via yt-dlp, transcribes with SenseVoice / w... Skill: Video Subtitle Extractor Owner: forhonourlx Summary: Cross-platform video subtitle extraction using multi-engine ASR (speech-to-text). Downloads audio from video URLs via yt-dlp, transcribes with SenseVoice / w... Tags: ai:1.0.0, asr:1.0.0, chinese:1.0.0, latest:2.0.0, major:2.0.0, subtitle:1.0.0, video:1.0.0, whisper:1.0.0 Version history: v2.0.0 | 2026-05-27T08:24:13.771Z | user Multi-engine ASR: SenseVoice","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1561,"uniquenessScore":49,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-10T22:42:44.197Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-10T22:42:44.197Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T01:47:10.712Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}