{"id":"a7612ce9-e936-4e84-abb4-06c15abdfc81","entityType":"agent","slug":"clawhub-drakulavich-kesha-voice-kit","name":"kesha-voice-kit","canonicalUrl":"https://www.xpersona.co/agent/clawhub-drakulavich-kesha-voice-kit","canonicalPath":"/agent/clawhub-drakulavich-kesha-voice-kit","generatedAt":"2026-10-10T10:47:54.413Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-10T06:05:55.014Z","emptyReason":null},"description":"Local multilingual voice toolkit — speech-to-text (STT), text-to-speech (TTS), speaker diarization, and language detection, over a CLI or an MCP server. Runs entirely offline on Apple Silicon, Linux, and Windows. No API keys, no cloud. NVIDIA Parakeet TDT for STT across 25 European languages, Kokoro-82M + Vosk-TTS for TTS in 9 languages, plus macOS AVSpeechSynthesizer for ~180 system voices with zero install. Skill: kesha-voice-kit Owner: drakulavich Summary: Local multilingual voice toolkit — speech-to-text (STT), text-to-speech (TTS), speaker diarization, and language detection, over a CLI or an MCP server. Runs entirely offline on Apple Silicon, Linux, and Windows. No API keys, no cloud. NVIDIA Parakeet TDT for STT across 25 European languages, Kokoro-82M + Vosk-TTS for TTS in 9 languages, plus macOS AVSpeechSynthesize","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.6K downloads reported by the source. Last updated 10/10/2026.","installCommand":"clawhub skill install s17b3c9zks1vdxe5e7vt4f316h8571qe:kesha-voice-kit","sourceUrl":"https://clawhub.ai/drakulavich/kesha-voice-kit","homepage":"https://clawhub.ai/drakulavich/skills/kesha-voice-kit","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/drakulavich/kesha-voice-kit","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/drakulavich/skills/kesha-voice-kit","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":64,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Local multilingual voice toolkit — speech-to-text (STT), text-to-speech (TTS), speaker diarization, and language detection, over a CLI or an MCP server. Runs en"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-10T06:05:55.014Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T06:05:55.014Z","emptyReason":null},"stars":null,"forks":null,"downloads":1631,"packageName":null,"latestVersion":"1.6.1","tractionLabel":"1.6K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T06:05:54.988Z","emptyReason":null},"lastUpdatedAt":"2026-10-10T06:05:55.014Z","lastCrawledAt":"2026-10-10T06:05:54.988Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-11T06:05:54.988Z","lastVerifiedAt":null,"highlights":[{"version":"1.6.1","createdAt":"2026-08-05T16:49:21.169Z","changelog":"Retag 1.6.0 content onto every topic tag (they were still pinned to 1.5.0, so category browsing served the old description) and restore the Kesha Voice Kit display name. No content change from 1.6.0.","fileCount":4,"zipByteSize":12980},{"version":"1.6.0","createdAt":"2026-08-05T16:47:21.107Z","changelog":"Refresh from repo main (#745): TTS now covers 9 languages (was documented as English + Russian only), speaker diarization, and a new MCP server section (transcribe_audio, synthesize_speech, list_voices, list_languages). Replaces stale install size figures with per-platform numbers and kesha install --plan.","fileCount":4,"zipByteSize":13296},{"version":"1.5.0","createdAt":"2026-05-23T12:19:12.363Z","changelog":"Refresh skill docs from repo main: Bun-only install, Vosk-TTS Russian voice, OGG/Opus voice-note output, timestamp/diarization guidance, and OpenClaw audio/TTS routing.","fileCount":5,"zipByteSize":14270},{"version":"1.4.4","createdAt":"2026-04-26T08:34:27.688Z","changelog":"v1.4.4: include LICENSE in bundle (1.4.3 dropped it accidentally).","fileCount":3,"zipByteSize":5347},{"version":"1.4.3","createdAt":"2026-04-26T08:31:37.433Z","changelog":"v1.4.3: ONNX G2P replaces espeak-ng (no system deps); README trimmed; SKILL.md reframed as multilingual.","fileCount":3,"zipByteSize":5347},{"version":"1.4.1","createdAt":"2026-04-23T07:40:25.541Z","changelog":"v1.4.1: ONNX G2P replaces espeak-ng — no system deps on any platform. VAD auto-trigger for audio ≥120s. SSML <phoneme alphabet='ipa' ph='...'> override. AVSpeechSynthesizer for ~180 macOS system voices. Fixes displayName (was 'Parakeet Cli Ssml').","fileCount":3,"zipByteSize":8106},{"version":"1.3.2","createdAt":"2026-04-20T20:58:05.715Z","changelog":"- Updated description and documentation to clarify multilingual support for both STT and TTS features. - Improved wording and examples for text-to-speech, highlighting usage of additional macOS system voices in multiple languages. - Revised trigger keywords and usage sections to emphasize multilingual voice and language detection capabilities. - No code changes — documentation update only.","fileCount":3,"zipByteSize":7132},{"version":"1.3.1","createdAt":"2026-04-20T20:46:30.333Z","changelog":"- Removed 94 files, including documentation, plans, specs, benchmarks, and various assets. - No changes to functionality, install process, or user-facing features. - Skill remains focused on local speech-to-text, text-to-speech, and language detection.","fileCount":3,"zipByteSize":7140}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17b3c9zks1vdxe5e7vt4f316h8571qe:kesha-voice-kit","setupComplexity":"low","setupSteps":["Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-drakulavich-kesha-voice-kit/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-drakulavich-kesha-voice-kit/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-drakulavich-kesha-voice-kit/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-drakulavich-kesha-voice-kit/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-drakulavich-kesha-voice-kit/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-drakulavich-kesha-voice-kit/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T10:47:54.409Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-drakulavich-kesha-voice-kit/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-drakulavich-kesha-voice-kit/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-drakulavich-kesha-voice-kit/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-drakulavich-kesha-voice-kit/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-10T06:05:55.014Z","emptyReason":null},"readme":"Skill: kesha-voice-kit\n\nOwner: drakulavich\n\nSummary: Local multilingual voice toolkit — speech-to-text (STT), text-to-speech (TTS), speaker diarization, and language detection, over a CLI or an MCP server. Runs entirely offline on Apple Silicon, Linux, and Windows. No API keys, no cloud. NVIDIA Parakeet TDT for STT across 25 European languages, Kokoro-82M + Vosk-TTS for TTS in 9 languages, plus macOS AVSpeechSynthesizer for ~180 system voices with zero install.\n\nTags: apple-silicon:1.6.1, audio:1.6.1, latest:1.6.1, offline:1.6.1, russian:1.6.1, speech-and-transcription:1.6.1, stt:1.6.1, tts:1.6.1, voice:1.6.1\n\nVersion history:\n\nv1.6.1 | 2026-08-05T16:49:21.169Z | user\n\nRetag 1.6.0 content onto every topic tag (they were still pinned to 1.5.0, so category browsing served the old description) and restore the Kesha Voice Kit display name. No content change from 1.6.0.\n\nv1.6.0 | 2026-08-05T16:47:21.107Z | user\n\nRefresh from repo main (#745): TTS now covers 9 languages (was documented as English + Russian only), speaker diarization, and a new MCP server section (transcribe_audio, synthesize_speech, list_voices, list_languages). Replaces stale install size figures with per-platform numbers and kesha install --plan.\n\nv1.5.0 | 2026-05-23T12:19:12.363Z | user\n\nRefresh skill docs from repo main: Bun-only install, Vosk-TTS Russian voice, OGG/Opus voice-note output, timestamp/diarization guidance, and OpenClaw audio/TTS routing.\n\nv1.4.4 | 2026-04-26T08:34:27.688Z | user\n\nv1.4.4: include LICENSE in bundle (1.4.3 dropped it accidentally).\n\nv1.4.3 | 2026-04-26T08:31:37.433Z | user\n\nv1.4.3: ONNX G2P replaces espeak-ng (no system deps); README trimmed; SKILL.md reframed as multilingual.\n\nv1.4.1 | 2026-04-23T07:40:25.541Z | user\n\nv1.4.1: ONNX G2P replaces espeak-ng — no system deps on any platform. VAD auto-trigger for audio ≥120s. SSML <phoneme alphabet='ipa' ph='...'> override. AVSpeechSynthesizer for ~180 macOS system voices. Fixes displayName (was 'Parakeet Cli Ssml').\n\nv1.3.2 | 2026-04-20T20:58:05.715Z | auto\n\n- Updated description and documentation to clarify multilingual support for both STT and TTS features.\n- Improved wording and examples for text-to-speech, highlighting usage of additional macOS system voices in multiple languages.\n- Revised trigger keywords and usage sections to emphasize multilingual voice and language detection capabilities.\n- No code changes — documentation update only.\n\nv1.3.1 | 2026-04-20T20:46:30.333Z | auto\n\n- Removed 94 files, including documentation, plans, specs, benchmarks, and various assets.\n- No changes to functionality, install process, or user-facing features.\n- Skill remains focused on local speech-to-text, text-to-speech, and language detection.\n\nv1.3.0 | 2026-04-20T20:39:56.015Z | auto\n\n- Added detailed documentation clarifying usage, supported features, and installation steps.\n- Explained offline operation, wide audio format support, and privacy benefits.\n- Listed supported languages for both STT (speech-to-text) and TTS (text-to-speech).\n- Described performance improvements over OpenAI Whisper and introduced automatic language detection and voice routing.\n- Provided clear platform-specific installation and voice usage instructions.\n\nArchive index:\n\nArchive v1.6.1: 4 files, 12980 bytes\n\nFiles: README.md (10241b), skill-card.md (2173b), SKILL.md (15630b), _meta.json (134b)\n\nFile v1.6.1:SKILL.md\n\n---\nname: kesha-voice-kit\ndescription: Local multilingual voice toolkit — speech-to-text (STT), text-to-speech (TTS), speaker diarization, and language detection, over a CLI or an MCP server. Runs entirely offline on Apple Silicon, Linux, and Windows. No API keys, no cloud. NVIDIA Parakeet TDT for STT across 25 European languages, Kokoro-82M + Vosk-TTS for TTS in 9 languages, plus macOS AVSpeechSynthesizer for ~180 system voices with zero install.\nemoji: 🎙️\n\nrequires:\n  bins: [kesha]\n\ninstall:\n  - kind: bash\n    cmd: bun add -g \"@drakulavich/kesha-voice-kit\"\n  - kind: bash\n    cmd: kesha install\n---\n\n# kesha-voice-kit\n\nLocal voice toolkit: transcribe voice messages to text, synthesize speech, detect language of audio or text. Fully offline after `kesha install`. No API keys, no per-minute billing.\n\n**Trigger keywords for when to use this skill:** voice message, voice memo, voice note, .ogg, .opus, .wav, .mp3, audio file, transcribe, transcription, speech-to-text, STT, text-to-speech, TTS, synthesize speech, say, telegram voice note, whatsapp voice note, ogg-opus, opus, multilingual voice, multilingual ASR, language detection, speaker diarization, who said what, meeting transcript, MCP server, offline voice, privacy, Apple Silicon, CoreML.\n\n## When to use\n\n- **Voice memo arrived** (Telegram, WhatsApp, Slack, Signal .ogg/.opus/.m4a): transcribe with `kesha --json <path>` and branch on the detected language.\n- **Need to send a voice note (Telegram, WhatsApp, Signal, Discord)**: synthesize directly into messenger-native OGG/Opus with `kesha say --format ogg-opus --out reply.ogg \"<text>\"`. Default is mono 24 kHz @ 32 kbps - what Telegram `sendVoice` expects. No WAV redirect and no `ffmpeg` round-trip.\n- **Need local file playback/debug output**: WAV is still available with `kesha say --out reply.wav \"<text>\"`, but do not use WAV for Telegram voice replies. Auto-routes by detected language (Kokoro-82M for English, Vosk-TTS for Russian). On darwin-arm64, English Kokoro uses FluidAudio CoreML instead of ONNX. For other languages and ~180 more voices use `--voice macos-*` on macOS (zero model download).\n- **Need to detect what language a file is in** before choosing a pipeline: `kesha --json audio.ogg` returns both audio-based and text-based language detection with confidence scores.\n- **Need to capture your own voice** for transcription or as a voice-note source: `kesha record --out clip.wav` records up to 120s (override with `--max-seconds`) of mono 16 kHz WAV from the default microphone. Pipe straight into `kesha --json clip.wav` to close the loop.\n\n## OpenClaw plugin setup\n\nInstall the plugin, then explicitly route OpenClaw audio understanding through the CLI model entry. The plugin registration makes Kesha discoverable, but real voice-message transcription uses `tools.media.audio.models` with a `type: \"cli\"` entry.\n\n```bash\nbun add -g @drakulavich/kesha-voice-kit\nkesha install\nopenclaw plugins install @drakulavich/kesha-voice-kit\nopenclaw config patch --stdin <<'JSON5'\n{\n  tools: {\n    media: {\n      audio: {\n        enabled: true,\n        models: [\n          {\n            type: \"cli\",\n            command: \"kesha\",\n            args: [\"{{MediaPath}}\"],\n            timeoutSeconds: 15,\n          },\n        ],\n        echoTranscript: true,\n        echoFormat: '🦜 \"{transcript}\"',\n      },\n    },\n  },\n}\nJSON5\n```\n\nUse Kesha's default output for OpenClaw's normal voice-message path: stdout is the bare transcript text, while progress and errors stay off the transcript payload. The default setup echoes each transcript back to chat as `🦜 \"{transcript}\"` before the agent responds.\n\nFor agents that need timestamped segments, switch the model entry to JSON output and allow a longer timeout:\n\n```bash\nopenclaw config set tools.media.audio.models \\\n  '[{\"type\":\"cli\",\"command\":\"kesha\",\"args\":[\"--json\",\"--timestamps\",\"{{MediaPath}}\"],\"timeoutSeconds\":30}]'\n```\n\nVerification checklist:\n\n```bash\nwhich kesha\nkesha status\nopenclaw plugins list\nopenclaw config get tools.media.audio.models\nopenclaw config get tools.media.audio.echoTranscript\nopenclaw config get tools.media.audio.echoFormat\n```\n\nDo not rely on `openclaw.plugin.json` to patch `tools.media.audio.models`; OpenClaw ignores non-schema fields such as `configPatch`. Keep the CLI route in user config.\n\nFor OpenClaw TTS replies, route the local TTS provider to Kesha OGG/Opus output. This is the Telegram-safe path:\n\n```bash\nopenclaw config patch --stdin <<'JSON5'\n{\n  messages: {\n    tts: {\n      auto: \"always\",\n      provider: \"tts-local-cli\",\n      providers: {\n        \"tts-local-cli\": {\n          command: \"kesha\",\n          args: [\"say\", \"--format\", \"ogg-opus\", \"--out\", \"{{OutputPath}}\", \"{{Text}}\"],\n          outputFormat: \"opus\",\n          timeoutMs: 120000,\n        },\n      },\n    },\n  },\n}\nJSON5\n```\n\nWhen invoking Kesha manually from an OpenClaw flow, write OGG/Opus into an OpenClaw-owned temp path, for example `kesha say --format ogg-opus --out /tmp/openclaw/reply.ogg \"<text>\"`, after ensuring the directory exists. The configured `tts-local-cli` provider should use OpenClaw's `{{OutputPath}}` placeholder instead of a hardcoded path.\n\nDo not configure OpenClaw Telegram TTS as `kesha say \"<text>\" > reply.wav`; that creates a WAV file and will not render as a native Telegram voice note.\n\n## MCP server\n\n`kesha mcp` serves the same capabilities over Model Context Protocol (stdio), exposing `transcribe_audio`, `synthesize_speech`, `list_voices`, and `list_languages` to any MCP client:\n\n```json\n{ \"mcpServers\": { \"kesha\": { \"command\": \"kesha\", \"args\": [\"mcp\"] } } }\n```\n\nModels are never auto-downloaded — tools fail with a `kesha install` / `kesha install --tts` hint when missing. Full client setup (Claude Code, Claude Desktop, Codex, Cursor): [docs/mcp.md](docs/mcp.md).\n\n## STT: transcribe audio\n\n```bash\n# JSON output with language detection (recommended for automation)\nkesha --json voice.ogg\n```\n\n```json\n[{\n  \"file\": \"voice.ogg\",\n  \"text\": \"Привет, как дела?\",\n  \"lang\": \"ru\",\n  \"audioLanguage\": { \"code\": \"ru\", \"confidence\": 0.98 },\n  \"textLanguage\": { \"code\": \"ru\", \"confidence\": 0.99 }\n}]\n```\n\nUse `lang` (or the more detailed `audioLanguage`/`textLanguage`) to decide how to respond.\n\nNeed timestamped transcript segments for navigation, chapters, or downstream editing:\n\n```bash\nkesha --json --timestamps voice.ogg > voice.timestamps.json\njq '.[0].segments' voice.timestamps.json\n```\n\nEach segment has `start`, `end`, and `text` fields. `--timestamps` is available for machine-readable output (`--json`, `--toon`, or `--format json`).\n\n**Speaker diarization** (darwin-arm64, post-v1.12.0). Add `--speakers` to label each segment with a cluster ID — useful for transcribing multi-person calls / meetings:\n\n```bash\nkesha install --diarize                                  # one-time, ~245MB\nkesha --json --vad --speakers meeting.m4a > out.json\njq '.[0].segments[] | \"\\(.speaker)\\t\\(.text)\"' out.json\n```\n\nEach `segment.speaker` is a number (cluster id, stable within one file). On Linux / Windows the engine returns a clear \"currently darwin-arm64 only\" error — see [#199](https://github.com/drakulavich/kesha-voice-kit/issues/199).\n\n**Formats:** .ogg, .opus, .mp3, .m4a, .wav, .flac, .webm — decoded via symphonia, no ffmpeg required.\n\n**Other output modes:**\n- `kesha audio.ogg` — plain transcript on stdout\n- `kesha --format transcript audio.ogg` — transcript + `[lang: ru, confidence: 0.99]` footer\n- `kesha --json --timestamps audio.ogg` — JSON with timestamped `segments`\n- `kesha --toon audio.ogg` — TOON (compact, LLM-friendly JSON encoding); preferred when piping multi-file results to an LLM/agent\n- `kesha --verbose audio.ogg` — human-readable with language info\n- `kesha --lang en audio.ogg` — warn if detected language differs (useful sanity check)\n\n**Long audio:** files ≥ 120 s auto-engage Silero VAD chunking; force on with `--vad` or off with `--no-vad`. Short files use full-file ASR by default.\n\n## TTS: synthesize speech\n\n```bash\nkesha say \"Hello, world\" > hello.wav               # auto-routes en → Kokoro-82M\nkesha say \"Привет, мир\" > privet.wav              # auto-routes ru → Vosk-TTS\nkesha say --voice macos-de-DE \"Guten Tag\" > de.wav # any macOS system voice — German, French, Italian, ...\nkesha say --list-voices                            # Kokoro + Vosk-TTS + ~180 macos-* voices\n```\n\nOutput: WAV mono float32 by default. `--out <path>` writes to a file instead of stdout. For Telegram/OpenClaw replies, prefer `--format ogg-opus --out reply.ogg` or the OpenClaw-provided `{{OutputPath}}`.\n\n**Output formats** (`--format`, or inferred from the `--out` extension): `wav` (default, uncompressed), `ogg-opus` (messenger voice notes), `flac` (lossless, royalty-free, plays in every browser incl. Safari/iOS — the format for web-embeddable samples). FLAC keeps the engine's native rate; `--bitrate` / `--sample-rate` apply only to `ogg-opus`.\n\n```bash\nkesha say --format flac --out sample.flac \"Hello\"   # web-embeddable, Safari-safe\n```\n\n**Voice notes (Telegram / WhatsApp / Signal / Discord):** add `--format ogg-opus` to emit OGG/Opus directly — the format messenger APIs render as a native voice message:\n\n```bash\nkesha say --format ogg-opus --out reply.ogg \"Hello there\"                  # 24 kHz @ 32 kbps mono - Telegram-grade\nkesha say --voice ru-vosk-m02 --format ogg-opus --out reply.ogg \"Привет\"   # Russian voice note\nkesha say --format ogg-opus --bitrate 16000 --out tiny.ogg \"Hi\"            # tinier file, intelligible but lossy\n```\n\nFormat is also inferred from `--out` extension (`.ogg` / `.opus` / `.oga` → OGG/Opus). `--bitrate` (6 000–510 000 bps) and `--sample-rate` (8 000 / 12 000 / 16 000 / 24 000 / 48 000 Hz) tune the encoder.\n\n**Russian abbreviations** (`ru-vosk-*`): all-uppercase Cyrillic 2-5-char tokens auto-expand letter-by-letter when not pronounceable as a Russian syllable (ФСБ → \"эф-эс-бэ\", ВОЗ → \"воз\"). Disable with `--no-expand-abbrev`. See [docs/tts.md#russian-abbreviation-auto-expansion](docs/tts.md#russian-abbreviation-auto-expansion).\n\n**English acronyms** (`en-*`, Kokoro): three-table mechanism (letter-spell rule + STOP_LIST + IPA_LEXICON) auto-expands FBI → \"ef bee eye\" and gives EPAM/JSON/Anthropic the right IPA. Disable letter-spell with `--no-expand-abbrev`. See [docs/tts.md#english-acronym-auto-expansion](docs/tts.md#english-acronym-auto-expansion).\n\n**Russian word stress** (`ru-vosk-*` only): `<emphasis>сл+ово</emphasis>` shifts stress to the vowel marked with `+`. `<emphasis level=\"none\">сл+ово</emphasis>` strips the `+` (cancel inherited emphasis). Other voices (`en-*`, `macos-*`) silently strip the `+` and warn once per process. Auto-stress dictionary not provided — caller writes the `+` manually. Closes [#233](https://github.com/drakulavich/kesha-voice-kit/issues/233).\n\n**Speech rate via SSML** (`ru-vosk-*` and `en-*` voices): wrap the utterance in `<prosody rate=\"…\">` to slow down or speed up synthesis. Supports SSML named values (`x-slow`/`slow`/`medium`/`fast`/`x-fast`), absolute `N%` (e.g. `120%`), and relative `+N%`/`-N%`. Honored only when `<prosody>` wraps the whole utterance — mid-utterance prosody warns and synthesizes at default rate. `--rate` and `<prosody rate>` compose multiplicatively; result is clamped to 0.5×–2.0×. AVSpeech (`macos-*` voices) does not yet accept SSML — see [#236](https://github.com/drakulavich/kesha-voice-kit/issues/236).\n\n## Language detection standalone\n\n`kesha --json audio.ogg` includes both audio-based (`audioLanguage`) and text-based (`textLanguage`) detection. Use audio detection to identify the language before running language-specific logic.\n\n## Install\n\n**Humans use `kesha init` (guided). Agents and scripts use `kesha install` (deterministic).**\n\n```bash\nbun add -g @drakulavich/kesha-voice-kit          # global CLI install (always first)\n\n# For humans: interactive setup that prompts for backend / TTS / VAD / diarize\nkesha init\n\n# For agents and CI: explicit, scriptable install commands\nkesha install                                    # engine only — ~0.6 GB on Apple Silicon, ~2.5 GB on Linux/Windows\nkesha install --plan                             # preview exact download/disk sizes first — downloads nothing\nkesha install --tts                              # + English Kokoro; add langs additively: --tts en ru es\nkesha install --tts --vad                        # + Silero VAD (long-audio chunking)\nkesha install --tts --vad --diarize              # + speaker diarization (darwin-arm64 only)\n```\n\nKesha's runtime error/warning messages adapt to the same split: when `kesha` is invoked from a TTY, hints suggest `kesha init`; when stderr is piped (CI logs, OpenClaw, agent subprocess), hints suggest the equivalent `kesha install [...flags]`. Both run the same install code under the hood — pick the one your caller is.\n\nFor pre-release builds: `bun add -g @drakulavich/kesha-voice-kit@beta` (current `beta` channel; `@latest` stays on the last stable release).\n\nNo system deps — English G2P is embedded (`misaki-rs`); Russian G2P is bundled inside Vosk-TTS. `macos-*` voices need no install either — they use voices already on the Mac.\n\n## Supported languages\n\n**Speech-to-text (25):** Bulgarian, Croatian, Czech, Danish, Dutch, English, Estonian, Finnish, French, German, Greek, Hungarian, Italian, Latvian, Lithuanian, Maltese, Polish, Portuguese, Romanian, Russian, Slovak, Slovenian, Spanish, Swedish, Ukrainian.\n\n**Text-to-speech (9):** English (Kokoro-82M; FluidAudio CoreML on darwin-arm64, ONNX elsewhere), Russian (Vosk-TTS, 5 baked-in speakers — default `ru-vosk-m02`), Spanish, French, Italian, Portuguese (Kokoro — CharsiuG2P on ONNX builds, FluidAudio G2P on darwin-arm64), plus Hindi, Japanese, Chinese (Kokoro, darwin-arm64 only). `hi`/`ja` accept romanized Latin input only; native Devanagari and kana/kanji are rejected with `E_SCRIPT_UNSUPPORTED` — see [#492](https://github.com/drakulavich/kesha-voice-kit/issues/492). Any macOS system voice is also available via `--voice macos-*`.\n\n## Performance\n\n- ASR: ~19× faster than OpenAI Whisper on Apple Silicon (CoreML via FluidAudio), ~2.5× on CPU (ONNX via `ort`).\n- TTS: sub-second latency for short utterances on Apple Silicon.\n\n## Troubleshooting\n\n- `kesha doctor` — collect support diagnostics without changing local state. Add `--json` for machine-readable output, `--redact` to scrub secrets and home paths before sharing.\n- `kesha logs` — manage privacy-safe local diagnostic logs. Default mode is `retain-on-failure` (events buffered in memory, flushed to disk only when a command fails). `kesha logs mode on` captures every run, `kesha logs path` prints the NDJSON file, `kesha logs disable` turns it off entirely.\n- `kesha support-bundle --output bundle.tar.gz` — produce a redacted `.tar.gz` for filing an issue. Add `--include-logs` to bundle a bounded tail of diagnostic logs.\n- `kesha stats` — manage local anonymous performance stats (per-command latency percentiles). Actions: `enable | disable | status | week | errors | export | reset | vacuum | retention`. Stays on the machine.\n\n## Why local\n\nNo API keys to manage. No per-minute billing. Voice data never leaves the machine — important for regulated industries, personal messaging, and anything that shouldn't be in a third-party log.\n\n## Links\n\n- Source: https://github.com/drakulavich/kesha-voice-kit\n- npm: https://www.npmjs.com/package/@drakulavich/kesha-voice-kit\n- Releases: https://github.com/drakulavich/kesha-voice-kit/releases\n\nFile v1.6.1:README.md\n\n<p align=\"center\">\n  <img src=\"https://github.com/drakulavich/kesha-voice-kit/raw/main/docs/assets/logo.png\" alt=\"Kesha Voice Kit\" width=\"200\">\n</p>\n\n<h1 align=\"center\">Kesha Voice Kit</h1>\n\n<p align=\"center\">\n  <a href=\"https://flakiness.io/Laputa/kesha-voice-kit\"><img src=\"https://img.shields.io/endpoint?url=https%3A%2F%2Fflakiness.io%2Fapi%2Fbadge%3Finput%3D%257B%2522badgeToken%2522%253A%2522badge-2IKMRRqUxh9P3w8Ym3Szf0%2522%257D\" alt=\"Tests\"></a>\n  <a href=\"https://www.npmjs.com/package/@drakulavich/kesha-voice-kit\"><img src=\"https://img.shields.io/npm/v/@drakulavich/kesha-voice-kit\" alt=\"npm version\"></a>\n  <a href=\"https://opensource.org/licenses/MIT\"><img src=\"https://img.shields.io/badge/License-MIT-blue.svg\" alt=\"License: MIT\"></a>\n  <a href=\"https://bun.sh\"><img src=\"https://img.shields.io/badge/runtime-Bun-f9f1e1?logo=bun\" alt=\"Bun\"></a>\n</p>\n\n<p align=\"center\"><b>Give your local tools and LLM agents a voice.</b><br>Fast speech-to-text, text-to-speech, voice-activity detection, and language detection in one local-first CLI: Apple Silicon CoreML first, ONNX fallback on supported Linux/Windows builds.</p>\n\n- **Transcribe locally** — [25 languages](docs/languages.md#speech-to-text-25), up to ~19x faster than Whisper on Apple Silicon, ~2.5x on CPU\n- **Speak back** — text-to-speech in [9 languages](docs/languages.md#text-to-speech)\n- **Plug into agents** — ship voice workflows as CLI commands, an MCP server, an <a href=\"docs/openclaw.md\">OpenClaw</a> skill, or a <a href=\"docs/hermes.md\">Hermes</a> agent\n- **Small Rust engine** — single ~60MB binary, no ffmpeg, no Python, no native Node addons\n\n<p align=\"center\">\n  <img src=\"https://github.com/drakulavich/kesha-voice-kit/raw/main/demo.gif\" alt=\"kesha demo — English + Russian transcription with automatic language detection\" width=\"800\">\n</p>\n\n## Quick Start\n\nRuntime: **[Bun](https://bun.sh)** >= 1.3.0 · Platforms: macOS arm64, Linux x64, Windows x64. Linux and Windows run the ONNX engine — everything except microphone capture (`kesha record`), macOS system voices, speaker diarization, and text language detection, which need Apple frameworks.\n\n```bash\n# 1. Install Bun (skip if you have it) — Linux & macOS:\ncurl -fsSL https://bun.sh/install | bash        # or: brew install oven-sh/bun/bun\n# Windows: powershell -c \"irm bun.sh/install.ps1 | iex\"\n# if `bun --version` fails, reload PATH: exec $SHELL -l\n\n# 2. Install Kesha:\nbun add -g @drakulavich/kesha-voice-kit\nkesha --version                                 # confirms `kesha` resolved on PATH\nkesha install --plan                            # preview exact download/disk sizes first — downloads nothing\nkesha install        # ~2.5 GB on Linux/Windows; ~0.6 GB on Apple Silicon, whose CoreML\n                      # engine uses a different, smaller model set. Explicit — never automatic.\n                      # No progress bar during the model step; can take several minutes.\n                      # Prefer a guided wizard? `kesha init` walks through the same choices interactively.\n\n# 3. Transcribe:\nkesha audio.ogg      # transcript to stdout\n```\n\nPrefer Homebrew, `.deb`/`.rpm`, Docker, or Nix? See [Other install methods](#other-install-methods).\nAir-gapped or behind a corporate mirror? See [docs/model-mirror.md](docs/model-mirror.md).\n\n## Speech-to-text\n\n```bash\nkesha audio.ogg                            # transcribe (plain text)\nkesha --format transcript audio.ogg        # text + language/confidence\nkesha --format json audio.ogg              # full JSON with lang fields\nkesha --json --timestamps audio.ogg        # JSON with timestamped segments\nkesha --toon audio.ogg                     # compact LLM-friendly TOON\nkesha status                               # show installed backend info\nkesha status --disk                        # + recursive cache disk usage\nkesha status --json                        # machine-readable, for scripts\n```\n\nMultiple files get `head`-style headers; stdout is the transcript, stderr is errors — pipe-friendly:\n\n```bash\n$ kesha freedom.ogg tahiti.ogg\n=== freedom.ogg ===\nСвободу попугаям! Свободу!\n\n=== tahiti.ogg ===\nТаити, Таити! Не были мы ни в какой Таити! Нас и тут неплохо кормят.\n```\n\n- **Record from the mic (macOS):** `kesha record --out hello.wav` writes microphone audio to a WAV file (`kesha hello.wav` transcribes it). macOS prompts for microphone access on first use — grant it under System Settings → Privacy & Security → Microphone if it was denied. On Linux/Windows or headless boxes, pass any existing audio file straight to `kesha` instead.\n- **Long / silence-heavy audio:** install VAD (`kesha install --vad`); Kesha auto-uses it past 120 s. Without VAD, long audio falls back to fixed ASR chunks. See [docs/vad.md](docs/vad.md).\n- **Speaker diarization** (darwin-arm64): `kesha install --diarize`, then `kesha --json --vad --speakers meeting.m4a` stamps each segment with a `speaker` id. Linux/Windows return a clear \"darwin-arm64 only\" error ([#199](https://github.com/drakulavich/kesha-voice-kit/issues/199)).\n\n## Text-to-speech\n\nKesha speaks back in [9 languages](docs/languages.md#text-to-speech), auto-picking the voice from the text's language. Override with `--lang <code>` or `--voice <id>`.\n\n```bash\nkesha install --tts                              # English voices; sizes differ per platform — preview: kesha install --plan\nkesha install --tts en ru                        # + Russian (+~890 MB, Vosk)\nkesha say \"Hello, world\" > hello.wav\nkesha say \"Привет, мир\" > privet.wav             # auto-routes by language\nkesha say --voice ru-vosk-m02 \"Голос в текст.\" > ru.wav\n```\n\n**Output formats** (`--format`, or inferred from the `--out` extension):\n\n```bash\nkesha say \"Hello\" --out hi.wav                    # WAV (default, uncompressed)\nkesha say \"Hello\" --format ogg-opus --out hi.ogg  # OGG/Opus — messenger voice notes\nkesha say \"Hello\" --format flac --out hi.flac     # FLAC — lossless, plays in every browser incl. Safari/iOS\n```\n\n`kesha say --list-voices` lists what's installed. Voices, the full catalogue, macOS system voices, SSML, speaking rate (`--rate`, `<prosody>`), Russian word stress, and Russian/English abbreviation handling are all in **[docs/tts.md](docs/tts.md)**.\n\n## Languages\n\n**Speech-to-text** spans 25 languages and **text-to-speech** covers English, Russian, and select multilingual voices — full tables with codes and flags in **[docs/languages.md](docs/languages.md)**. Audio language detection identifies [107 languages](https://huggingface.co/speechbrain/lang-id-voxlingua107-ecapa).\n\n## Performance\n\n> **Up to ~19x faster than Whisper** on Apple Silicon (M2), **~2.5x faster** on CPU\n\nCompared against Whisper `large-v3-turbo`, all engines auto-detecting language:\n\n![Benchmark: openai-whisper vs faster-whisper vs Kesha Voice Kit](https://github.com/drakulavich/kesha-voice-kit/raw/main/docs/assets/benchmark.svg)\n\nFull per-file breakdown (Russian + English): [BENCHMARK.md](BENCHMARK.md).\n\n## Other install methods\n\nAll of these install the Bun CLI wrapper; engine + models still download explicitly via `kesha install`.\n\n- **Homebrew** — `brew install drakulavich/tap/kesha-voice-kit` · [docs/homebrew.md](docs/homebrew.md)\n- **Linux packages** (`.deb`/`.rpm`, x64) — [docs/linux-packages.md](docs/linux-packages.md)\n- **Docker** (GHCR image) — [docs/docker.md](docs/docker.md)\n- **Nix** (`aarch64-darwin` / `x86_64-linux`) — `nix run github:drakulavich/kesha-voice-kit -- install` · [docs/nix-install.md](docs/nix-install.md)\n- **Shell completions + manpage** — `kesha completions bash|zsh|fish` and `kesha manpage` print the packaged files to install wherever your shell expects them.\n\n## Integrations\n\n- **MCP server** — `kesha mcp` exposes transcribe/synthesize/list tools to any MCP client (Claude, Cursor, Codex, Gemini). Setup: [docs/mcp.md](docs/mcp.md).\n- **OpenClaw** — give your LLM agent ears. Install & config: [docs/openclaw.md](docs/openclaw.md).\n- **Hermes Agent** — local STT/TTS through Hermes command providers. Setup: [docs/hermes.md](docs/hermes.md).\n- **Raycast** (macOS) — offline microphone dictation from the launcher: *Dictate to Clipboard* records with a live signal meter, auto-stops on silence, transcribes locally, and copies the text. [Install from the Raycast Store](https://www.raycast.com/drakulavich/kesha-voice-kit) · source: [`raycast/`](raycast/).\n- **Programmatic API** — `@drakulavich/kesha-voice-kit/core` for use inside a Bun program. See [docs/api.md](docs/api.md).\n\n## More\n\n- [Architecture](docs/architecture.md) — runtime data flow, the models that ship, the CLI ↔ Rust engine boundary, model pinning, and where tests live.\n- [Use cases](docs/use-cases.md) — copy-paste recipes (transcribe a meeting, speak from OpenClaw, run offline, move the cache).\n- [Product positioning](docs/product-positioning.md) — supported workflows, non-goals, maturity labels, platform matrix.\n- **Diagnostics:** `kesha doctor`, `kesha support-bundle` (redacted `.tar.gz` for issues), and `kesha logs` produce local, content-free diagnostics — see [docs/diagnostic-logs.md](docs/diagnostic-logs.md). Every failure prints a stable `error [CODE]: …` line and a documented [process exit code](docs/errors.md#process-exit-codes).\n- **Scripting & CI:** `--json` (or `--toon`) for machine-readable output, `--quiet`/`-q` to silence progress, and `--no-color` (or `NO_COLOR=1`) for plain logs. Colors switch off automatically when `CI=true`.\n- **Privacy / Local Stats:** Stats are **off by default** and fully local. Opt in with `kesha stats enable` to record content-free operational metrics in a local SQLite database — never networked, never storing audio, transcripts, text, or paths. Full commands & lifecycle: [docs/local-stats.md](docs/local-stats.md).\n\n## Contributing\n\nSee [CONTRIBUTING.md](CONTRIBUTING.md), the [Roadmap](ROADMAP.md) (Now / Next / Later), and the [Decision log](docs/decision-log.md) (why platform/model choices were made — and reversed). Dev setup: `make dev-setup` (Bun, Rust, nextest, platform libs).\n\n## License\n\nMade with 💛🩵 and 🥤 energy under MIT License\n\nFile v1.6.1:_meta.json\n\n{\n  \"ownerId\": \"kn70xsptbaknapzrxhhsqepa4x80ynkp\",\n  \"slug\": \"kesha-voice-kit\",\n  \"version\": \"1.6.1\",\n  \"publishedAt\": 1785948561169\n}\n\nFile v1.6.1:skill-card.md\n\n## Description:\n\nKesha Voice Kit lets agents transcribe audio, synthesize speech, perform speaker diarization, and detect language locally through a CLI or MCP server.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[drakulavich](https://clawhub.ai/user/drakulavich)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and agent builders use this skill to add local voice-message transcription, text-to-speech voice-note generation, language detection, and MCP or OpenClaw audio workflows without cloud speech APIs.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The install path uses mutable remote and global installers with broad local code execution authority.\n\nMitigation: Review install commands before execution, prefer package-manager or pinned installation methods, avoid piping remote installers directly into a shell, and run the tool as a normal non-admin user.\n\nRisk: Installation persists a global CLI plus model assets and can change OpenClaw audio or TTS behavior.\n\nMitigation: Preview planned downloads with installation plan commands, verify OpenClaw audio and TTS configuration after install, and keep the CLI route explicit in user configuration.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/drakulavich/skills/kesha-voice-kit)\n- [npm package](https://www.npmjs.com/package/@drakulavich/kesha-voice-kit)\n- [Bun runtime](https://bun.sh)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, JSON, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown with inline shell commands and JSON configuration examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Agent-facing guidance may direct local CLI calls that produce transcripts, timestamped JSON, language metadata, or generated audio files.]\n\n## Skill Version(s):\n\n1.6.1 (source: server release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v1.6.0: 4 files, 13296 bytes\n\nFiles: README.md (10241b), skill-card.md (3059b), SKILL.md (15630b), _meta.json (134b)\n\nFile v1.6.0:SKILL.md\n\n---\nname: kesha-voice-kit\ndescription: Local multilingual voice toolkit — speech-to-text (STT), text-to-speech (TTS), speaker diarization, and language detection, over a CLI or an MCP server. Runs entirely offline on Apple Silicon, Linux, and Windows. No API keys, no cloud. NVIDIA Parakeet TDT for STT across 25 European languages, Kokoro-82M + Vosk-TTS for TTS in 9 languages, plus macOS AVSpeechSynthesizer for ~180 system voices with zero install.\nemoji: 🎙️\n\nrequires:\n  bins: [kesha]\n\ninstall:\n  - kind: bash\n    cmd: bun add -g \"@drakulavich/kesha-voice-kit\"\n  - kind: bash\n    cmd: kesha install\n---\n\n# kesha-voice-kit\n\nLocal voice toolkit: transcribe voice messages to text, synthesize speech, detect language of audio or text. Fully offline after `kesha install`. No API keys, no per-minute billing.\n\n**Trigger keywords for when to use this skill:** voice message, voice memo, voice note, .ogg, .opus, .wav, .mp3, audio file, transcribe, transcription, speech-to-text, STT, text-to-speech, TTS, synthesize speech, say, telegram voice note, whatsapp voice note, ogg-opus, opus, multilingual voice, multilingual ASR, language detection, speaker diarization, who said what, meeting transcript, MCP server, offline voice, privacy, Apple Silicon, CoreML.\n\n## When to use\n\n- **Voice memo arrived** (Telegram, WhatsApp, Slack, Signal .ogg/.opus/.m4a): transcribe with `kesha --json <path>` and branch on the detected language.\n- **Need to send a voice note (Telegram, WhatsApp, Signal, Discord)**: synthesize directly into messenger-native OGG/Opus with `kesha say --format ogg-opus --out reply.ogg \"<text>\"`. Default is mono 24 kHz @ 32 kbps - what Telegram `sendVoice` expects. No WAV redirect and no `ffmpeg` round-trip.\n- **Need local file playback/debug output**: WAV is still available with `kesha say --out reply.wav \"<text>\"`, but do not use WAV for Telegram voice replies. Auto-routes by detected language (Kokoro-82M for English, Vosk-TTS for Russian). On darwin-arm64, English Kokoro uses FluidAudio CoreML instead of ONNX. For other languages and ~180 more voices use `--voice macos-*` on macOS (zero model download).\n- **Need to detect what language a file is in** before choosing a pipeline: `kesha --json audio.ogg` returns both audio-based and text-based language detection with confidence scores.\n- **Need to capture your own voice** for transcription or as a voice-note source: `kesha record --out clip.wav` records up to 120s (override with `--max-seconds`) of mono 16 kHz WAV from the default microphone. Pipe straight into `kesha --json clip.wav` to close the loop.\n\n## OpenClaw plugin setup\n\nInstall the plugin, then explicitly route OpenClaw audio understanding through the CLI model entry. The plugin registration makes Kesha discoverable, but real voice-message transcription uses `tools.media.audio.models` with a `type: \"cli\"` entry.\n\n```bash\nbun add -g @drakulavich/kesha-voice-kit\nkesha install\nopenclaw plugins install @drakulavich/kesha-voice-kit\nopenclaw config patch --stdin <<'JSON5'\n{\n  tools: {\n    media: {\n      audio: {\n        enabled: true,\n        models: [\n          {\n            type: \"cli\",\n            command: \"kesha\",\n            args: [\"{{MediaPath}}\"],\n            timeoutSeconds: 15,\n          },\n        ],\n        echoTranscript: true,\n        echoFormat: '🦜 \"{transcript}\"',\n      },\n    },\n  },\n}\nJSON5\n```\n\nUse Kesha's default output for OpenClaw's normal voice-message path: stdout is the bare transcript text, while progress and errors stay off the transcript payload. The default setup echoes each transcript back to chat as `🦜 \"{transcript}\"` before the agent responds.\n\nFor agents that need timestamped segments, switch the model entry to JSON output and allow a longer timeout:\n\n```bash\nopenclaw config set tools.media.audio.models \\\n  '[{\"type\":\"cli\",\"command\":\"kesha\",\"args\":[\"--json\",\"--timestamps\",\"{{MediaPath}}\"],\"timeoutSeconds\":30}]'\n```\n\nVerification checklist:\n\n```bash\nwhich kesha\nkesha status\nopenclaw plugins list\nopenclaw config get tools.media.audio.models\nopenclaw config get tools.media.audio.echoTranscript\nopenclaw config get tools.media.audio.echoFormat\n```\n\nDo not rely on `openclaw.plugin.json` to patch `tools.media.audio.models`; OpenClaw ignores non-schema fields such as `configPatch`. Keep the CLI route in user config.\n\nFor OpenClaw TTS replies, route the local TTS provider to Kesha OGG/Opus output. This is the Telegram-safe path:\n\n```bash\nopenclaw config patch --stdin <<'JSON5'\n{\n  messages: {\n    tts: {\n      auto: \"always\",\n      provider: \"tts-local-cli\",\n      providers: {\n        \"tts-local-cli\": {\n          command: \"kesha\",\n          args: [\"say\", \"--format\", \"ogg-opus\", \"--out\", \"{{OutputPath}}\", \"{{Text}}\"],\n          outputFormat: \"opus\",\n          timeoutMs: 120000,\n        },\n      },\n    },\n  },\n}\nJSON5\n```\n\nWhen invoking Kesha manually from an OpenClaw flow, write OGG/Opus into an OpenClaw-owned temp path, for example `kesha say --format ogg-opus --out /tmp/openclaw/reply.ogg \"<text>\"`, after ensuring the directory exists. The configured `tts-local-cli` provider should use OpenClaw's `{{OutputPath}}` placeholder instead of a hardcoded path.\n\nDo not configure OpenClaw Telegram TTS as `kesha say \"<text>\" > reply.wav`; that creates a WAV file and will not render as a native Telegram voice note.\n\n## MCP server\n\n`kesha mcp` serves the same capabilities over Model Context Protocol (stdio), exposing `transcribe_audio`, `synthesize_speech`, `list_voices`, and `list_languages` to any MCP client:\n\n```json\n{ \"mcpServers\": { \"kesha\": { \"command\": \"kesha\", \"args\": [\"mcp\"] } } }\n```\n\nModels are never auto-downloaded — tools fail with a `kesha install` / `kesha install --tts` hint when missing. Full client setup (Claude Code, Claude Desktop, Codex, Cursor): [docs/mcp.md](docs/mcp.md).\n\n## STT: transcribe audio\n\n```bash\n# JSON output with language detection (recommended for automation)\nkesha --json voice.ogg\n```\n\n```json\n[{\n  \"file\": \"voice.ogg\",\n  \"text\": \"Привет, как дела?\",\n  \"lang\": \"ru\",\n  \"audioLanguage\": { \"code\": \"ru\", \"confidence\": 0.98 },\n  \"textLanguage\": { \"code\": \"ru\", \"confidence\": 0.99 }\n}]\n```\n\nUse `lang` (or the more detailed `audioLanguage`/`textLanguage`) to decide how to respond.\n\nNeed timestamped transcript segments for navigation, chapters, or downstream editing:\n\n```bash\nkesha --json --timestamps voice.ogg > voice.timestamps.json\njq '.[0].segments' voice.timestamps.json\n```\n\nEach segment has `start`, `end`, and `text` fields. `--timestamps` is available for machine-readable output (`--json`, `--toon`, or `--format json`).\n\n**Speaker diarization** (darwin-arm64, post-v1.12.0). Add `--speakers` to label each segment with a cluster ID — useful for transcribing multi-person calls / meetings:\n\n```bash\nkesha install --diarize                                  # one-time, ~245MB\nkesha --json --vad --speakers meeting.m4a > out.json\njq '.[0].segments[] | \"\\(.speaker)\\t\\(.text)\"' out.json\n```\n\nEach `segment.speaker` is a number (cluster id, stable within one file). On Linux / Windows the engine returns a clear \"currently darwin-arm64 only\" error — see [#199](https://github.com/drakulavich/kesha-voice-kit/issues/199).\n\n**Formats:** .ogg, .opus, .mp3, .m4a, .wav, .flac, .webm — decoded via symphonia, no ffmpeg required.\n\n**Other output modes:**\n- `kesha audio.ogg` — plain transcript on stdout\n- `kesha --format transcript audio.ogg` — transcript + `[lang: ru, confidence: 0.99]` footer\n- `kesha --json --timestamps audio.ogg` — JSON with timestamped `segments`\n- `kesha --toon audio.ogg` — TOON (compact, LLM-friendly JSON encoding); preferred when piping multi-file results to an LLM/agent\n- `kesha --verbose audio.ogg` — human-readable with language info\n- `kesha --lang en audio.ogg` — warn if detected language differs (useful sanity check)\n\n**Long audio:** files ≥ 120 s auto-engage Silero VAD chunking; force on with `--vad` or off with `--no-vad`. Short files use full-file ASR by default.\n\n## TTS: synthesize speech\n\n```bash\nkesha say \"Hello, world\" > hello.wav               # auto-routes en → Kokoro-82M\nkesha say \"Привет, мир\" > privet.wav              # auto-routes ru → Vosk-TTS\nkesha say --voice macos-de-DE \"Guten Tag\" > de.wav # any macOS system voice — German, French, Italian, ...\nkesha say --list-voices                            # Kokoro + Vosk-TTS + ~180 macos-* voices\n```\n\nOutput: WAV mono float32 by default. `--out <path>` writes to a file instead of stdout. For Telegram/OpenClaw replies, prefer `--format ogg-opus --out reply.ogg` or the OpenClaw-provided `{{OutputPath}}`.\n\n**Output formats** (`--format`, or inferred from the `--out` extension): `wav` (default, uncompressed), `ogg-opus` (messenger voice notes), `flac` (lossless, royalty-free, plays in every browser incl. Safari/iOS — the format for web-embeddable samples). FLAC keeps the engine's native rate; `--bitrate` / `--sample-rate` apply only to `ogg-opus`.\n\n```bash\nkesha say --format flac --out sample.flac \"Hello\"   # web-embeddable, Safari-safe\n```\n\n**Voice notes (Telegram / WhatsApp / Signal / Discord):** add `--format ogg-opus` to emit OGG/Opus directly — the format messenger APIs render as a native voice message:\n\n```bash\nkesha say --format ogg-opus --out reply.ogg \"Hello there\"                  # 24 kHz @ 32 kbps mono - Telegram-grade\nkesha say --voice ru-vosk-m02 --format ogg-opus --out reply.ogg \"Привет\"   # Russian voice note\nkesha say --format ogg-opus --bitrate 16000 --out tiny.ogg \"Hi\"            # tinier file, intelligible but lossy\n```\n\nFormat is also inferred from `--out` extension (`.ogg` / `.opus` / `.oga` → OGG/Opus). `--bitrate` (6 000–510 000 bps) and `--sample-rate` (8 000 / 12 000 / 16 000 / 24 000 / 48 000 Hz) tune the encoder.\n\n**Russian abbreviations** (`ru-vosk-*`): all-uppercase Cyrillic 2-5-char tokens auto-expand letter-by-letter when not pronounceable as a Russian syllable (ФСБ → \"эф-эс-бэ\", ВОЗ → \"воз\"). Disable with `--no-expand-abbrev`. See [docs/tts.md#russian-abbreviation-auto-expansion](docs/tts.md#russian-abbreviation-auto-expansion).\n\n**English acronyms** (`en-*`, Kokoro): three-table mechanism (letter-spell rule + STOP_LIST + IPA_LEXICON) auto-expands FBI → \"ef bee eye\" and gives EPAM/JSON/Anthropic the right IPA. Disable letter-spell with `--no-expand-abbrev`. See [docs/tts.md#english-acronym-auto-expansion](docs/tts.md#english-acronym-auto-expansion).\n\n**Russian word stress** (`ru-vosk-*` only): `<emphasis>сл+ово</emphasis>` shifts stress to the vowel marked with `+`. `<emphasis level=\"none\">сл+ово</emphasis>` strips the `+` (cancel inherited emphasis). Other voices (`en-*`, `macos-*`) silently strip the `+` and warn once per process. Auto-stress dictionary not provided — caller writes the `+` manually. Closes [#233](https://github.com/drakulavich/kesha-voice-kit/issues/233).\n\n**Speech rate via SSML** (`ru-vosk-*` and `en-*` voices): wrap the utterance in `<prosody rate=\"…\">` to slow down or speed up synthesis. Supports SSML named values (`x-slow`/`slow`/`medium`/`fast`/`x-fast`), absolute `N%` (e.g. `120%`), and relative `+N%`/`-N%`. Honored only when `<prosody>` wraps the whole utterance — mid-utterance prosody warns and synthesizes at default rate. `--rate` and `<prosody rate>` compose multiplicatively; result is clamped to 0.5×–2.0×. AVSpeech (`macos-*` voices) does not yet accept SSML — see [#236](https://github.com/drakulavich/kesha-voice-kit/issues/236).\n\n## Language detection standalone\n\n`kesha --json audio.ogg` includes both audio-based (`audioLanguage`) and text-based (`textLanguage`) detection. Use audio detection to identify the language before running language-specific logic.\n\n## Install\n\n**Humans use `kesha init` (guided). Agents and scripts use `kesha install` (deterministic).**\n\n```bash\nbun add -g @drakulavich/kesha-voice-kit          # global CLI install (always first)\n\n# For humans: interactive setup that prompts for backend / TTS / VAD / diarize\nkesha init\n\n# For agents and CI: explicit, scriptable install commands\nkesha install                                    # engine only — ~0.6 GB on Apple Silicon, ~2.5 GB on Linux/Windows\nkesha install --plan                             # preview exact download/disk sizes first — downloads nothing\nkesha install --tts                              # + English Kokoro; add langs additively: --tts en ru es\nkesha install --tts --vad                        # + Silero VAD (long-audio chunking)\nkesha install --tts --vad --diarize              # + speaker diarization (darwin-arm64 only)\n```\n\nKesha's runtime error/warning messages adapt to the same split: when `kesha` is invoked from a TTY, hints suggest `kesha init`; when stderr is piped (CI logs, OpenClaw, agent subprocess), hints suggest the equivalent `kesha install [...flags]`. Both run the same install code under the hood — pick the one your caller is.\n\nFor pre-release builds: `bun add -g @drakulavich/kesha-voice-kit@beta` (current `beta` channel; `@latest` stays on the last stable release).\n\nNo system deps — English G2P is embedded (`misaki-rs`); Russian G2P is bundled inside Vosk-TTS. `macos-*` voices need no install either — they use voices already on the Mac.\n\n## Supported languages\n\n**Speech-to-text (25):** Bulgarian, Croatian, Czech, Danish, Dutch, English, Estonian, Finnish, French, German, Greek, Hungarian, Italian, Latvian, Lithuanian, Maltese, Polish, Portuguese, Romanian, Russian, Slovak, Slovenian, Spanish, Swedish, Ukrainian.\n\n**Text-to-speech (9):** English (Kokoro-82M; FluidAudio CoreML on darwin-arm64, ONNX elsewhere), Russian (Vosk-TTS, 5 baked-in speakers — default `ru-vosk-m02`), Spanish, French, Italian, Portuguese (Kokoro — CharsiuG2P on ONNX builds, FluidAudio G2P on darwin-arm64), plus Hindi, Japanese, Chinese (Kokoro, darwin-arm64 only). `hi`/`ja` accept romanized Latin input only; native Devanagari and kana/kanji are rejected with `E_SCRIPT_UNSUPPORTED` — see [#492](https://github.com/drakulavich/kesha-voice-kit/issues/492). Any macOS system voice is also available via `--voice macos-*`.\n\n## Performance\n\n- ASR: ~19× faster than OpenAI Whisper on Apple Silicon (CoreML via FluidAudio), ~2.5× on CPU (ONNX via `ort`).\n- TTS: sub-second latency for short utterances on Apple Silicon.\n\n## Troubleshooting\n\n- `kesha doctor` — collect support diagnostics without changing local state. Add `--json` for machine-readable output, `--redact` to scrub secrets and home paths before sharing.\n- `kesha logs` — manage privacy-safe local diagnostic logs. Default mode is `retain-on-failure` (events buffered in memory, flushed to disk only when a command fails). `kesha logs mode on` captures every run, `kesha logs path` prints the NDJSON file, `kesha logs disable` turns it off entirely.\n- `kesha support-bundle --output bundle.tar.gz` — produce a redacted `.tar.gz` for filing an issue. Add `--include-logs` to bundle a bounded tail of diagnostic logs.\n- `kesha stats` — manage local anonymous performance stats (per-command latency percentiles). Actions: `enable | disable | status | week | errors | export | reset | vacuum | retention`. Stays on the machine.\n\n## Why local\n\nNo API keys to manage. No per-minute billing. Voice data never leaves the machine — important for regulated industries, personal messaging, and anything that shouldn't be in a third-party log.\n\n## Links\n\n- Source: https://github.com/drakulavich/kesha-voice-kit\n- npm: https://www.npmjs.com/package/@drakulavich/kesha-voice-kit\n- Releases: https://github.com/drakulavich/kesha-voice-kit/releases\n\nFile v1.6.0:README.md\n\n<p align=\"center\">\n  <img src=\"https://github.com/drakulavich/kesha-voice-kit/raw/main/docs/assets/logo.png\" alt=\"Kesha Voice Kit\" width=\"200\">\n</p>\n\n<h1 align=\"center\">Kesha Voice Kit</h1>\n\n<p align=\"center\">\n  <a href=\"https://flakiness.io/Laputa/kesha-voice-kit\"><img src=\"https://img.shields.io/endpoint?url=https%3A%2F%2Fflakiness.io%2Fapi%2Fbadge%3Finput%3D%257B%2522badgeToken%2522%253A%2522badge-2IKMRRqUxh9P3w8Ym3Szf0%2522%257D\" alt=\"Tests\"></a>\n  <a href=\"https://www.npmjs.com/package/@drakulavich/kesha-voice-kit\"><img src=\"https://img.shields.io/npm/v/@drakulavich/kesha-voice-kit\" alt=\"npm version\"></a>\n  <a href=\"https://opensource.org/licenses/MIT\"><img src=\"https://img.shields.io/badge/License-MIT-blue.svg\" alt=\"License: MIT\"></a>\n  <a href=\"https://bun.sh\"><img src=\"https://img.shields.io/badge/runtime-Bun-f9f1e1?logo=bun\" alt=\"Bun\"></a>\n</p>\n\n<p align=\"center\"><b>Give your local tools and LLM agents a voice.</b><br>Fast speech-to-text, text-to-speech, voice-activity detection, and language detection in one local-first CLI: Apple Silicon CoreML first, ONNX fallback on supported Linux/Windows builds.</p>\n\n- **Transcribe locally** — [25 languages](docs/languages.md#speech-to-text-25), up to ~19x faster than Whisper on Apple Silicon, ~2.5x on CPU\n- **Speak back** — text-to-speech in [9 languages](docs/languages.md#text-to-speech)\n- **Plug into agents** — ship voice workflows as CLI commands, an MCP server, an <a href=\"docs/openclaw.md\">OpenClaw</a> skill, or a <a href=\"docs/hermes.md\">Hermes</a> agent\n- **Small Rust engine** — single ~60MB binary, no ffmpeg, no Python, no native Node addons\n\n<p align=\"center\">\n  <img src=\"https://github.com/drakulavich/kesha-voice-kit/raw/main/demo.gif\" alt=\"kesha demo — English + Russian transcription with automatic language detection\" width=\"800\">\n</p>\n\n## Quick Start\n\nRuntime: **[Bun](https://bun.sh)** >= 1.3.0 · Platforms: macOS arm64, Linux x64, Windows x64. Linux and Windows run the ONNX engine — everything except microphone capture (`kesha record`), macOS system voices, speaker diarization, and text language detection, which need Apple frameworks.\n\n```bash\n# 1. Install Bun (skip if you have it) — Linux & macOS:\ncurl -fsSL https://bun.sh/install | bash        # or: brew install oven-sh/bun/bun\n# Windows: powershell -c \"irm bun.sh/install.ps1 | iex\"\n# if `bun --version` fails, reload PATH: exec $SHELL -l\n\n# 2. Install Kesha:\nbun add -g @drakulavich/kesha-voice-kit\nkesha --version                                 # confirms `kesha` resolved on PATH\nkesha install --plan                            # preview exact download/disk sizes first — downloads nothing\nkesha install        # ~2.5 GB on Linux/Windows; ~0.6 GB on Apple Silicon, whose CoreML\n                      # engine uses a different, smaller model set. Explicit — never automatic.\n                      # No progress bar during the model step; can take several minutes.\n                      # Prefer a guided wizard? `kesha init` walks through the same choices interactively.\n\n# 3. Transcribe:\nkesha audio.ogg      # transcript to stdout\n```\n\nPrefer Homebrew, `.deb`/`.rpm`, Docker, or Nix? See [Other install methods](#other-install-methods).\nAir-gapped or behind a corporate mirror? See [docs/model-mirror.md](docs/model-mirror.md).\n\n## Speech-to-text\n\n```bash\nkesha audio.ogg                            # transcribe (plain text)\nkesha --format transcript audio.ogg        # text + language/confidence\nkesha --format json audio.ogg              # full JSON with lang fields\nkesha --json --timestamps audio.ogg        # JSON with timestamped segments\nkesha --toon audio.ogg                     # compact LLM-friendly TOON\nkesha status                               # show installed backend info\nkesha status --disk                        # + recursive cache disk usage\nkesha status --json                        # machine-readable, for scripts\n```\n\nMultiple files get `head`-style headers; stdout is the transcript, stderr is errors — pipe-friendly:\n\n```bash\n$ kesha freedom.ogg tahiti.ogg\n=== freedom.ogg ===\nСвободу попугаям! Свободу!\n\n=== tahiti.ogg ===\nТаити, Таити! Не были мы ни в какой Таити! Нас и тут неплохо кормят.\n```\n\n- **Record from the mic (macOS):** `kesha record --out hello.wav` writes microphone audio to a WAV file (`kesha hello.wav` transcribes it). macOS prompts for microphone access on first use — grant it under System Settings → Privacy & Security → Microphone if it was denied. On Linux/Windows or headless boxes, pass any existing audio file straight to `kesha` instead.\n- **Long / silence-heavy audio:** install VAD (`kesha install --vad`); Kesha auto-uses it past 120 s. Without VAD, long audio falls back to fixed ASR chunks. See [docs/vad.md](docs/vad.md).\n- **Speaker diarization** (darwin-arm64): `kesha install --diarize`, then `kesha --json --vad --speakers meeting.m4a` stamps each segment with a `speaker` id. Linux/Windows return a clear \"darwin-arm64 only\" error ([#199](https://github.com/drakulavich/kesha-voice-kit/issues/199)).\n\n## Text-to-speech\n\nKesha speaks back in [9 languages](docs/languages.md#text-to-speech), auto-picking the voice from the text's language. Override with `--lang <code>` or `--voice <id>`.\n\n```bash\nkesha install --tts                              # English voices; sizes differ per platform — preview: kesha install --plan\nkesha install --tts en ru                        # + Russian (+~890 MB, Vosk)\nkesha say \"Hello, world\" > hello.wav\nkesha say \"Привет, мир\" > privet.wav             # auto-routes by language\nkesha say --voice ru-vosk-m02 \"Голос в текст.\" > ru.wav\n```\n\n**Output formats** (`--format`, or inferred from the `--out` extension):\n\n```bash\nkesha say \"Hello\" --out hi.wav                    # WAV (default, uncompressed)\nkesha say \"Hello\" --format ogg-opus --out hi.ogg  # OGG/Opus — messenger voice notes\nkesha say \"Hello\" --format flac --out hi.flac     # FLAC — lossless, plays in every browser incl. Safari/iOS\n```\n\n`kesha say --list-voices` lists what's installed. Voices, the full catalogue, macOS system voices, SSML, speaking rate (`--rate`, `<prosody>`), Russian word stress, and Russian/English abbreviation handling are all in **[docs/tts.md](docs/tts.md)**.\n\n## Languages\n\n**Speech-to-text** spans 25 languages and **text-to-speech** covers English, Russian, and select multilingual voices — full tables with codes and flags in **[docs/languages.md](docs/languages.md)**. Audio language detection identifies [107 languages](https://huggingface.co/speechbrain/lang-id-voxlingua107-ecapa).\n\n## Performance\n\n> **Up to ~19x faster than Whisper** on Apple Silicon (M2), **~2.5x faster** on CPU\n\nCompared against Whisper `large-v3-turbo`, all engines auto-detecting language:\n\n![Benchmark: openai-whisper vs faster-whisper vs Kesha Voice Kit](https://github.com/drakulavich/kesha-voice-kit/raw/main/docs/assets/benchmark.svg)\n\nFull per-file breakdown (Russian + English): [BENCHMARK.md](BENCHMARK.md).\n\n## Other install methods\n\nAll of these install the Bun CLI wrapper; engine + models still download explicitly via `kesha install`.\n\n- **Homebrew** — `brew install drakulavich/tap/kesha-voice-kit` · [docs/homebrew.md](docs/homebrew.md)\n- **Linux packages** (`.deb`/`.rpm`, x64) — [docs/linux-packages.md](docs/linux-packages.md)\n- **Docker** (GHCR image) — [docs/docker.md](docs/docker.md)\n- **Nix** (`aarch64-darwin` / `x86_64-linux`) — `nix run github:drakulavich/kesha-voice-kit -- install` · [docs/nix-install.md](docs/nix-install.md)\n- **Shell completions + manpage** — `kesha completions bash|zsh|fish` and `kesha manpage` print the packaged files to install wherever your shell expects them.\n\n## Integrations\n\n- **MCP server** — `kesha mcp` exposes transcribe/synthesize/list tools to any MCP client (Claude, Cursor, Codex, Gemini). Setup: [docs/mcp.md](docs/mcp.md).\n- **OpenClaw** — give your LLM agent ears. Install & config: [docs/openclaw.md](docs/openclaw.md).\n- **Hermes Agent** — local STT/TTS through Hermes command providers. Setup: [docs/hermes.md](docs/hermes.md).\n- **Raycast** (macOS) — offline microphone dictation from the launcher: *Dictate to Clipboard* records with a live signal meter, auto-stops on silence, transcribes locally, and copies the text. [Install from the Raycast Store](https://www.raycast.com/drakulavich/kesha-voice-kit) · source: [`raycast/`](raycast/).\n- **Programmatic API** — `@drakulavich/kesha-voice-kit/core` for use inside a Bun program. See [docs/api.md](docs/api.md).\n\n## More\n\n- [Architecture](docs/architecture.md) — runtime data flow, the models that ship, the CLI ↔ Rust engine boundary, model pinning, and where tests live.\n- [Use cases](docs/use-cases.md) — copy-paste recipes (transcribe a meeting, speak from OpenClaw, run offline, move the cache).\n- [Product positioning](docs/product-positioning.md) — supported workflows, non-goals, maturity labels, platform matrix.\n- **Diagnostics:** `kesha doctor`, `kesha support-bundle` (redacted `.tar.gz` for issues), and `kesha logs` produce local, content-free diagnostics — see [docs/diagnostic-logs.md](docs/diagnostic-logs.md). Every failure prints a stable `error [CODE]: …` line and a documented [process exit code](docs/errors.md#process-exit-codes).\n- **Scripting & CI:** `--json` (or `--toon`) for machine-readable output, `--quiet`/`-q` to silence progress, and `--no-color` (or `NO_COLOR=1`) for plain logs. Colors switch off automatically when `CI=true`.\n- **Privacy / Local Stats:** Stats are **off by default** and fully local. Opt in with `kesha stats enable` to record content-free operational metrics in a local SQLite database — never networked, never storing audio, transcripts, text, or paths. Full commands & lifecycle: [docs/local-stats.md](docs/local-stats.md).\n\n## Contributing\n\nSee [CONTRIBUTING.md](CONTRIBUTING.md), the [Roadmap](ROADMAP.md) (Now / Next / Later), and the [Decision log](docs/decision-log.md) (why platform/model choices were made — and reversed). Dev setup: `make dev-setup` (Bun, Rust, nextest, platform libs).\n\n## License\n\nMade with 💛🩵 and 🥤 energy under MIT License\n\nFile v1.6.0:_meta.json\n\n{\n  \"ownerId\": \"kn70xsptbaknapzrxhhsqepa4x80ynkp\",\n  \"slug\": \"kesha-voice-kit\",\n  \"version\": \"1.6.0\",\n  \"publishedAt\": 1785948441107\n}\n\nFile v1.6.0:skill-card.md\n\n## Description:\n\nLocal multilingual voice toolkit for speech-to-text, text-to-speech, speaker diarization, and language detection over a CLI or MCP server, running offline on Apple Silicon, Linux, and Windows.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[drakulavich](https://clawhub.ai/user/drakulavich)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and agent builders use this skill to add local voice workflows to agents: transcribing audio files or voice notes, synthesizing spoken replies, detecting language, and configuring OpenClaw or MCP clients to call the local Kesha CLI.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill installs a global Bun package and can download large local voice models.\n\nMitigation: Install only when a local voice CLI is intended, and run `kesha install --plan` before downloading models to review platform-specific size and disk impact.\n\nRisk: Recording workflows may request microphone access on supported systems.\n\nMitigation: Grant microphone permissions only when local recording is required; otherwise pass existing audio files directly to the CLI.\n\nRisk: OpenClaw configuration can echo transcripts or generate user-visible TTS replies.\n\nMitigation: Review the OpenClaw audio and TTS configuration before enabling it, especially `echoTranscript`, transcript formatting, and automatic TTS behavior.\n\nRisk: Broad activation keywords may cause the agent to consider this skill for many audio-adjacent requests.\n\nMitigation: Confirm that the user actually wants local transcription, synthesis, language detection, or diarization before running CLI commands.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/drakulavich/skills/kesha-voice-kit)\n- [Publisher profile](https://clawhub.ai/user/drakulavich)\n- [npm package](https://www.npmjs.com/package/@drakulavich/kesha-voice-kit)\n- [Project repository](https://github.com/drakulavich/kesha-voice-kit)\n- [Project releases](https://github.com/drakulavich/kesha-voice-kit/releases)\n- [Speaker diarization platform limitation](https://github.com/drakulavich/kesha-voice-kit/issues/199)\n- [Unsupported native scripts for hi/ja TTS](https://github.com/drakulavich/kesha-voice-kit/issues/492)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, code, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown guidance with inline shell, JSON, and JSON5 configuration examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Guides agents toward local CLI and MCP voice workflows; runtime commands may produce plain transcript text, JSON, TOON, WAV, OGG/Opus, or FLAC depending on the selected Kesha command.]\n\n## Skill Version(s):\n\n1.6.0 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v1.5.0: 5 files, 14270 bytes\n\nFiles: LICENSE (1068b), README.md (17208b), skill-card.md (2394b), SKILL.md (11683b), _meta.json (134b)\n\nFile v1.5.0:SKILL.md\n\n---\nname: kesha-voice-kit\ndescription: Local multilingual voice toolkit — speech-to-text (STT), text-to-speech (TTS), and language detection. Runs entirely offline on Apple Silicon, Linux, and Windows. No API keys, no cloud. NVIDIA Parakeet TDT for STT across 25 European languages, Kokoro-82M + Vosk-TTS for TTS, plus macOS AVSpeechSynthesizer for ~180 system voices with zero install.\nemoji: 🎙️\n\nrequires:\n  bins: [kesha]\n\ninstall:\n  - kind: bash\n    cmd: bun add -g \"@drakulavich/kesha-voice-kit\"\n  - kind: bash\n    cmd: kesha install\n---\n\n# kesha-voice-kit\n\nLocal voice toolkit: transcribe voice messages to text, synthesize speech, detect language of audio or text. Fully offline after `kesha install`. No API keys, no per-minute billing.\n\n**Trigger keywords for when to use this skill:** voice message, voice memo, voice note, .ogg, .opus, .wav, .mp3, audio file, transcribe, transcription, speech-to-text, STT, text-to-speech, TTS, synthesize speech, say, telegram voice note, whatsapp voice note, ogg-opus, opus, multilingual voice, multilingual ASR, language detection, offline voice, privacy, Apple Silicon, CoreML.\n\n## When to use\n\n- **Voice memo arrived** (Telegram, WhatsApp, Slack, Signal .ogg/.opus/.m4a): transcribe with `kesha --json <path>` and branch on the detected language.\n- **Need to send a voice note (Telegram, WhatsApp, Signal, Discord)**: synthesize directly into messenger-native OGG/Opus with `kesha say --format ogg-opus --out reply.ogg \"<text>\"`. Default is mono 24 kHz @ 32 kbps - what Telegram `sendVoice` expects. No WAV redirect and no `ffmpeg` round-trip.\n- **Need local file playback/debug output**: WAV is still available with `kesha say --out reply.wav \"<text>\"`, but do not use WAV for Telegram voice replies. Auto-routes by detected language (Kokoro-82M for English, Vosk-TTS for Russian). On darwin-arm64, English Kokoro uses FluidAudio CoreML instead of ONNX. For other languages and ~180 more voices use `--voice macos-*` on macOS (zero model download).\n- **Need to detect what language a file is in** before choosing a pipeline: `kesha --json audio.ogg` returns both audio-based and text-based language detection with confidence scores.\n\n## OpenClaw plugin setup\n\nInstall the plugin, then explicitly route OpenClaw audio understanding through the CLI model entry. The plugin registration makes Kesha discoverable, but real voice-message transcription uses `tools.media.audio.models` with a `type: \"cli\"` entry.\n\n```bash\nbun add -g @drakulavich/kesha-voice-kit\nkesha install\nopenclaw plugins install @drakulavich/kesha-voice-kit\nopenclaw config patch --stdin <<'JSON5'\n{\n  tools: {\n    media: {\n      audio: {\n        enabled: true,\n        models: [\n          {\n            type: \"cli\",\n            command: \"kesha\",\n            args: [\"{{MediaPath}}\"],\n            timeoutSeconds: 15,\n          },\n        ],\n        echoTranscript: true,\n        echoFormat: '🦜 \"{transcript}\"',\n      },\n    },\n  },\n}\nJSON5\n```\n\nUse Kesha's default output for OpenClaw's normal voice-message path: stdout is the bare transcript text, while progress and errors stay off the transcript payload. The default setup echoes each transcript back to chat as `🦜 \"{transcript}\"` before the agent responds.\n\nFor agents that need timestamped segments, switch the model entry to JSON output and allow a longer timeout:\n\n```bash\nopenclaw config set tools.media.audio.models \\\n  '[{\"type\":\"cli\",\"command\":\"kesha\",\"args\":[\"--json\",\"--timestamps\",\"{{MediaPath}}\"],\"timeoutSeconds\":30}]'\n```\n\nVerification checklist:\n\n```bash\nwhich kesha\nkesha status\nopenclaw plugins list\nopenclaw config get tools.media.audio.models\nopenclaw config get tools.media.audio.echoTranscript\nopenclaw config get tools.media.audio.echoFormat\n```\n\nDo not rely on `openclaw.plugin.json` to patch `tools.media.audio.models`; OpenClaw ignores non-schema fields such as `configPatch`. Keep the CLI route in user config.\n\nFor OpenClaw TTS replies, route the local TTS provider to Kesha OGG/Opus output. This is the Telegram-safe path:\n\n```bash\nopenclaw config patch --stdin <<'JSON5'\n{\n  messages: {\n    tts: {\n      auto: \"always\",\n      provider: \"tts-local-cli\",\n      providers: {\n        \"tts-local-cli\": {\n          command: \"kesha\",\n          args: [\"say\", \"--format\", \"ogg-opus\", \"--out\", \"{{OutputPath}}\", \"{{Text}}\"],\n          outputFormat: \"opus\",\n          timeoutMs: 120000,\n        },\n      },\n    },\n  },\n}\nJSON5\n```\n\nWhen invoking Kesha manually from an OpenClaw flow, write OGG/Opus into an OpenClaw-owned temp path, for example `kesha say --format ogg-opus --out /tmp/openclaw/reply.ogg \"<text>\"`, after ensuring the directory exists. The configured `tts-local-cli` provider should use OpenClaw's `{{OutputPath}}` placeholder instead of a hardcoded path.\n\nDo not configure OpenClaw Telegram TTS as `kesha say \"<text>\" > reply.wav`; that creates a WAV file and will not render as a native Telegram voice note.\n\n## STT: transcribe audio\n\n```bash\n# JSON output with language detection (recommended for automation)\nkesha --json voice.ogg\n```\n\n```json\n[{\n  \"file\": \"voice.ogg\",\n  \"text\": \"Привет, как дела?\",\n  \"lang\": \"ru\",\n  \"audioLanguage\": { \"code\": \"ru\", \"confidence\": 0.98 },\n  \"textLanguage\": { \"code\": \"ru\", \"confidence\": 0.99 }\n}]\n```\n\nUse `lang` (or the more detailed `audioLanguage`/`textLanguage`) to decide how to respond.\n\nNeed timestamped transcript segments for navigation, chapters, or downstream editing:\n\n```bash\nkesha --json --timestamps voice.ogg > voice.timestamps.json\njq '.[0].segments' voice.timestamps.json\n```\n\nEach segment has `start`, `end`, and `text` fields. `--timestamps` is available for machine-readable output (`--json`, `--toon`, or `--format json`).\n\n**Speaker diarization** (darwin-arm64, post-v1.12.0). Add `--speakers` to label each segment with a cluster ID — useful for transcribing multi-person calls / meetings:\n\n```bash\nkesha install --diarize                                  # one-time, ~245MB\nkesha --json --vad --speakers meeting.m4a > out.json\njq '.[0].segments[] | \"\\(.speaker)\\t\\(.text)\"' out.json\n```\n\nEach `segment.speaker` is a number (cluster id, stable within one file). On Linux / Windows the engine returns a clear \"currently darwin-arm64 only\" error — see [#199](https://github.com/drakulavich/kesha-voice-kit/issues/199).\n\n**Formats:** .ogg, .opus, .mp3, .m4a, .wav, .flac, .webm — decoded via symphonia, no ffmpeg required.\n\n**Other output modes:**\n- `kesha audio.ogg` — plain transcript on stdout\n- `kesha --format transcript audio.ogg` — transcript + `[lang: ru, confidence: 0.99]` footer\n- `kesha --json --timestamps audio.ogg` — JSON with timestamped `segments`\n- `kesha --verbose audio.ogg` — human-readable with language info\n- `kesha --lang en audio.ogg` — warn if detected language differs (useful sanity check)\n\n## TTS: synthesize speech\n\n```bash\nkesha say \"Hello, world\" > hello.wav               # auto-routes en → Kokoro-82M\nkesha say \"Привет, мир\" > privet.wav              # auto-routes ru → Vosk-TTS\nkesha say --voice macos-de-DE \"Guten Tag\" > de.wav # any macOS system voice — German, French, Italian, ...\nkesha say --list-voices                            # Kokoro + Vosk-TTS + ~180 macos-* voices\n```\n\nOutput: WAV mono float32 by default. `--out <path>` writes to a file instead of stdout. For Telegram/OpenClaw replies, prefer `--format ogg-opus --out reply.ogg` or the OpenClaw-provided `{{OutputPath}}`.\n\n**Voice notes (Telegram / WhatsApp / Signal / Discord):** add `--format ogg-opus` to emit OGG/Opus directly — the format messenger APIs render as a native voice message:\n\n```bash\nkesha say --format ogg-opus --out reply.ogg \"Hello there\"                  # 24 kHz @ 32 kbps mono - Telegram-grade\nkesha say --voice ru-vosk-m02 --format ogg-opus --out reply.ogg \"Привет\"   # Russian voice note\nkesha say --format ogg-opus --bitrate 16000 --out tiny.ogg \"Hi\"            # tinier file, intelligible but lossy\n```\n\nFormat is also inferred from `--out` extension (`.ogg` / `.opus` / `.oga` → OGG/Opus). `--bitrate` (6 000–510 000 bps) and `--sample-rate` (8 000 / 12 000 / 16 000 / 24 000 / 48 000 Hz) tune the encoder.\n\n**Russian abbreviations** (`ru-vosk-*`): all-uppercase Cyrillic 2-5-char tokens auto-expand letter-by-letter when not pronounceable as a Russian syllable (ФСБ → \"эф-эс-бэ\", ВОЗ → \"воз\"). Disable with `--no-expand-abbrev`. See [docs/tts.md#russian-abbreviation-auto-expansion](docs/tts.md#russian-abbreviation-auto-expansion).\n\n**English acronyms** (`en-*`, Kokoro): three-table mechanism (letter-spell rule + STOP_LIST + IPA_LEXICON) auto-expands FBI → \"ef bee eye\" and gives EPAM/JSON/Anthropic the right IPA. Disable letter-spell with `--no-expand-abbrev`. See [docs/tts.md#english-acronym-auto-expansion](docs/tts.md#english-acronym-auto-expansion).\n\n**Russian word stress** (`ru-vosk-*` only): `<emphasis>сл+ово</emphasis>` shifts stress to the vowel marked with `+`. `<emphasis level=\"none\">сл+ово</emphasis>` strips the `+` (cancel inherited emphasis). Other voices (`en-*`, `macos-*`) silently strip the `+` and warn once per process. Auto-stress dictionary not provided — caller writes the `+` manually. Closes [#233](https://github.com/drakulavich/kesha-voice-kit/issues/233).\n\n**Speech rate via SSML** (`ru-vosk-*` and `en-*` voices): wrap the utterance in `<prosody rate=\"…\">` to slow down or speed up synthesis. Supports SSML named values (`x-slow`/`slow`/`medium`/`fast`/`x-fast`), absolute `N%` (e.g. `120%`), and relative `+N%`/`-N%`. Honored only when `<prosody>` wraps the whole utterance — mid-utterance prosody warns and synthesizes at default rate. `--rate` and `<prosody rate>` compose multiplicatively; result is clamped to 0.5×–2.0×. AVSpeech (`macos-*` voices) does not yet accept SSML — see [#236](https://github.com/drakulavich/kesha-voice-kit/issues/236).\n\n## Language detection standalone\n\n`kesha --json audio.ogg` includes both audio-based (`audioLanguage`) and text-based (`textLanguage`) detection. Use audio detection to identify the language before running language-specific logic.\n\n## Install\n\n```bash\nbun add -g @drakulavich/kesha-voice-kit          # global CLI install\nkesha install                                    # downloads engine (~350 MB)\nkesha install --tts                              # adds Kokoro + Vosk-TTS RU (~990 MB more, for TTS)\n```\n\nNo system deps — English G2P is embedded (`misaki-rs`); Russian G2P is bundled inside Vosk-TTS. `macos-*` voices need no install either — they use voices already on the Mac.\n\n## Supported languages\n\n**Speech-to-text (25):** Bulgarian, Croatian, Czech, Danish, Dutch, English, Estonian, Finnish, French, German, Greek, Hungarian, Italian, Latvian, Lithuanian, Maltese, Polish, Portuguese, Romanian, Russian, Slovak, Slovenian, Spanish, Swedish, Ukrainian.\n\n**Text-to-speech:** English (Kokoro-82M; FluidAudio CoreML on darwin-arm64, ONNX elsewhere), Russian (Vosk-TTS, 5 baked-in speakers — default `ru-vosk-m02`), plus any macOS system voice via `--voice macos-*`.\n\n## Performance\n\n- ASR: ~19× faster than OpenAI Whisper on Apple Silicon (CoreML via FluidAudio), ~2.5× on CPU (ONNX via `ort`).\n- TTS: sub-second latency for short utterances on Apple Silicon.\n\n## Why local\n\nNo API keys to manage. No per-minute billing. Voice data never leaves the machine — important for regulated industries, personal messaging, and anything that shouldn't be in a third-party log.\n\n## Links\n\n- Source: https://github.com/drakulavich/kesha-voice-kit\n- npm: https://www.npmjs.com/package/@drakulavich/kesha-voice-kit\n- Releases: https://github.com/drakulavich/kesha-voice-kit/releases\n\nFile v1.5.0:README.md\n\n<p align=\"center\">\n  <img src=\"docs/assets/logo.png\" alt=\"Kesha Voice Kit\" width=\"200\">\n</p>\n\n<h1 align=\"center\">Kesha Voice Kit</h1>\n\n<p align=\"center\">\n  <a href=\"https://flakiness.io/Laputa/kesha-voice-kit\"><img src=\"https://img.shields.io/endpoint?url=https%3A%2F%2Fflakiness.io%2Fapi%2Fbadge%3Finput%3D%257B%2522badgeToken%2522%253A%2522badge-2IKMRRqUxh9P3w8Ym3Szf0%2522%257D\" alt=\"Tests\"></a>\n  <a href=\"https://www.npmjs.com/package/@drakulavich/kesha-voice-kit\"><img src=\"https://img.shields.io/npm/v/@drakulavich/kesha-voice-kit\" alt=\"npm version\"></a>\n  <a href=\"https://opensource.org/licenses/MIT\"><img src=\"https://img.shields.io/badge/License-MIT-blue.svg\" alt=\"License: MIT\"></a>\n  <a href=\"https://bun.sh\"><img src=\"https://img.shields.io/badge/runtime-Bun-f9f1e1?logo=bun\" alt=\"Bun\"></a>\n</p>\n\n<p align=\"center\"><b>Give your local tools and LLM agents a voice.</b><br>Fast speech-to-text, text-to-speech, voice-activity detection, and language detection in one local-first CLI: Apple Silicon CoreML first, ONNX fallback on supported Linux/Windows builds.</p>\n\n- **Transcribe locally** — 25 languages, up to ~19x faster than Whisper on Apple Silicon, ~2.5x on CPU\n- **Speak back** — Kokoro (EN), Vosk-TTS (RU), macOS system voices, and SSML preview\n- **Plug into agents** — ship voice workflows as CLI commands or an <a href=\"https://github.com/openclaw/openclaw\">OpenClaw</a> skill; setup docs: [OpenClaw](docs/openclaw.md), [Hermes](docs/hermes.md)\n- **Small Rust engine** — single ~20MB binary, no ffmpeg, no Python, no native Node addons\n\nSee [Product positioning](docs/product-positioning.md) for supported workflows, non-goals, maturity labels, and the platform matrix.\n\n<p align=\"center\">\n  <img src=\"./demo.gif\" alt=\"kesha demo — English + Russian transcription with automatic language detection\" width=\"800\">\n</p>\n\n## Quick Start\n\nRuntime: **[Bun](https://bun.sh)** >= 1.3.0.\n\nInstall Bun (skip if already installed) — pick one:\n\n```bash\n# Linux & macOS\ncurl -fsSL https://bun.sh/install | bash       # upstream installer\nbrew install oven-sh/bun/bun                   # Homebrew\n```\n\n```powershell\n# Windows\npowershell -c \"irm bun.sh/install.ps1 | iex\"\n```\n\nThen install Kesha:\n\n```bash\nbun add -g @drakulavich/kesha-voice-kit\nkesha install       # downloads engine + models\nkesha audio.ogg     # transcript to stdout\n```\n\nAir-gapped or behind a corporate mirror? See [docs/model-mirror.md](docs/model-mirror.md).\n\n## Requirements\n\n- [Bun](https://bun.sh) >= 1.3\n- macOS arm64, Linux x64, or Windows x64\n\n## Speech-to-text\n\n```bash\nkesha audio.ogg                            # transcribe (plain text)\nkesha --format transcript audio.ogg        # text + language/confidence\nkesha --format json audio.ogg              # full JSON with lang fields\nkesha --json --timestamps audio.ogg        # JSON with timestamped segments\nkesha --toon audio.ogg                     # compact LLM-friendly TOON\nkesha --verbose audio.ogg                  # show language detection details\nkesha --lang en audio.ogg                  # warn if detected language differs\nkesha status                               # show installed backend info\n```\n\nMultiple files — headers per file, like `head`:\n\n```bash\n$ kesha freedom.ogg tahiti.ogg\n=== freedom.ogg ===\nСвободу попугаям! Свободу!\n\n=== tahiti.ogg ===\nТаити, Таити! Не были мы ни в какой Таити! Нас и тут неплохо кормят.\n```\n\nStdout: transcript. Stderr: errors. Pipe-friendly.\n\nFor long / silence-heavy audio, install VAD (`kesha install --vad`) and run without `--no-vad`. Kesha auto-uses VAD past 120 s when installed; without VAD, very long audio falls back to fixed ASR chunks. Details: [docs/vad.md](docs/vad.md).\n\n**Speaker diarization** (darwin-arm64, post-v1.12.0):\n\n```bash\nkesha install --diarize                        # one-time, ~245MB Sortformer model\nkesha --json --vad --speakers meeting.m4a > out.json\njq '.[0].segments[] | \"\\(.speaker)\\t\\(.text)\"' out.json\n```\n\nEach segment gets a `speaker` integer (cluster ID, stable within one file). Linux / Windows: `--speakers` returns a clear \"currently darwin-arm64 only\" error — see [#199](https://github.com/drakulavich/kesha-voice-kit/issues/199).\n\n## Text-to-speech\n\nKesha speaks back via Kokoro-82M (English) and Vosk-TTS (Russian) — voice auto-picks from the text's language. On darwin-arm64, Kokoro uses FluidAudio CoreML instead of ONNX; FluidAudio stores that CoreML cache at `~/.cache/fluidaudio/Models/kokoro`, outside Kesha's pinned model cache:\n\n```bash\nkesha install --tts                      # TTS, opt-in; Darwin Kokoro uses FluidAudio cache\nkesha say \"Hello, world\" > hello.wav\nkesha say \"Привет, мир\" > privet.wav     # auto-routes (Milena on darwin, ru-vosk-m02 elsewhere)\n```\n\n**Russian abbreviations** (`ru-vosk-*`): all-uppercase Cyrillic 2-5-char tokens auto-expand letter-by-letter when not pronounceable as a Russian syllable (ФСБ → \"эф-эс-бэ\", ВОЗ → \"воз\"). Disable with `--no-expand-abbrev`. See [docs/tts.md#russian-abbreviation-auto-expansion](docs/tts.md#russian-abbreviation-auto-expansion).\n\n**English acronyms** (`en-*`, Kokoro): three-table mechanism (letter-spell rule + STOP_LIST + IPA_LEXICON) auto-expands FBI → \"ef bee eye\" and gives EPAM/JSON/Anthropic the right IPA. Disable letter-spell with `--no-expand-abbrev`. See [docs/tts.md#english-acronym-auto-expansion](docs/tts.md#english-acronym-auto-expansion).\n\n**Russian word stress** (`ru-vosk-*` voices):\n\n```bash\n# Caller provides `+` before the stressed vowel; engine passes it to Vosk\nkesha say --voice ru-vosk-m02 --ssml \\\n  '<speak><emphasis>дом+а</emphasis></speak>'   # genitive до-МА́\n\n# Suppress an inherited <emphasis> with level=\"none\"\nkesha say --voice ru-vosk-m02 --ssml \\\n  '<speak><emphasis level=\"none\">дом+а</emphasis></speak>'   # default ДО́ма\n```\n\nVosk-TTS 0.9-multi honors a `+` placed BEFORE the target stressed vowel — but only when the marker shifts stress AWAY from the model's default (first-syllable). `+` agreeing with the default is a no-op. See [#233](https://github.com/drakulavich/kesha-voice-kit/issues/233).\n\n**Speech rate via SSML** (`ru-vosk-*` and `en-*` voices):\n\n```bash\nkesha say --voice ru-vosk-m02 --ssml \\\n  '<speak><prosody rate=\"slow\">Привет, как дела.</prosody></speak>' --out slow.wav\n\nkesha say --voice en-am_michael --ssml \\\n  '<speak><prosody rate=\"x-fast\">Read this fast.</prosody></speak>' --out fast.wav\n```\n\nHonored when `<prosody rate>` wraps the whole utterance. Mid-utterance prosody warns and synthesizes at default rate (whole-segment-only is a v1 limitation; mid-utterance support tracked in [#236](https://github.com/drakulavich/kesha-voice-kit/issues/236)). `--rate` and `<prosody rate>` compose multiplicatively. Range clamped to 0.5×–2.0×.\n\nmacOS system voices, SSML, voice listing, and the full voice catalogue: [docs/tts.md](docs/tts.md).\n\n## Homebrew Install\n\nHomebrew installs the Bun-based CLI wrapper. Engine and model downloads remain\nexplicit:\n\n```bash\nbrew tap oven-sh/bun\nbrew install drakulavich/tap/kesha-voice-kit\nkesha install\nkesha audio.ogg\n```\n\nSee [Homebrew install](docs/homebrew.md) for package scope and maintainer\nvalidation.\n\n## Linux Packages\n\nStable engine releases also publish `.deb` and `.rpm` packages for Linux x64.\nThey install the standalone CLI wrapper; engine and model downloads remain explicit:\n\n```bash\nkesha install\nkesha audio.ogg\n```\n\nSee [Linux packages](docs/linux-packages.md) for install commands and package\nscope.\n\n## Docker\n\nLinux x64 CLI image, published to GHCR:\n\n```bash\ndocker run --rm \\\n  -v kesha-cache:/cache/kesha \\\n  -v \"$PWD:/work\" -w /work \\\n  ghcr.io/drakulavich/kesha-voice-kit:latest install\n\ndocker run --rm \\\n  -v kesha-cache:/cache/kesha \\\n  -v \"$PWD:/work\" -w /work \\\n  ghcr.io/drakulavich/kesha-voice-kit:latest audio.ogg\n```\n\nThe image keeps model downloads and the engine cache under `/cache/kesha`.\nMount that path as a named volume so `kesha install`, TTS models, VAD, and future\nruns reuse the same cache. `compose.yml` provides the same layout:\n\n```bash\ndocker compose run --rm kesha install\ndocker compose run --rm kesha audio.ogg\n```\n\n## Nix Install\n\nAlternative reproducible-build path on `aarch64-darwin` / `x86_64-linux`:\n\n```bash\nnix run github:drakulavich/kesha-voice-kit -- install      # downloads models (engine is bundled)\nnix run github:drakulavich/kesha-voice-kit -- audio.ogg    # transcribe\n```\n\nFull recipes (one-liner, profile install, engine-only, dev shell) live in [docs/nix-install.md](docs/nix-install.md).\n\n## Shell Completions and Manpage\n\nThe npm package includes bash, zsh, and fish completions plus `kesha(1)`.\nThe CLI can print the packaged files, so install paths do not depend on the\nBun global package layout:\n\n```bash\n# bash\nmkdir -p ~/.local/share/bash-completion/completions\nkesha completions bash > ~/.local/share/bash-completion/completions/kesha\n\n# zsh\nmkdir -p ~/.zsh/completions\nkesha completions zsh > ~/.zsh/completions/_kesha\n# add to ~/.zshrc once: fpath=(~/.zsh/completions $fpath); autoload -Uz compinit; compinit\n\n# fish\nmkdir -p ~/.config/fish/completions\nkesha completions fish > ~/.config/fish/completions/kesha.fish\n\n# manpage\nmkdir -p ~/.local/share/man/man1\nkesha manpage > ~/.local/share/man/man1/kesha.1\nmandb ~/.local/share/man 2>/dev/null || true\n```\n\n## Performance\n\n> **Up to ~19x faster than Whisper** on Apple Silicon (M2), **~2.5x faster** on CPU\n\nCompared against Whisper `large-v3-turbo` — all engines auto-detect language.\n\n![Benchmark: openai-whisper vs faster-whisper vs Kesha Voice Kit](docs/assets/benchmark.svg)\n\nSee [BENCHMARK.md](BENCHMARK.md) for the full per-file breakdown (Russian + English).\n\n## Architecture\n\n```text\nUsers / agents\n  shell | scripts | OpenClaw | Hermes | Raycast | @drakulavich/kesha-voice-kit/core\n        |\n        v\n+------------------------------- Kesha CLI -------------------------------+\n| Bun + TypeScript wrapper                                                |\n| - parses commands and formats stdout/stderr                             |\n| - installs pinned engine/model assets only when explicitly requested    |\n| - keeps cache, support bundles, and local Stats in the CLI              |\n+-----------------------------------+-------------------------------------+\n                                   |\n                                   | spawns one local process\n                                   v\n+----------------------------- kesha-engine ------------------------------+\n| Rust binary, no cloud calls, no Python, no ffmpeg                       |\n|                                                                         |\n|  Audio input                 Text input                  Diagnostics    |\n|  WAV/MP3/OGG/FLAC/AAC/M4A   plain text / SSML           status/support  |\n|       |                           |                           |         |\n|       v                           v                           v         |\n|  Symphonia decode            TTS preprocessing           runtime probes |\n|       |                           |                                     |\n|       +--> optional VAD           +--> voice routing                    |\n|       |    + diarization          |    Kokoro / Vosk / macOS voices     |\n|       |                           |                                     |\n|       +--> audio lang ID          +--> speech synthesis                 |\n|       |    SpeechBrain ONNX                                             |\n|       |                                                                 |\n|       +--> ASR backend                                                  |\n|            CoreML on Apple Silicon                                      |\n|            ONNX Runtime on Linux/Windows/fallback                       |\n+-----------------------------------+-------------------------------------+\n                                   |\n                                   v\n                         transcript | JSON/TOON | WAV | local diagnostics\n```\n\nCache boundary: `kesha install` and opt-in feature installs populate the local\ncache; ordinary transcription and speech commands fail fast if required assets\nare missing.\n\n## What's Inside\n\n| Model | Task | Size | Source |\n|---|---|---|---|\n| NVIDIA Parakeet TDT 0.6B v3 | Speech-to-text | ~2.5GB | [HuggingFace](https://huggingface.co/nvidia/parakeet-tdt-0.6b-v3) |\n| SpeechBrain ECAPA-TDNN | Audio language detection | ~86MB | [HuggingFace](https://huggingface.co/speechbrain/lang-id-voxlingua107-ecapa) |\n| Apple NLLanguageRecognizer | Text language detection | built-in | macOS system framework |\n| Silero VAD v5 (opt-in) | Voice activity detection | ~2.3MB | [snakers4/silero-vad](https://github.com/snakers4/silero-vad) |\n| Kokoro-82M / Vosk-TTS (opt-in) | Text-to-speech | ~990MB | [FluidAudio Kokoro](https://github.com/FluidInference/FluidAudio) on darwin-arm64 (FluidAudio cache, not Kesha-verified); ONNX Kokoro elsewhere · [Vosk-TTS](https://github.com/alphacep/vosk-tts) |\n\nAll models run through `kesha-engine` — a Rust binary using [FluidAudio](https://github.com/FluidInference/FluidAudio) (CoreML) on Apple Silicon and [ort](https://github.com/pykeio/ort) (ONNX Runtime) on other platforms.\n\nAudio decoding via [symphonia](https://github.com/pdeljanov/Symphonia) — WAV, MP3, OGG/Opus, FLAC, AAC, M4A. No ffmpeg.\n\n## Languages\n\n- **Speech-to-text (25):** Bulgarian, Croatian, Czech, Danish, Dutch, English, Estonian, Finnish, French, German, Greek, Hungarian, Italian, Latvian, Lithuanian, Maltese, Polish, Portuguese, Romanian, Russian, Slovak, Slovenian, Spanish, Swedish, Ukrainian.\n- **Audio language detection (107):** [full list](https://huggingface.co/speechbrain/lang-id-voxlingua107-ecapa).\n\n## Integrations\n\n- **OpenClaw** — give your LLM agent ears. Install & config: [docs/openclaw.md](docs/openclaw.md).\n- **Hermes Agent** — local STT/TTS through Hermes command providers. Setup: [docs/hermes.md](docs/hermes.md).\n- **Raycast** (macOS) — transcribe selected audio & speak clipboard from the launcher. Source + install: [`raycast/`](raycast/).\n\n## Programmatic API\n\n```typescript\nimport { transcribe, downloadModel } from \"@drakulavich/kesha-voice-kit/core\";\n\nawait downloadModel();                       // install engine + models\nconst text = await transcribe(\"audio.ogg\");  // transcribe\n```\n\n## Support diagnostics\n\nKesha can collect local diagnostics without downloading models or mutating cache state:\n\n```bash\nkesha doctor --json --redact\nkesha support-bundle --output kesha-support.tar.gz\nkesha support-bundle --include-logs --output kesha-support-with-logs.tar.gz\nkesha logs status\n```\n\n`support-bundle` creates a redacted `.tar.gz` archive for GitHub issues. It includes runtime, engine, cache, optional-component, Stats status, and known Kesha environment settings. It does not include audio, transcripts, model files, or the Stats database.\nDiagnostic log contents are excluded by default; pass `--include-logs` to add a\nbounded tail of Kesha's privacy-safe NDJSON diagnostic log.\n\n`kesha logs` manages local, rotated diagnostic logs for troubleshooting. Logs\ndefault to `retain-on-failure` and also support explicit `off` and `on` modes.\nThey use content-free NDJSON events: command/stage names, versions, durations,\nexit codes, and coarse audio metadata only. They must not store audio,\ntranscripts, input text, generated speech text, file names, full paths, raw\nstdout/stderr, environment variables, tokens, or URLs. See [Diagnostic logs](docs/diagnostic-logs.md).\n\n## Local Stats privacy and lifecycle\n\nKesha Stats is disabled by default. When you opt in with `kesha stats enable`,\nKesha writes a local SQLite database only on your machine:\n\n```bash\nkesha stats status\nkesha stats week\nkesha stats errors\nkesha stats export --format json   # or csv\nkesha stats retention 30           # default: 90 days\nkesha stats retention off          # keep until reset\nkesha stats reset                  # delete recorded stats rows\nkesha stats vacuum                 # compact the SQLite file\n```\n\nThe database stores content-free operational records only: command name\n(`transcribe` or `say`), timestamps, success/failure status, app version, item\ncount, anonymous stage timings, input/output artifact kind, file extension,\nsize, optional duration/sample-rate/channel counts, and sanitized error\nclass/code/message.\n\nStats never stores audio bytes, transcripts, input text, generated speech text,\nfile names, full file paths, raw stdout/stderr, environment variables, model\nfiles, API tokens, or cloud identifiers. `support-bundle` reports Stats status\nonly; it never includes the Stats SQLite database.\n\nBy default, Stats prunes rows older than 90 days before writing or exporting\ndata. Use `kesha stats retention <days>` to change the TTL or `kesha stats\nretention off` to disable TTL pruning. `kesha stats reset` deletes recorded\nruns, artifacts, timings, and errors while preserving settings such as enabled\nstate and retention.\n\n## Contributing\n\nSee [CONTRIBUTING.md](CONTRIBUTING.md).\n\n## License\n\nMade with 💛🩵 and 🥤 energy under MIT License\n\nFile v1.5.0:_meta.json\n\n{\n  \"ownerId\": \"kn70xsptbaknapzrxhhsqepa4x80ynkp\",\n  \"slug\": \"kesha-voice-kit\",\n  \"version\": \"1.5.0\",\n  \"publishedAt\": 1779538752363\n}\n\nFile v1.5.0:skill-card.md\n\n## Description: <br>\nLocal multilingual voice toolkit for speech-to-text, text-to-speech, and language detection that runs offline on Apple Silicon, Linux, and Windows. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[drakulavich](https://clawhub.ai/user/drakulavich) <br>\n\n### License/Terms of Use: <br>\nMIT <br>\n\n\n## Use Case: <br>\nDevelopers and agent operators use this skill to transcribe voice notes, synthesize voice replies, detect language, and route local audio workflows through a CLI or OpenClaw configuration. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Voice notes may contain sensitive speech, and the documented OpenClaw setup can echo recognized speech into chat history and agent context. <br>\nMitigation: Review OpenClaw transcript echo settings before use and disable or change transcript echoing when handling sensitive audio. <br>\nRisk: Installation uses a global Bun package and downloads local engine and model assets. <br>\nMitigation: Install only in environments where global Bun packages and model downloads are approved, and verify the installed `kesha` binary before routing agent audio through it. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/drakulavich/kesha-voice-kit) <br>\n- [Source repository](https://github.com/drakulavich/kesha-voice-kit) <br>\n- [npm package](https://www.npmjs.com/package/@drakulavich/kesha-voice-kit) <br>\n- [Release notes](https://github.com/drakulavich/kesha-voice-kit/releases) <br>\n- [OpenClaw](https://github.com/openclaw/openclaw) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [Text, Markdown, Shell commands, Configuration, Guidance] <br>\n**Output Format:** [Markdown with inline shell commands, JSON examples, and configuration snippets] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Can guide agents to produce transcripts, timestamped JSON, OGG/Opus voice-note files, WAV audio, and OpenClaw routing configuration.] <br>\n\n## Skill Version(s): <br>\n1.5.0 (source: ClawHub release metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.5.0:LICENSE\n\nMIT License\n\nCopyright (c) 2026 drakulavich\n\nPermission is hereby granted, free of charge, to any person obtaining a copy\nof this software and associated documentation files (the \"Software\"), to deal\nin the Software without restriction, including without limitation the rights\nto use, copy, modify, merge, publish, distribute, sublicense, and/or sell\ncopies of the Software, and to permit persons to whom the Software is\nfurnished to do so, subject to the following conditions:\n\nThe above copyright notice and this permission notice shall be included in all\ncopies or substantial portions of the Software.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR\nIMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,\nFITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE\nAUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER\nLIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,\nOUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE\nSOFTWARE.\n\nArchive v1.4.4: 3 files, 5347 bytes\n\nFiles: README.md (5820b), SKILL.md (4738b), _meta.json (134b)\n\nFile v1.4.4:SKILL.md\n\n---\nname: kesha-voice-kit\ndescription: Local multilingual voice toolkit — speech-to-text (STT), text-to-speech (TTS), and language detection. Runs entirely offline on Apple Silicon, Linux, and Windows. No API keys, no cloud. NVIDIA Parakeet TDT for STT across 25 European languages, Kokoro-82M + Piper VITS for TTS, plus macOS AVSpeechSynthesizer for ~180 system voices with zero install.\nemoji: 🎙️\n\nrequires:\n  bins: [kesha]\n\ninstall:\n  - kind: npm\n    packages: [@drakulavich/kesha-voice-kit]\n    flags: [--global]\n  - kind: bash\n    cmd: kesha install\n---\n\n# kesha-voice-kit\n\nLocal voice toolkit: transcribe voice messages to text, synthesize speech, detect language of audio or text. Fully offline after `kesha install`. No API keys, no per-minute billing.\n\n**Trigger keywords for when to use this skill:** voice message, voice memo, .ogg, .wav, .mp3, audio file, transcribe, transcription, speech-to-text, STT, text-to-speech, TTS, synthesize speech, say, multilingual voice, multilingual ASR, language detection, offline voice, privacy, Apple Silicon, CoreML.\n\n## When to use\n\n- **Voice memo arrived** (Telegram, WhatsApp, Slack, Signal .ogg/.opus/.m4a): transcribe with `kesha --json <path>` and branch on the detected language.\n- **Need to reply with audio**: synthesize with `kesha say \"<text>\" > reply.wav`. Auto-routes by detected language (Kokoro-82M for English, Piper for Russian). For other languages and ~180 more voices use `--voice macos-*` on macOS (zero model download).\n- **Need to detect what language a file is in** before choosing a pipeline: `kesha --json audio.ogg` returns both audio-based and text-based language detection with confidence scores.\n\n## STT: transcribe audio\n\n```bash\n# JSON output with language detection (recommended for automation)\nkesha --json voice.ogg\n```\n\n```json\n[{\n  \"file\": \"voice.ogg\",\n  \"text\": \"Привет, как дела?\",\n  \"lang\": \"ru\",\n  \"audioLanguage\": { \"code\": \"ru\", \"confidence\": 0.98 },\n  \"textLanguage\": { \"code\": \"ru\", \"confidence\": 0.99 }\n}]\n```\n\nUse `lang` (or the more detailed `audioLanguage`/`textLanguage`) to decide how to respond.\n\n**Formats:** .ogg, .opus, .mp3, .m4a, .wav, .flac, .webm — decoded via symphonia, no ffmpeg required.\n\n**Other output modes:**\n- `kesha audio.ogg` — plain transcript on stdout\n- `kesha --format transcript audio.ogg` — transcript + `[lang: ru, confidence: 0.99]` footer\n- `kesha --verbose audio.ogg` — human-readable with language info\n- `kesha --lang en audio.ogg` — warn if detected language differs (useful sanity check)\n\n## TTS: synthesize speech\n\n```bash\nkesha say \"Hello, world\" > hello.wav               # auto-routes en → Kokoro-82M\nkesha say \"Привет, мир\" > privet.wav              # auto-routes ru → Piper\nkesha say --voice macos-de-DE \"Guten Tag\" > de.wav # any macOS system voice — German, French, Italian, ...\nkesha say --list-voices                            # Kokoro + Piper + ~180 macos-* voices\n```\n\nOutput: WAV mono float32. `--out <path>` writes to a file instead of stdout.\n\n## Language detection standalone\n\n`kesha --json audio.ogg` includes both audio-based (`audioLanguage`) and text-based (`textLanguage`) detection. Use audio detection to identify the language before running language-specific logic.\n\n## Install\n\n```bash\nbun add --global @drakulavich/kesha-voice-kit    # or: npm i -g @drakulavich/kesha-voice-kit\nkesha install                                    # downloads engine (~350 MB)\nkesha install --tts                              # adds Kokoro + Piper RU + ONNX G2P (~490 MB more, for TTS)\n```\n\nNo system deps — G2P runs as ONNX alongside Kokoro/Piper. `macos-*` voices need no install either — they use voices already on the Mac.\n\n## Supported languages\n\n**Speech-to-text (25):** Bulgarian, Croatian, Czech, Danish, Dutch, English, Estonian, Finnish, French, German, Greek, Hungarian, Italian, Latvian, Lithuanian, Maltese, Polish, Portuguese, Romanian, Russian, Slovak, Slovenian, Spanish, Swedish, Ukrainian.\n\n**Text-to-speech:** English (Kokoro-82M, ~70 voices), Russian (Piper `ru-denis`), plus any macOS system voice via `--voice macos-*`.\n\n## Performance\n\n- ASR: ~19× faster than OpenAI Whisper on Apple Silicon (CoreML via FluidAudio), ~2.5× on CPU (ONNX via `ort`).\n- TTS: sub-second latency for short utterances on Apple Silicon.\n\n## Why local\n\nNo API keys to manage. No per-minute billing. Voice data never leaves the machine — important for regulated industries, personal messaging, and anything that shouldn't be in a third-party log.\n\n## Links\n\n- Source: https://github.com/drakulavich/kesha-voice-kit\n- npm: https://www.npmjs.com/package/@drakulavich/kesha-voice-kit\n- Releases: https://github.com/drakulavich/kesha-voice-kit/releases\n\nFile v1.4.4:README.md\n\n<p align=\"center\">\n  <img src=\"assets/logo.png\" alt=\"Kesha Voice Kit\" width=\"200\">\n</p>\n\n<h1 align=\"center\">Kesha Voice Kit</h1>\n\n<p align=\"center\">\n  <a href=\"https://github.com/drakulavich/kesha-voice-kit/actions/workflows/ci.yml\"><img src=\"https://github.com/drakulavich/kesha-voice-kit/actions/workflows/ci.yml/badge.svg\" alt=\"CI\"></a>\n  <a href=\"https://www.npmjs.com/package/@drakulavich/kesha-voice-kit\"><img src=\"https://img.shields.io/npm/v/@drakulavich/kesha-voice-kit\" alt=\"npm version\"></a>\n  <a href=\"https://opensource.org/licenses/MIT\"><img src=\"https://img.shields.io/badge/License-MIT-blue.svg\" alt=\"License: MIT\"></a>\n  <a href=\"https://bun.sh\"><img src=\"https://img.shields.io/badge/runtime-Bun-f9f1e1?logo=bun\" alt=\"Bun\"></a>\n</p>\n\n<p align=\"center\"><b>Open-source voice toolkit.</b> Optimized for Apple Silicon (CoreML), works on any platform (ONNX fallback).<br>A collection of small, fast, open-source audio models — packaged as CLI tools and an <a href=\"https://github.com/openclaw/openclaw\">OpenClaw</a> skill for LLM agents.</p>\n\n- **Speech-to-text** — 25 languages, ~15x faster than Whisper on Apple Silicon, ~2.5x on CPU\n- **Text-to-speech** — Kokoro (EN) + Piper (RU) + macOS system voices, SSML preview\n- **Rust engine** — single 20MB binary, no ffmpeg, no Python, no native Node addons\n- **OpenClaw-ready** — plug into your LLM agent as a voice processing skill\n\n## Quick Start\n\nRuntime: **[Bun](https://bun.sh)** >= 1.3.0.\n\n```bash\ncurl -fsSL https://bun.sh/install | bash   # skip if Bun is already installed\n\nbun install -g @drakulavich/kesha-voice-kit\nkesha install       # downloads engine + models\nkesha audio.ogg     # transcript to stdout\n```\n\nAir-gapped or behind a corporate mirror? See [docs/model-mirror.md](docs/model-mirror.md).\n\n## Speech-to-text\n\n```bash\nkesha audio.ogg                            # transcribe (plain text)\nkesha --format transcript audio.ogg        # text + language/confidence\nkesha --format json audio.ogg              # full JSON with lang fields\nkesha --toon audio.ogg                     # compact LLM-friendly TOON\nkesha --verbose audio.ogg                  # show language detection details\nkesha --lang en audio.ogg                  # warn if detected language differs\nkesha status                               # show installed backend info\n```\n\nMultiple files — headers per file, like `head`:\n\n```bash\n$ kesha freedom.ogg tahiti.ogg\n=== freedom.ogg ===\nСвободу попугаям! Свободу!\n\n=== tahiti.ogg ===\nТаити, Таити! Не были мы ни в какой Таити! Нас и тут неплохо кормят.\n```\n\nStdout: transcript. Stderr: errors. Pipe-friendly. Also available as `parakeet` command (backward-compatible alias).\n\nFor long / silence-heavy audio, use `--vad` (auto-on past 120 s). Details: [docs/vad.md](docs/vad.md).\n\n## Text-to-speech\n\nKesha speaks back via Kokoro-82M (English) and Piper (Russian) — voice auto-picks from the text's language:\n\n```bash\nkesha install --tts                      # ~490MB (Kokoro + Piper RU + ONNX G2P, opt-in)\nkesha say \"Hello, world\" > hello.wav\nkesha say \"Привет, мир\" > privet.wav     # auto-routes to ru-denis\n```\n\nmacOS system voices, SSML, voice listing, and the full voice catalogue: [docs/tts.md](docs/tts.md).\n\n## Performance\n\n> **~15x faster than Whisper** on Apple Silicon (M3 Pro), **~2.5x faster** on CPU\n\nCompared against Whisper `large-v3-turbo` — all engines auto-detect language.\n\n![Benchmark: openai-whisper vs faster-whisper vs Kesha Voice Kit](assets/benchmark.svg)\n\nSee [BENCHMARK.md](BENCHMARK.md) for the full per-file breakdown (Russian + English).\n\n## What's Inside\n\n| Model | Task | Size | Source |\n|---|---|---|---|\n| NVIDIA Parakeet TDT 0.6B v3 | Speech-to-text | ~2.5GB | [HuggingFace](https://huggingface.co/nvidia/parakeet-tdt-0.6b-v3) |\n| SpeechBrain ECAPA-TDNN | Audio language detection | ~86MB | [HuggingFace](https://huggingface.co/speechbrain/lang-id-voxlingua107-ecapa) |\n| Apple NLLanguageRecognizer | Text language detection | built-in | macOS system framework |\n| Silero VAD v5 (opt-in) | Voice activity detection | ~2.3MB | [snakers4/silero-vad](https://github.com/snakers4/silero-vad) |\n| Kokoro-82M / Piper (opt-in) | Text-to-speech | ~490MB | [Kokoro](https://huggingface.co/hexgrad/Kokoro-82M) · [Piper](https://github.com/rhasspy/piper) |\n\nAll models run through `kesha-engine` — a Rust binary using [FluidAudio](https://github.com/FluidInference/FluidAudio) (CoreML) on Apple Silicon and [ort](https://github.com/pykeio/ort) (ONNX Runtime) on other platforms.\n\nAudio decoding via [symphonia](https://github.com/pdeljanov/Symphonia) — WAV, MP3, OGG/Opus, FLAC, AAC, M4A. No ffmpeg.\n\n## Languages\n\n- **Speech-to-text (25):** Bulgarian, Croatian, Czech, Danish, Dutch, English, Estonian, Finnish, French, German, Greek, Hungarian, Italian, Latvian, Lithuanian, Maltese, Polish, Portuguese, Romanian, Russian, Slovak, Slovenian, Spanish, Swedish, Ukrainian.\n- **Audio language detection (107):** [full list](https://huggingface.co/speechbrain/lang-id-voxlingua107-ecapa).\n\n## Integrations\n\n- **OpenClaw** — give your LLM agent ears. Install & config: [docs/openclaw.md](docs/openclaw.md).\n- **Raycast** (macOS) — transcribe selected audio & speak clipboard from the launcher. Source + install: [`raycast/`](raycast/).\n\n## Programmatic API\n\n```typescript\nimport { transcribe, downloadModel } from \"@drakulavich/kesha-voice-kit/core\";\n\nawait downloadModel();                       // install engine + models\nconst text = await transcribe(\"audio.ogg\");  // transcribe\n```\n\n## Requirements\n\n- [Bun](https://bun.sh) >= 1.3\n- macOS arm64, Linux x64, or Windows x64\n\n## Contributing\n\nSee [CONTRIBUTING.md](CONTRIBUTING.md).\n\n## License\n\nMade with 💛🩵 and 🥤 energy under MIT License\n\nFile v1.4.4:_meta.json\n\n{\n  \"ownerId\": \"kn70xsptbaknapzrxhhsqepa4x80ynkp\",\n  \"slug\": \"kesha-voice-kit\",\n  \"version\": \"1.4.4\",\n  \"publishedAt\": 1777192467688\n}\n\nArchive v1.4.3: 3 files, 5347 bytes\n\nFiles: README.md (5820b), SKILL.md (4738b), _meta.json (134b)\n\nFile v1.4.3:SKILL.md\n\n---\nname: kesha-voice-kit\ndescription: Local multilingual voice toolkit — speech-to-text (STT), text-to-speech (TTS), and language detection. Runs entirely offline on Apple Silicon, Linux, and Windows. No API keys, no cloud. NVIDIA Parakeet TDT for STT across 25 European languages, Kokoro-82M + Piper VITS for TTS, plus macOS AVSpeechSynthesizer for ~180 system voices with zero install.\nemoji: 🎙️\n\nrequires:\n  bins: [kesha]\n\ninstall:\n  - kind: npm\n    packages: [@drakulavich/kesha-voice-kit]\n    flags: [--global]\n  - kind: bash\n    cmd: kesha install\n---\n\n# kesha-voice-kit\n\nLocal voice toolkit: transcribe voice messages to text, synthesize speech, detect language of audio or text. Fully offline after `kesha install`. No API keys, no per-minute billing.\n\n**Trigger keywords for when to use this skill:** voice message, voice memo, .ogg, .wav, .mp3, audio file, transcribe, transcription, speech-to-text, STT, text-to-speech, TTS, synthesize speech, say, multilingual voice, multilingual ASR, language detection, offline voice, privacy, Apple Silicon, CoreML.\n\n## When to use\n\n- **Voice memo arrived** (Telegram, WhatsApp, Slack, Signal .ogg/.opus/.m4a): transcribe with `kesha --json <path>` and branch on the detected language.\n- **Need to reply with audio**: synthesize with `kesha say \"<text>\" > reply.wav`. Auto-routes by detected language (Kokoro-82M for English, Piper for Russian). For other languages and ~180 more voices use `--voice macos-*` on macOS (zero model download).\n- **Need to detect what language a file is in** before choosing a pipeline: `kesha --json audio.ogg` returns both audio-based and text-based language detection with confidence scores.\n\n## STT: transcribe audio\n\n```bash\n# JSON output with language detection (recommended for automation)\nkesha --json voice.ogg\n```\n\n```json\n[{\n  \"file\": \"voice.ogg\",\n  \"text\": \"Привет, как дела?\",\n  \"lang\": \"ru\",\n  \"audioLanguage\": { \"code\": \"ru\", \"confidence\": 0.98 },\n  \"textLanguage\": { \"code\": \"ru\", \"confidence\": 0.99 }\n}]\n```\n\nUse `lang` (or the more detailed `audioLanguage`/`textLanguage`) to decide how to respond.\n\n**Formats:** .ogg, .opus, .mp3, .m4a, .wav, .flac, .webm — decoded via symphonia, no ffmpeg required.\n\n**Other output modes:**\n- `kesha audio.ogg` — plain transcript on stdout\n- `kesha --format transcript audio.ogg` — transcript + `[lang: ru, confidence: 0.99]` footer\n- `kesha --verbose audio.ogg` — human-readable with language info\n- `kesha --lang en audio.ogg` — warn if detected language differs (useful sanity check)\n\n## TTS: synthesize speech\n\n```bash\nkesha say \"Hello, world\" > hello.wav               # auto-routes en → Kokoro-82M\nkesha say \"Привет, мир\" > privet.wav              # auto-routes ru → Piper\nkesha say --voice macos-de-DE \"Guten Tag\" > de.wav # any macOS system voice — German, French, Italian, ...\nkesha say --list-voices                            # Kokoro + Piper + ~180 macos-* voices\n```\n\nOutput: WAV mono float32. `--out <path>` writes to a file instead of stdout.\n\n## Language detection standalone\n\n`kesha --json audio.ogg` includes both audio-based (`audioLanguage`) and text-based (`textLanguage`) detection. Use audio detection to identify the language before running language-specific logic.\n\n## Install\n\n```bash\nbun add --global @drakulavich/kesha-voice-kit    # or: npm i -g @drakulavich/kesha-voice-kit\nkesha install                                    # downloads engine (~350 MB)\nkesha install --tts                              # adds Kokoro + Piper RU + ONNX G2P (~490 MB more, for TTS)\n```\n\nNo system deps — G2P runs as ONNX alongside Kokoro/Piper. `macos-*` voices need no install either — they use voices already on the Mac.\n\n## Supported languages\n\n**Speech-to-text (25):** Bulgarian, Croatian, Czech, Danish, Dutch, English, Estonian, Finnish, French, German, Greek, Hungarian, Italian, Latvian, Lithuanian, Maltese, Polish, Portuguese, Romanian, Russian, Slovak, Slovenian, Spanish, Swedish, Ukrainian.\n\n**Text-to-speech:** English (Kokoro-82M, ~70 voices), Russian (Piper `ru-denis`), plus any macOS system voice via `--voice macos-*`.\n\n## Performance\n\n- ASR: ~19× faster than OpenAI Whisper on Apple Silicon (CoreML via FluidAudio), ~2.5× on CPU (ONNX via `ort`).\n- TTS: sub-second latency for short utterances on Apple Silicon.\n\n## Why local\n\nNo API keys to manage. No per-minute billing. Voice data never leaves the machine — important for regulated industries, personal messaging, and anything that shouldn't be in a third-party log.\n\n## Links\n\n- Source: https://github.com/drakulavich/kesha-voice-kit\n- npm: https://www.npmjs.com/package/@drakulavich/kesha-voice-kit\n- Releases: https://github.com/drakulavich/kesha-voice-kit/releases\n\nFile v1.4.3:README.md\n\n<p align=\"center\">\n  <img src=\"assets/logo.png\" alt=\"Kesha Voice Kit\" width=\"200\">\n</p>\n\n<h1 align=\"center\">Kesha Voice Kit</h1>\n\n<p align=\"center\">\n  <a href=\"https://github.com/drakulavich/kesha-voice-kit/actions/workflows/ci.yml\"><img src=\"https://github.com/drakulavich/kesha-voice-kit/actions/workflows/ci.yml/badge.svg\" alt=\"CI\"></a>\n  <a href=\"https://www.npmjs.com/package/@drakulavich/kesha-voice-kit\"><img src=\"https://img.shields.io/npm/v/@drakulavich/kesha-voice-kit\" alt=\"npm version\"></a>\n  <a href=\"https://opensource.org/licenses/MIT\"><img src=\"https://img.shields.io/badge/License-MIT-blue.svg\" alt=\"License: MIT\"></a>\n  <a href=\"https://bun.sh\"><img src=\"https://img.shields.io/badge/runtime-Bun-f9f1e1?logo=bun\" alt=\"Bun\"></a>\n</p>\n\n<p align=\"center\"><b>Open-source voice toolkit.</b> Optimized for Apple Silicon (CoreML), works on any platform (ONNX fallback).<br>A collection of small, fast, open-source audio models — packaged as CLI tools and an <a href=\"https://github.com/openclaw/openclaw\">OpenClaw</a> skill for LLM agents.</p>\n\n- **Speech-to-text** — 25 languages, ~15x faster than Whisper on Apple Silicon, ~2.5x on CPU\n- **Text-to-speech** — Kokoro (EN) + Piper (RU) + macOS system voices, SSML preview\n- **Rust engine** — single 20MB binary, no ffmpeg, no Python, no native Node addons\n- **OpenClaw-ready** — plug into your LLM agent as a voice processing skill\n\n## Quick Start\n\nRuntime: **[Bun](https://bun.sh)** >= 1.3.0.\n\n```bash\ncurl -fsSL https://bun.sh/install | bash   # skip if Bun is already installed\n\nbun install -g @drakulavich/kesha-voice-kit\nkesha install       # downloads engine + models\nkesha audio.ogg     # transcript to stdout\n```\n\nAir-gapped or behind a corporate mirror? See [docs/model-mirror.md](docs/model-mirror.md).\n\n## Speech-to-text\n\n```bash\nkesha audio.ogg                            # transcribe (plain text)\nkesha --format transcript audio.ogg        # text + language/confidence\nkesha --format json audio.ogg              # full JSON with lang fields\nkesha --toon audio.ogg                     # compact LLM-friendly TOON\nkesha --verbose audio.ogg                  # show language detection details\nkesha --lang en audio.ogg                  # warn if detected language differs\nkesha status                               # show installed backend info\n```\n\nMultiple files — headers per file, like `head`:\n\n```bash\n$ kesha freedom.ogg tahiti.ogg\n=== freedom.ogg ===\nСвободу попугаям! Свободу!\n\n=== tahiti.ogg ===\nТаити, Таити! Не были мы ни в какой Таити! Нас и тут неплохо кормят.\n```\n\nStdout: transcript. Stderr: errors. Pipe-friendly. Also available as `parakeet` command (backward-compatible alias).\n\nFor long / silence-heavy audio, use `--vad` (auto-on past 120 s). Details: [docs/vad.md](docs/vad.md).\n\n## Text-to-speech\n\nKesha speaks back via Kokoro-82M (English) and Piper (Russian) — voice auto-picks from the text's language:\n\n```bash\nkesha install --tts                      # ~490MB (Kokoro + Piper RU + ONNX G2P, opt-in)\nkesha say \"Hello, world\" > hello.wav\nkesha say \"Привет, мир\" > privet.wav     # auto-routes to ru-denis\n```\n\nmacOS system voices, SSML, voice listing, and the full voice catalogue: [docs/tts.md](docs/tts.md).\n\n## Performance\n\n> **~15x faster than Whisper** on Apple Silicon (M3 Pro), **~2.5x faster** on CPU\n\nCompared against Whisper `large-v3-turbo` — all engines auto-detect language.\n\n![Benchmark: openai-whisper vs faster-whisper vs Kesha Voice Kit](assets/benchmark.svg)\n\nSee [BENCHMARK.md](BENCHMARK.md) for the full per-file breakdown (Russian + English).\n\n## What's Inside\n\n| Model | Task | Size | Source |\n|---|---|---|---|\n| NVIDIA Parakeet TDT 0.6B v3 | Speech-to-text | ~2.5GB | [HuggingFace](https://huggingface.co/nvidia/parakeet-tdt-0.6b-v3) |\n| SpeechBrain ECAPA-TDNN | Audio language detection | ~86MB | [HuggingFace](https://huggingface.co/speechbrain/lang-id-voxlingua107-ecapa) |\n| Apple NLLanguageRecognizer | Text language detection | built-in | macOS system framework |\n| Silero VAD v5 (opt-in) | Voice activity detection | ~2.3MB | [snakers4/silero-vad](https://github.com/snakers4/silero-vad) |\n| Kokoro-82M / Piper (opt-in) | Text-to-speech | ~490MB | [Kokoro](https://huggingface.co/hexgrad/Kokoro-82M) · [Piper](https://github.com/rhasspy/piper) |\n\nAll models run through `kesha-engine` — a Rust binary using [FluidAudio](https://github.com/FluidInference/FluidAudio) (CoreML) on Apple Silicon and [ort](https://github.com/pykeio/ort) (ONNX Runtime) on other platforms.\n\nAudio decoding via [symphonia](https://github.com/pdeljanov/Symphonia) — WAV, MP3, OGG/Opus, FLAC, AAC, M4A. No ffmpeg.\n\n## Languages\n\n- **Speech-to-text (25):** Bulgarian, Croatian, Czech, Danish, Dutch, English, Estonian, Finnish, French, German, Greek, Hungarian, Italian, Latvian, Lithuanian, Maltese, Polish, Portuguese, Romanian, Russian, Slovak, Slovenian, Spanish, Swedish, Ukrainian.\n- **Audio language detection (107):** [full list](https://huggingface.co/speechbrain/lang-id-voxlingua107-ecapa).\n\n## Integrations\n\n- **OpenClaw** — give your LLM agent ears. Install & config: [docs/openclaw.md](docs/openclaw.md).\n- **Raycast** (macOS) — transcribe selected audio & speak clipboard from the launcher. Source + install: [`raycast/`](raycast/).\n\n## Programmatic API\n\n```typescript\nimport { transcribe, downloadModel } from \"@drakulavich/kesha-voice-kit/core\";\n\nawait downloadModel();                       // install engine + models\nconst text = await transcribe(\"audio.ogg\");  // transcribe\n```\n\n## Requirements\n\n- [Bun](https://bun.sh) >= 1.3\n- macOS arm64, Linux x64, or Windows x64\n\n## Contributing\n\nSee [CONTRIBUTING.md](CONTRIBUTING.md).\n\n## License\n\nMade with 💛🩵 and 🥤 energy under MIT License\n\nFile v1.4.3:_meta.json\n\n{\n  \"ownerId\": \"kn70xsptbaknapzrxhhsqepa4x80ynkp\",\n  \"slug\": \"kesha-voice-kit\",\n  \"version\": \"1.4.3\",\n  \"publishedAt\": 1777192297433\n}\n\nArchive v1.4.1: 3 files, 8106 bytes\n\nFiles: README.md (12369b), SKILL.md (4738b), _meta.json (134b)\n\nFile v1.4.1:SKILL.md\n\n---\nname: kesha-voice-kit\ndescription: Local multilingual voice toolkit — speech-to-text (STT), text-to-speech (TTS), and language detection. Runs entirely offline on Apple Silicon, Linux, and Windows. No API keys, no cloud. NVIDIA Parakeet TDT for STT across 25 European languages, Kokoro-82M + Piper VITS for TTS, plus macOS AVSpeechSynthesizer for ~180 system voices with zero install.\nemoji: 🎙️\n\nrequires:\n  bins: [kesha]\n\ninstall:\n  - kind: npm\n    packages: [@drakulavich/kesha-voice-kit]\n    flags: [--global]\n  - kind: bash\n    cmd: kesha install\n---\n\n# kesha-voice-kit\n\nLocal voice toolkit: transcribe voice messages to text, synthesize speech, detect language of audio or text. Fully offline after `kesha install`. No API keys, no per-minute billing.\n\n**Trigger keywords for when to use this skill:** voice message, voice memo, .ogg, .wav, .mp3, audio file, transcribe, transcription, speech-to-text, STT, text-to-speech, TTS, synthesize speech, say, multilingual voice, multilingual ASR, language detection, offline voice, privacy, Apple Silicon, CoreML.\n\n## When to use\n\n- **Voice memo arrived** (Telegram, WhatsApp, Slack, Signal .ogg/.opus/.m4a): transcribe with `kesha --json <path>` and branch on the detected language.\n- **Need to reply with audio**: synthesize with `kesha say \"<text>\" > reply.wav`. Auto-routes by detected language (Kokoro-82M for English, Piper for Russian). For other languages and ~180 more voices use `--voice macos-*` on macOS (zero model download).\n- **Need to detect what language a file is in** before choosing a pipeline: `kesha --json audio.ogg` returns both audio-based and text-based language detection with confidence scores.\n\n## STT: transcribe audio\n\n```bash\n# JSON output with language detection (recommended for automation)\nkesha --json voice.ogg\n```\n\n```json\n[{\n  \"file\": \"voice.ogg\",\n  \"text\": \"Привет, как дела?\",\n  \"lang\": \"ru\",\n  \"audioLanguage\": { \"code\": \"ru\", \"confidence\": 0.98 },\n  \"textLanguage\": { \"code\": \"ru\", \"confidence\": 0.99 }\n}]\n```\n\nUse `lang` (or the more detailed `audioLanguage`/`textLanguage`) to decide how to respond.\n\n**Formats:** .ogg, .opus, .mp3, .m4a, .wav, .flac, .webm — decoded via symphonia, no ffmpeg required.\n\n**Other output modes:**\n- `kesha audio.ogg` — plain transcript on stdout\n- `kesha --format transcript audio.ogg` — transcript + `[lang: ru, confidence: 0.99]` footer\n- `kesha --verbose audio.ogg` — human-readable with language info\n- `kesha --lang en audio.ogg` — warn if detected language differs (useful sanity check)\n\n## TTS: synthesize speech\n\n```bash\nkesha say \"Hello, world\" > hello.wav               # auto-routes en → Kokoro-82M\nkesha say \"Привет, мир\" > privet.wav              # auto-routes ru → Piper\nkesha say --voice macos-de-DE \"Guten Tag\" > de.wav # any macOS system voice — German, French, Italian, ...\nkesha say --list-voices                            # Kokoro + Piper + ~180 macos-* voices\n```\n\nOutput: WAV mono float32. `--out <path>` writes to a file instead of stdout.\n\n## Language detection standalone\n\n`kesha --json audio.ogg` includes both audio-based (`audioLanguage`) and text-based (`textLanguage`) detection. Use audio detection to identify the language before running language-specific logic.\n\n## Install\n\n```bash\nbun add --global @drakulavich/kesha-voice-kit    # or: npm i -g @drakulavich/kesha-voice-kit\nkesha install                                    # downloads engine (~350 MB)\nkesha install --tts                              # adds Kokoro + Piper RU + ONNX G2P (~490 MB more, for TTS)\n```\n\nNo system deps — G2P runs as ONNX alongside Kokoro/Piper. `macos-*` voices need no install either — they use voices already on the Mac.\n\n## Supported languages\n\n**Speech-to-text (25):** Bulgarian, Croatian, Czech, Danish, Dutch, English, Estonian, Finnish, French, German, Greek, Hungarian, Italian, Latvian, Lithuanian, Maltese, Polish, Portuguese, Romanian, Russian, Slovak, Slovenian, Spanish, Swedish, Ukrainian.\n\n**Text-to-speech:** English (Kokoro-82M, ~70 voices), Russian (Piper `ru-denis`), plus any macOS system voice via `--voice macos-*`.\n\n## Performance\n\n- ASR: ~19× faster than OpenAI Whisper on Apple Silicon (CoreML via FluidAudio), ~2.5× on CPU (ONNX via `ort`).\n- TTS: sub-second latency for short utterances on Apple Silicon.\n\n## Why local\n\nNo API keys to manage. No per-minute billing. Voice data never leaves the machine — important for regulated industries, personal messaging, and anything that shouldn't be in a third-party log.\n\n## Links\n\n- Source: https://github.com/drakulavich/kesha-voice-kit\n- npm: https://www.npmjs.com/package/@drakulavich/kesha-voice-kit\n- Releases: https://github.com/drakulavich/kesha-voice-kit/releases\n\nFile v1.4.1:README.md\n\n<p align=\"center\">\n  <img src=\"assets/logo.png\" alt=\"Kesha Voice Kit\" width=\"200\">\n</p>\n\n<h1 align=\"center\">Kesha Voice Kit</h1>\n\n<p align=\"center\">\n  <a href=\"https://github.com/drakulavich/kesha-voice-kit/actions/workflows/ci.yml\"><img src=\"https://github.com/drakulavich/kesha-voice-kit/actions/workflows/ci.yml/badge.svg\" alt=\"CI\"></a>\n  <a href=\"https://www.npmjs.com/package/@drakulavich/kesha-voice-kit\"><img src=\"https://img.shields.io/npm/v/@drakulavich/kesha-voice-kit\" alt=\"npm version\"></a>\n  <a href=\"https://opensource.org/licenses/MIT\"><img src=\"https://img.shields.io/badge/License-MIT-blue.svg\" alt=\"License: MIT\"></a>\n  <a href=\"https://bun.sh\"><img src=\"https://img.shields.io/badge/runtime-Bun-f9f1e1?logo=bun\" alt=\"Bun\"></a>\n</p>\n\n<p align=\"center\"><b>Open-source voice toolkit.</b> Optimized for Apple Silicon (CoreML), works on any platform (ONNX fallback).<br>A collection of small, fast, open-source audio models — packaged as CLI tools and an <a href=\"https://github.com/openclaw/openclaw\">OpenClaw</a> skill for LLM agents.</p>\n\n- **Speech-to-text** — 25 languages, ~15x faster than Whisper on Apple Silicon, ~2.5x on CPU\n- **Language detection** — 107 languages from audio, text language via NLLanguageRecognizer\n- **Rust engine** — single 20MB binary, no ffmpeg, no Python, no native Node addons\n- **OpenClaw-ready** — plug into your LLM agent as a voice processing skill\n\n## Quick Start\n\nRuntime: **[Bun](https://bun.sh)** >= 1.3.0.\n\n```bash\ncurl -fsSL https://bun.sh/install | bash   # skip if Bun is already installed\n\nbun install -g @drakulavich/kesha-voice-kit\nkesha install       # downloads engine + models\nkesha audio.ogg     # transcript to stdout\n```\n\n### Air-gapped / corporate mirrors\n\nSet `KESHA_MODEL_MIRROR` to redirect all HuggingFace model downloads to an internal mirror ([#121](https://github.com/drakulavich/kesha-voice-kit/issues/121)). The HF path hierarchy is preserved, so any HTTP-readable mirror populated with `wget --mirror` or `rsync` works:\n\n```bash\nexport KESHA_MODEL_MIRROR=https://models.corp.internal/kesha\nkesha install        # ASR + lang-id + TTS models fetch from your mirror\nkesha status         # confirms the active Mirror URL\n```\n\nUnset / empty falls back to `huggingface.co` with no regression. The engine binary itself still comes from GitHub Releases — this env var only redirects model downloads.\n\n## OpenClaw Integration\n\nKesha Voice Kit ships as a plugin for [OpenClaw](https://github.com/openclaw/openclaw) — give your LLM agent ears. No API keys, everything runs locally on your machine.\n\n```bash\nbun add -g @drakulavich/kesha-voice-kit && kesha install\nopenclaw plugins install @drakulavich/kesha-voice-kit\nopenclaw config set tools.media.audio.models \\\n  '[{\"type\":\"cli\",\"command\":\"kesha\",\"args\":[\"--format\",\"transcript\",\"{{MediaPath}}\"],\"timeoutSeconds\":15}]'\n```\n\n> If audio transcription is not already enabled: `openclaw config set tools.media.audio.enabled true`\n\nYour agent receives a voice message in Telegram/WhatsApp/Slack, Kesha transcribes it locally, and the agent sees enriched context:\n\n```\nТаити, Таити! Не были мы ни в какой Таити! Нас и тут неплохо кормят.\n[lang: ru, confidence: 1.00]\n```\n\nManage the plugin with `openclaw plugins list`, `openclaw plugins disable kesha-voice-kit`, or `openclaw plugins uninstall kesha-voice-kit`.\n\n## Raycast Extension (macOS)\n\nTranscribe audio and speak clipboard text from Raycast's launcher without opening a terminal. Two commands:\n\n- **Transcribe Selected Audio** — pick an audio file in Finder, hit the command, transcript lands on your clipboard.\n- **Speak Clipboard** — synthesize whatever's on your clipboard and play it through the default output.\n\nSource + install instructions: [`raycast/`](raycast/). Tracked in [#145](https://github.com/drakulavich/kesha-voice-kit/issues/145); not yet on the Raycast Store — install via `ray develop` from the subdirectory.\n\n## CLI Tools\n\n```bash\nkesha install                              # download engine and models\nkesha audio.ogg                            # transcribe (plain text)\nkesha --format transcript audio.ogg        # text + language/confidence\nkesha --format json audio.ogg              # full JSON with lang fields\nkesha --json audio.ogg                     # alias for --format json\nkesha --toon audio.ogg                     # compact LLM-friendly TOON (same data as --json)\nkesha --verbose audio.ogg                  # show language detection details\nkesha --lang en audio.ogg                  # warn if detected language differs\nkesha --vad lecture.m4a                    # segment with Silero VAD first (long/silence-heavy audio)\nkesha status                               # show installed backend info\n```\n\nMultiple files — headers per file, like `head`:\n\n```bash\n$ kesha freedom.ogg tahiti.ogg\n=== freedom.ogg ===\nСвободу попугаям! Свободу!\n\n=== tahiti.ogg ===\nТаити, Таити! Не были мы ни в какой Таити! Нас и тут неплохо кормят.\n```\n\nStdout: transcript. Stderr: errors. Pipe-friendly. Also available as `parakeet` command (backward-compatible alias).\n\n### Long / silence-heavy audio: `--vad`\n\nFor meetings, lectures, and podcasts, enable Silero VAD so Parakeet only sees the speech bits. Segment boundaries land at natural speech starts/ends instead of arbitrary cuts, and long silences are skipped entirely.\n\n```bash\nkesha install --vad                   # one-time, ~2.3MB\nkesha lecture.m4a                     # auto-on when audio ≥ 120s and VAD installed\nkesha --vad short-clip.ogg            # force VAD on any input\nkesha --no-vad meeting.m4a            # force VAD off even on long audio\n```\n\nAuto-triggers at 120 s so voice messages (< 30 s of near-pure speech) stay on the fast path. If you have long audio without VAD installed, Kesha prints a one-time stderr hint. Defaults: threshold 0.5, min-speech 250 ms, min-silence 100 ms, 30 ms edge padding. See issues #128 (base) and #187 (auto-trigger).\n\n## Text-to-Speech\n\nKesha speaks back via Kokoro-82M (English) and Piper (Russian). Voice is auto-picked from the input text's language — `en` routes to Kokoro, `ru` to Piper. Pass `--voice` to override.\n\n```bash\nkesha install --tts                 # ~490MB (Kokoro + Piper RU + ONNX G2P, opt-in)\nkesha say \"Hello, world\" > hello.wav\nkesha say \"Привет, мир\" > privet.wav    # auto-routes to ru-denis\necho \"long text\" | kesha say > reply.wav\nkesha say --out reply.wav \"text\"\nkesha say --voice en-af_heart \"text\"    # explicit voice overrides auto-routing\nkesha say --list-voices\n```\n\nOutput format: WAV mono float32 (24 kHz for Kokoro, 22.05 kHz for Piper). OGG/Opus and MP3 are tracked in follow-up issues. Grapheme-to-phoneme runs entirely through ONNX (CharsiuG2P ByT5-tiny, [#123](https://github.com/drakulavich/kesha-voice-kit/issues/123)) — no `espeak-ng` system dep.\n\n**Supported voices:**\n- English: `en-af_heart` (default), plus any Kokoro voice you download into `~/.cache/kesha/models/kokoro-82m/voices/`\n- Russian: `ru-denis` (default). More speakers (dmitri, irina, ruslan) are ready to drop in once needed.\n- macOS system voices: `macos-<identifier-or-language>` routes to `AVSpeechSynthesizer`. Zero install, any of the 180+ voices already on your Mac.\n\n### macOS system voices\n\n`kesha say --voice macos-*` routes through `AVSpeechSynthesizer` on macOS, so you get voice synthesis for free — no 490 MB TTS bundle. The sidecar binary ships alongside `kesha-engine` on darwin-arm64 releases (#141); `kesha install` places both in `~/.cache/kesha/bin/`.\n\n```bash\nkesha say --list-voices | grep ^macos-                                       # discover installed voices\nkesha say --voice macos-com.apple.voice.compact.en-US.Samantha \"Hello\" > out.wav\nkesha say --voice macos-ru-RU \"Привет, мир\" > hello-ru.wav                   # language-code fallback\n```\n\nVoice id format: `macos-<id>` where `<id>` is either a full Apple identifier (`com.apple.voice.compact.en-US.Samantha`) or a language code (`en-US`, `ru-RU`) — the Swift helper tries the identifier first and falls back to the language. Output is mono float32 @ 22050 Hz, structurally identical to Piper.\n\nQuality tradeoff is honest: macOS system voices are notification-grade. Use them when you want zero-install TTS on macOS; keep Kokoro/Piper for anything that needs to sound good.\n\n### SSML (preview)\n\n`kesha say --ssml` accepts [SSML](https://www.w3.org/TR/speech-synthesis11/) for pauses and text-structuring. v1 is deliberately small:\n\n```bash\nkesha say --ssml '<speak>Hello <break time=\"500ms\"/> world.</speak>'\nkesha say --ssml --voice ru-denis '<speak>Привет <break time=\"1s\"/> мир.</speak>'\n```\n\n| Tag | Status |\n|---|---|\n| `<speak>` | ✅ required root |\n| `<break time=\"Nms\"\\|\"Ns\"\\|default>` | ✅ inserts silence of the given duration |\n| plain text inside `<speak>` | ✅ synthesized via the selected engine |\n| `<emphasis>`, `<prosody>`, `<phoneme>`, `<say-as>` | ⚠️ stripped with a stderr warning (contained text still synthesized); tracked in [#122](https://github.com/drakulavich/kesha-voice-kit/issues/122) |\n| `<!DOCTYPE>` | ❌ rejected (hardening against XXE) |\n\nSSML is opt-in via the explicit `--ssml` flag — inputs that happen to contain `<angle brackets>` aren't misinterpreted as SSML.\n\n## What's Inside\n\nKesha Voice Kit bundles open-source models optimized for on-device inference:\n\n| Model | Task | Size | Source |\n|---|---|---|---|\n| NVIDIA Parakeet TDT 0.6B v3 | Speech-to-text | ~2.5GB | [HuggingFace](https://huggingface.co/nvidia/parakeet-tdt-0.6b-v3) |\n| SpeechBrain ECAPA-TDNN | Audio language detection | ~86MB | [HuggingFace](https://huggingface.co/speechbrain/lang-id-voxlingua107-ecapa) |\n| Apple NLLanguageRecognizer | Text language detection | built-in | macOS system framework |\n| Silero VAD v5 (opt-in) | Voice activity detection | ~2.3MB | [snakers4/silero-vad](https://github.com/snakers4/silero-vad) |\n\nAll models run through `kesha-engine` — a Rust binary using [FluidAudio](https://github.com/FluidInference/FluidAudio) (CoreML) on Apple Silicon and [ort](https://github.com/pykeio/ort) (ONNX Runtime) on other platforms.\n\n## Performance\n\n> **~15x faster than Whisper** on Apple Silicon (M3 Pro), **~2.5x faster** on CPU\n\nCompared against Whisper `large-v3-turbo` — all engines auto-detect language.\n\n![Benchmark: openai-whisper vs faster-whisper vs Kesha Voice Kit](assets/benchmark.svg)\n\n<details>\n<summary>Full results with per-file breakdown</summary>\n\nSee [BENCHMARK.md](BENCHMARK.md) — includes Russian (real voice messages) and English transcription results with all four engines.\n\n</details>\n\n## Supported Audio Formats\n\nBuilt-in audio decoding via [symphonia](https://github.com/pdeljanov/Symphonia) — no ffmpeg required:\n\n| Format | Extension |\n|---|---|\n| WAV | `.wav` |\n| MP3 | `.mp3` |\n| OGG Vorbis/Opus | `.ogg`, `.opus` |\n| FLAC | `.flac` |\n| AAC / M4A | `.aac`, `.m4a` |\n\n## Supported Languages\n\n**Speech-to-text (25):** :bulgaria: Bulgarian, :croatia: Croatian, :czech_republic: Czech, :denmark: Danish, :netherlands: Dutch, :gb: English, :estonia: Estonian, :finland: Finnish, :fr: French, :de: German, :greece: Greek, :hungary: Hungarian, :it: Italian, :latvia: Latvian, :lithuania: Lithuanian, :malta: Maltese, :poland: Polish, :portugal: Portuguese, :romania: Romanian, :ru: Russian, :slovakia: Slovak, :slovenia: Slovenian, :es: Spanish, :sweden: Swedish, :ukraine: Ukrainian\n\n**Audio language detection (107):** Full list at [speechbrain/lang-id-voxlingua107-ecapa](https://huggingface.co/speechbrain/lang-id-voxlingua107-ecapa)\n\n## Architecture\n\n```\nkesha audio.ogg\n  → kesha-engine (Rust binary)\n    ├── Apple Silicon? → FluidAudio (CoreML / Neural Engine)\n    └── Other?        → ort (ONNX Runtime / CPU)\n  → transcript to stdout\n```\n\n## Programmatic API\n\n```typescript\nimport { transcribe, downloadModel } from \"@drakulavich/kesha-voice-kit/core\";\n\nawait downloadModel();                    // install engine + models\nconst text = await transcribe(\"audio.ogg\"); // transcribe\n```\n\n## Requirements\n\n- [Bun](https://bun.sh) >= 1.3\n- macOS arm64, Linux x64, or Windows x64\n\n## Contributing\n\nSee [CONTRIBUTING.md](CONTRIBUTING.md).\n\n## License\n\nMade with 💛🩵 and 🥤 energy under MIT License\n\nFile v1.4.1:_meta.json\n\n{\n  \"ownerId\": \"kn70xsptbaknapzrxhhsqepa4x80ynkp\",\n  \"slug\": \"kesha-voice-kit\",\n  \"version\": \"1.4.1\",\n  \"publishedAt\": 1776930025541\n}\n\nArchive v1.3.2: 3 files, 7132 bytes\n\nFiles: README.md (10063b), SKILL.md (4820b), _meta.json (134b)\n\nFile v1.3.2:SKILL.md\n\n---\nname: kesha-voice-kit\ndescription: Local multilingual voice toolkit — speech-to-text (STT), text-to-speech (TTS), and language detection. Runs entirely offline on Apple Silicon, Linux, and Windows. No API keys, no cloud. NVIDIA Parakeet TDT for STT across 25 European languages, Kokoro-82M + Piper VITS for TTS, plus macOS AVSpeechSynthesizer for ~180 system voices with zero install.\nemoji: 🎙️\n\nrequires:\n  bins: [kesha]\n\ninstall:\n  - kind: npm\n    packages: [@drakulavich/kesha-voice-kit]\n    flags: [--global]\n  - kind: bash\n    cmd: kesha install\n---\n\n# kesha-voice-kit\n\nLocal voice toolkit: transcribe voice messages to text, synthesize speech, detect language of audio or text. Fully offline after `kesha install`. No API keys, no per-minute billing.\n\n**Trigger keywords for when to use this skill:** voice message, voice memo, .ogg, .wav, .mp3, audio file, transcribe, transcription, speech-to-text, STT, text-to-speech, TTS, synthesize speech, say, multilingual voice, multilingual ASR, language detection, offline voice, privacy, Apple Silicon, CoreML.\n\n## When to use\n\n- **Voice memo arrived** (Telegram, WhatsApp, Slack, Signal .ogg/.opus/.m4a): transcribe with `kesha --json <path>` and branch on the detected language.\n- **Need to reply with audio**: synthesize with `kesha say \"<text>\" > reply.wav`. Auto-routes by detected language (Kokoro-82M for English, Piper for Russian). For other languages and ~180 more voices use `--voice macos-*` on macOS (zero model download).\n- **Need to detect what language a file is in** before choosing a pipeline: `kesha --json audio.ogg` returns both audio-based and text-based language detection with confidence scores.\n\n## STT: transcribe audio\n\n```bash\n# JSON output with language detection (recommended for automation)\nkesha --json voice.ogg\n```\n\n```json\n[{\n  \"file\": \"voice.ogg\",\n  \"text\": \"Привет, как дела?\",\n  \"lang\": \"ru\",\n  \"audioLanguage\": { \"code\": \"ru\", \"confidence\": 0.98 },\n  \"textLanguage\": { \"code\": \"ru\", \"confidence\": 0.99 }\n}]\n```\n\nUse `lang` (or the more detailed `audioLanguage`/`textLanguage`) to decide how to respond.\n\n**Formats:** .ogg, .opus, .mp3, .m4a, .wav, .flac, .webm — decoded via symphonia, no ffmpeg required.\n\n**Other output modes:**\n- `kesha audio.ogg` — plain transcript on stdout\n- `kesha --format transcript audio.ogg` — transcript + `[lang: ru, confidence: 0.99]` footer\n- `kesha --verbose audio.ogg` — human-readable with language info\n- `kesha --lang en audio.ogg` — warn if detected language differs (useful sanity check)\n\n## TTS: synthesize speech\n\n```bash\nkesha say \"Hello, world\" > hello.wav               # auto-routes en → Kokoro-82M\nkesha say \"Привет, мир\" > privet.wav              # auto-routes ru → Piper\nkesha say --voice macos-de-DE \"Guten Tag\" > de.wav # any macOS system voice — German, French, Italian, ...\nkesha say --list-voices                            # Kokoro + Piper + ~180 macos-* voices\n```\n\nOutput: WAV mono float32. `--out <path>` writes to a file instead of stdout.\n\n## Language detection standalone\n\n`kesha --json audio.ogg` includes both audio-based (`audioLanguage`) and text-based (`textLanguage`) detection. Use audio detection to identify the language before running language-specific logic.\n\n## Install\n\n```bash\nbun add --global @drakulavich/kesha-voice-kit    # or: npm i -g @drakulavich/kesha-voice-kit\nkesha install                                    # downloads engine (~350 MB)\nkesha install --tts                              # adds Kokoro + Piper RU (~390 MB more, for TTS)\n```\n\nOne-time runtime prereq for TTS on each platform:\n- macOS: `brew install espeak-ng`\n- Linux: `sudo apt install espeak-ng`\n- Windows: `choco install espeak-ng`\n\n`macos-*` voices need no install — they use voices already on the Mac.\n\n## Supported languages\n\n**Speech-to-text (25):** Bulgarian, Croatian, Czech, Danish, Dutch, English, Estonian, Finnish, French, German, Greek, Hungarian, Italian, Latvian, Lithuanian, Maltese, Polish, Portuguese, Romanian, Russian, Slovak, Slovenian, Spanish, Swedish, Ukrainian.\n\n**Text-to-speech:** English (Kokoro-82M, ~70 voices), Russian (Piper `ru-denis`), plus any macOS system voice via `--voice macos-*`.\n\n## Performance\n\n- ASR: ~19× faster than OpenAI Whisper on Apple Silicon (CoreML via FluidAudio), ~2.5× on CPU (ONNX via `ort`).\n- TTS: sub-second latency for short utterances on Apple Silicon.\n\n## Why local\n\nNo API keys to manage. No per-minute billing. Voice data never leaves the machine — important for regulated industries, personal messaging, and anything that shouldn't be in a third-party log.\n\n## Links\n\n- Source: https://github.com/drakulavich/kesha-voice-kit\n- npm: https://www.npmjs.com/package/@drakulavich/kesha-voice-kit\n- Releases: https://github.com/drakulavich/kesha-voice-kit/releases\n\nFile v1.3.2:README.md\n\n<p align=\"center\">\n  <img src=\"assets/logo.png\" alt=\"Kesha Voice Kit\" width=\"200\">\n</p>\n\n<h1 align=\"center\">Kesha Voice Kit</h1>\n\n<p align=\"center\">\n  <a href=\"https://github.com/drakulavich/kesha-voice-kit/actions/workflows/ci.yml\"><img src=\"https://github.com/drakulavich/kesha-voice-kit/actions/workflows/ci.yml/badge.svg\" alt=\"CI\"></a>\n  <a href=\"https://www.npmjs.com/package/@drakulavich/kesha-voice-kit\"><img src=\"https://img.shields.io/npm/v/@drakulavich/kesha-voice-kit\" alt=\"npm version\"></a>\n  <a href=\"https://opensource.org/licenses/MIT\"><img src=\"https://img.shields.io/badge/License-MIT-blue.svg\" alt=\"License: MIT\"></a>\n  <a href=\"https://bun.sh\"><img src=\"https://img.shields.io/badge/runtime-Bun-f9f1e1?logo=bun\" alt=\"Bun\"></a>\n</p>\n\n<p align=\"center\"><b>Open-source voice toolkit.</b> Optimized for Apple Silicon (CoreML), works on any platform (ONNX fallback).<br>A collection of small, fast, open-source audio models — packaged as CLI tools and an <a href=\"https://github.com/openclaw/openclaw\">OpenClaw</a> skill for LLM agents.</p>\n\n- **Speech-to-text** — 25 languages, ~15x faster than Whisper on Apple Silicon, ~2.5x on CPU\n- **Language detection** — 107 languages from audio, text language via NLLanguageRecognizer\n- **Rust engine** — single 20MB binary, no ffmpeg, no Python, no native Node addons\n- **OpenClaw-ready** — plug into your LLM agent as a voice processing skill\n\n## Quick Start\n\nRuntime: **[Bun](https://bun.sh)** >= 1.3.0.\n\n```bash\ncurl -fsSL https://bun.sh/install | bash   # skip if Bun is already installed\n\nbun install -g @drakulavich/kesha-voice-kit\nkesha install       # downloads engine + models\nkesha audio.ogg     # transcript to stdout\n```\n\n## OpenClaw Integration\n\nKesha Voice Kit ships as a plugin for [OpenClaw](https://github.com/openclaw/openclaw) — give your LLM agent ears. No API keys, everything runs locally on your machine.\n\n```bash\nbun add -g @drakulavich/kesha-voice-kit && kesha install\nopenclaw plugins install @drakulavich/kesha-voice-kit\nopenclaw config set tools.media.audio.models \\\n  '[{\"type\":\"cli\",\"command\":\"kesha\",\"args\":[\"--format\",\"transcript\",\"{{MediaPath}}\"],\"timeoutSeconds\":15}]'\n```\n\n> If audio transcription is not already enabled: `openclaw config set tools.media.audio.enabled true`\n\nYour agent receives a voice message in Telegram/WhatsApp/Slack, Kesha transcribes it locally, and the agent sees enriched context:\n\n```\nТаити, Таити! Не были мы ни в какой Таити! Нас и тут неплохо кормят.\n[lang: ru, confidence: 1.00]\n```\n\nManage the plugin with `openclaw plugins list`, `openclaw plugins disable kesha-voice-kit`, or `openclaw plugins uninstall kesha-voice-kit`.\n\n## CLI Tools\n\n```bash\nkesha install                              # download engine and models\nkesha audio.ogg                            # transcribe (plain text)\nkesha --format transcript audio.ogg        # text + language/confidence\nkesha --format json audio.ogg              # full JSON with lang fields\nkesha --json audio.ogg                     # alias for --format json\nkesha --verbose audio.ogg                  # show language detection details\nkesha --lang en audio.ogg                  # warn if detected language differs\nkesha status                               # show installed backend info\n```\n\nMultiple files — headers per file, like `head`:\n\n```bash\n$ kesha freedom.ogg tahiti.ogg\n=== freedom.ogg ===\nСвободу попугаям! Свободу!\n\n=== tahiti.ogg ===\nТаити, Таити! Не были мы ни в какой Таити! Нас и тут неплохо кормят.\n```\n\nStdout: transcript. Stderr: errors. Pipe-friendly. Also available as `parakeet` command (backward-compatible alias).\n\n## Text-to-Speech\n\nKesha speaks back via Kokoro-82M (English) and Piper (Russian). Voice is auto-picked from the input text's language — `en` routes to Kokoro, `ru` to Piper. Pass `--voice` to override.\n\n```bash\nbrew install espeak-ng              # one-time system dep — macOS\n# Linux:   sudo apt install espeak-ng\n# Windows: choco install espeak-ng  (puts libespeak-ng.dll on PATH)\nkesha install --tts                 # ~390MB (Kokoro + Piper RU, opt-in)\nkesha say \"Hello, world\" > hello.wav\nkesha say \"Привет, мир\" > privet.wav    # auto-routes to ru-denis\necho \"long text\" | kesha say > reply.wav\nkesha say --out reply.wav \"text\"\nkesha say --voice en-af_heart \"text\"    # explicit voice overrides auto-routing\nkesha say --list-voices\n```\n\nOutput format: WAV mono float32 (24 kHz for Kokoro, 22.05 kHz for Piper). OGG/Opus and MP3 are tracked in follow-up issues. Static-linking of `espeak-ng` to remove the system dep is [#124](https://github.com/drakulavich/kesha-voice-kit/issues/124).\n\n**Supported voices:**\n- English: `en-af_heart` (default), plus any Kokoro voice you download into `~/.cache/kesha/models/kokoro-82m/voices/`\n- Russian: `ru-denis` (default). More speakers (dmitri, irina, ruslan) are ready to drop in once needed.\n- macOS system voices: `macos-<identifier-or-language>` routes to `AVSpeechSynthesizer`. Zero install, any of the 180+ voices already on your Mac.\n\n### macOS system voices\n\n`kesha say --voice macos-*` routes through `AVSpeechSynthesizer` on macOS, so you get voice synthesis for free — no 390 MB TTS bundle, no `espeak-ng` dep. The sidecar binary ships alongside `kesha-engine` on darwin-arm64 releases (#141); `kesha install` places both in `~/.cache/kesha/bin/`.\n\n```bash\nkesha say --list-voices | grep ^macos-                                       # discover installed voices\nkesha say --voice macos-com.apple.voice.compact.en-US.Samantha \"Hello\" > out.wav\nkesha say --voice macos-ru-RU \"Привет, мир\" > hello-ru.wav                   # language-code fallback\n```\n\nVoice id format: `macos-<id>` where `<id>` is either a full Apple identifier (`com.apple.voice.compact.en-US.Samantha`) or a language code (`en-US`, `ru-RU`) — the Swift helper tries the identifier first and falls back to the language. Output is mono float32 @ 22050 Hz, structurally identical to Piper.\n\nQuality tradeoff is honest: macOS system voices are notification-grade. Use them when you want zero-install TTS on macOS; keep Kokoro/Piper for anything that needs to sound good.\n\n### SSML (preview)\n\n`kesha say --ssml` accepts [SSML](https://www.w3.org/TR/speech-synthesis11/) for pauses and text-structuring. v1 is deliberately small:\n\n```bash\nkesha say --ssml '<speak>Hello <break time=\"500ms\"/> world.</speak>'\nkesha say --ssml --voice ru-denis '<speak>Привет <break time=\"1s\"/> мир.</speak>'\n```\n\n| Tag | Status |\n|---|---|\n| `<speak>` | ✅ required root |\n| `<break time=\"Nms\"\\|\"Ns\"\\|default>` | ✅ inserts silence of the given duration |\n| plain text inside `<speak>` | ✅ synthesized via the selected engine |\n| `<emphasis>`, `<prosody>`, `<phoneme>`, `<say-as>` | ⚠️ stripped with a stderr warning (contained text still synthesized); tracked in [#122](https://github.com/drakulavich/kesha-voice-kit/issues/122) |\n| `<!DOCTYPE>` | ❌ rejected (hardening against XXE) |\n\nSSML is opt-in via the explicit `--ssml` flag — inputs that happen to contain `<angle brackets>` aren't misinterpreted as SSML.\n\n## What's Inside\n\nKesha Voice Kit bundles open-source models optimized for on-device inference:\n\n| Model | Task | Size | Source |\n|---|---|---|---|\n| NVIDIA Parakeet TDT 0.6B v3 | Speech-to-text | ~2.5GB | [HuggingFace](https://huggingface.co/nvidia/parakeet-tdt-0.6b-v3) |\n| SpeechBrain ECAPA-TDNN | Audio language detection | ~86MB | [HuggingFace](https://huggingface.co/speechbrain/lang-id-voxlingua107-ecapa) |\n| Apple NLLanguageRecognizer | Text language detection | built-in | macOS system framework |\n\nAll models run through `kesha-engine` — a Rust binary using [FluidAudio](https://github.com/FluidInference/FluidAudio) (CoreML) on Apple Silicon and [ort](https://github.com/pykeio/ort) (ONNX Runtime) on other platforms.\n\n## Performance\n\n> **~15x faster than Whisper** on Apple Silicon (M3 Pro), **~2.5x faster** on CPU\n\nCompared against Whisper `large-v3-turbo` — all engines auto-detect language.\n\n![Benchmark: openai-whisper vs faster-whisper vs Kesha Voice Kit](assets/benchmark.svg)\n\n<details>\n<summary>Full results with per-file breakdown</summary>\n\nSee [BENCHMARK.md](BENCHMARK.md) — includes Russian (real voice messages) and English transcription results with all four engines.\n\n</details>\n\n## Supported Audio Formats\n\nBuilt-in audio decoding via [symphonia](https://github.com/pdeljanov/Symphonia) — no ffmpeg required:\n\n| Format | Extension |\n|---|---|\n| WAV | `.wav` |\n| MP3 | `.mp3` |\n| OGG Vorbis\n\nArchive v1.3.1: 3 files, 7140 bytes\n\nFiles: README.md (10063b), SKILL.md (4839b), _meta.json (134b)\n\nArchive v1.3.0: 97 files, 217099 bytes\n\nFiles: AGENTS.md (4063b), assets/benchmark.svg (2329b), BENCHMARK.md (11032b), bin/kesha.js (76b), CLAUDE.md (22439b), CONTRIBUTING.md (2901b), docs/superpowers/plans/2026-04-06-lib-split.md (7374b), docs/superpowers/plans/2026-04-07-coreml-backend.md (19654b), docs/superpowers/plans/2026-04-09-cli-help-citty.md (10995b), docs/superpowers/plans/2026-04-09-dx-improvements.md (21098b), docs/superpowers/plans/2026-04-09-ffmpeg-error-dx.md (5198b), docs/superpowers/plans/2026-04-09-lang-detection.md (9145b), docs/superpowers/plans/2026-04-12-coreml-lang-detection.md (34817b), docs/superpowers/plans/2026-04-14-benchmark.md (15644b), docs/superpowers/plans/2026-04-14-rust-engine.md (32985b), docs/superpowers/plans/2026-04-14-ts-migration.md (13311b), docs/superpowers/plans/2026-04-16-bidirectional-voice-m1-kokoro-en.md (66734b), docs/superpowers/specs/2026-04-06-lib-split-design.md (3895b), docs/superpowers/specs/2026-04-07-coreml-backend-design.md (6034b), docs/superpowers/specs/2026-04-07-scripts-ts-rewrite-design.md (2022b), docs/superpowers/specs/2026-04-09-cli-help-design.md (3962b), docs/superpowers/specs/2026-04-09-dx-improvements-design.md (7009b), docs/superpowers/specs/2026-04-09-ffmpeg-error-dx-design.md (1780b), docs/superpowers/specs/2026-04-09-lang-detection-design.md (2383b), docs/superpowers/specs/2026-04-12-coreml-lang-detection-design.md (6630b), docs/superpowers/specs/2026-04-14-benchmark-design.md (4237b), docs/superpowers/specs/2026-04-14-rust-engine-design.md (8046b), docs/superpowers/specs/2026-04-16-bidirectional-voice-design.md (13525b), fixtures/test-vocab.txt (52b), openclaw-plugin.cjs (2291b), openclaw.plugin.json (316b), package.json (1666b), README.md (10063b), rust/build.rs (2280b), rust/Cargo.toml (1780b), rust/ci/download-kokoro.sh (670b), rust/ci/run-cargo-test.sh (1180b), rust/ci/run-clippy.sh (384b), rust/fixtures/tts/kokoro_vocab.json (1375b), rust/src/audio.rs (8197b), rust/src/backend/fluidaudio.rs (763b), rust/src/backend/mod.rs (761b), rust/src/backend/onnx.rs (15859b), rust/src/capabilities.rs (731b), rust/src/debug.rs (2148b), rust/src/lang_id.rs (1997b), rust/src/lib.rs (194b), rust/src/main.rs (10456b), rust/src/models.rs (8761b), rust/src/text_lang.rs (1408b), rust/src/transcribe.rs (802b), rust/src/tts/avspeech.rs (7526b), rust/src/tts/g2p.rs (3864b), rust/src/tts/kokoro.rs (2522b), rust/src/tts/mod.rs (11070b), rust/src/tts/piper.rs (6108b), rust/src/tts/ssml.rs (10354b), rust/src/tts/tokenizer.rs (3105b), rust/src/tts/voices.rs (11965b), rust/src/tts/wav.rs (1766b), rust/swift/say-avspeech.swift (5105b), rust/tests/tts_e2e.rs (3892b), rust/tests/tts_smoke.rs (9320b), scripts/benchmark.ts (12130b), scripts/convert-lang-id-model.py (8191b), scripts/postinstall.cjs (710b), scripts/smoke-test.ts (3666b), SECURITY.md (619b), SKILL.md (4839b), src/cli.ts (15288b), src/engine-install.ts (6754b), src/engine.ts (3059b), src/lang-id.ts (127b), src/lib.ts (888b), src/log.ts (840b), src/models.ts (176b), src/progress.ts (2257b), src/say.ts (2820b), src/star.ts (1959b), src/status.ts (2724b)","readmeExcerpt":"Skill: kesha-voice-kit Owner: drakulavich Summary: Local multilingual voice toolkit — speech-to-text (STT), text-to-speech (TTS), speaker diarization, and language detection, over a CLI or an MCP server. Runs entirely offline on Apple Silicon, Linux, and Windows. No API keys, no cloud. NVIDIA Parakeet TDT for STT across 25 European languages, Kokoro-82M + Vosk-TTS for TTS in 9 languages, plus macOS AVSpeechSynthesize","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"bun add -g @drakulavich/kesha-voice-kit\nkesha install\nopenclaw plugins install @drakulavich/kesha-voice-kit\nopenclaw config patch --stdin <<'JSON5'\n{\n  tools: {\n    media: {\n      audio: {\n        enabled: true,\n        models: [\n          {\n            type: \"cli\",\n            command: \"kesha\",\n            args: [\"{{MediaPath}}\"],\n            timeoutSeconds: 15,\n          },\n        ],\n        echoTranscript: true,\n        echoFormat: '🦜 \"{transcript}\"',\n      },\n    },\n  },\n}\nJSON5"},{"language":"bash","snippet":"openclaw config set tools.media.audio.models \\\n  '[{\"type\":\"cli\",\"command\":\"kesha\",\"args\":[\"--json\",\"--timestamps\",\"{{MediaPath}}\"],\"timeoutSeconds\":30}]'"},{"language":"bash","snippet":"which kesha\nkesha status\nopenclaw plugins list\nopenclaw config get tools.media.audio.models\nopenclaw config get tools.media.audio.echoTranscript\nopenclaw config get tools.media.audio.echoFormat"},{"language":"bash","snippet":"openclaw config patch --stdin <<'JSON5'\n{\n  messages: {\n    tts: {\n      auto: \"always\",\n      provider: \"tts-local-cli\",\n      providers: {\n        \"tts-local-cli\": {\n          command: \"kesha\",\n          args: [\"say\", \"--format\", \"ogg-opus\", \"--out\", \"{{OutputPath}}\", \"{{Text}}\"],\n          outputFormat: \"opus\",\n          timeoutMs: 120000,\n        },\n      },\n    },\n  },\n}\nJSON5"},{"language":"json","snippet":"{ \"mcpServers\": { \"kesha\": { \"command\": \"kesha\", \"args\": [\"mcp\"] } } }"},{"language":"bash","snippet":"# JSON output with language detection (recommended for automation)\nkesha --json voice.ogg"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: kesha-voice-kit\ndescription: Local multilingual voice toolkit — speech-to-text (STT), text-to-speech (TTS), speaker diarization, and language detection, over a CLI or an MCP server. Runs entirely offline on Apple Silicon, Linux, and Windows. No API keys, no cloud. NVIDIA Parakeet TDT for STT across 25 European languages, Kokoro-82M + Vosk-TTS for TTS in 9 languages, plus macOS AVSpeechSynthesizer for ~180 system voices with zero install.\nemoji: 🎙️\n\nrequires:\n  bins: [kesha]\n\ninstall:\n  - kind: bash\n    cmd: bun add -g \"@drakulavich/kesha-voice-kit\"\n  - kind: bash\n    cmd: kesha install\n---\n\n# kesha-voice-kit\n\nLocal voice toolkit: transcribe voice messages to text, synthesize speech, detect language of audio or text. Fully offline after `kesha install`. No API keys, no per-minute billing.\n\n**Trigger keywords for when to use this skill:** voice message, voice memo, voice note, .ogg, .opus, .wav, .mp3, audio file, transcribe, transcription, speech-to-text, STT, text-to-speech, TTS, synthesize speech, say, telegram voice note, whatsapp voice note, ogg-opus, opus, multilingual voice, multilingual ASR, language detection, speaker diarization, who said what, meeting transcript, MCP server, offline voice, privacy, Apple Silicon, CoreML.\n\n## When to use\n\n- **Voice memo arrived** (Telegram, WhatsApp, Slack, Signal .ogg/.opus/.m4a): transcribe with `kesha --json <path>` and branch on the detected language.\n- **Need to send a voice note (Telegram, WhatsApp, Signal, Discord)**: synthesize directly into messenger-native OGG/Opus with `kesha say --format ogg-opus --out reply.ogg \"<text>\"`. Default is mono 24 kHz @ 32 kbps - what Telegram `sendVoice` expects. No WAV redirect and no `ffmpeg` round-trip.\n- **Need local file playback/debug output**: WAV is still available with `kesha say --out reply.wav \"<text>\"`, but do not use WAV for Telegram voice replies. Auto-routes by detected language (Kokoro-82M for English, Vosk-TTS for Russian). On darwin-arm64, English Kokoro uses FluidAudio CoreML instead of ONNX. For other languages and ~180 more voices use `--voice macos-*` on macOS (zero model download).\n- **Need to detect what language a file is in** before choosing a pipeline: `kesha --json audio.ogg` returns both audio-based and text-based language detection with confidence scores.\n- **Need to capture your own voice** for transcription or as a voice-note source: `kesha record --out clip.wav` records up to 120s (override with `--max-seconds`) of mono 16 kHz WAV from the default microphone. Pipe straight into `kesha --json clip.wav` to close the loop.\n\n## OpenClaw plugin setup\n\nInstall the plugin, then explicitly route OpenClaw audio understanding through the CLI model entry. The plugin registration makes Kesha discoverable, but real voice-message transcription uses `tools.media.audio.models` with a `type: \"cli\"` entry.\n\n```bash\nbun add -g @drakulavich/kesha-voice-kit\nkesha install\nopenclaw plugins install @drakulavich/kesha-voice-kit\nopenclaw config pat"},{"path":"README.md","content":"<p align=\"center\">\n  <img src=\"https://github.com/drakulavich/kesha-voice-kit/raw/main/docs/assets/logo.png\" alt=\"Kesha Voice Kit\" width=\"200\">\n</p>\n\n<h1 align=\"center\">Kesha Voice Kit</h1>\n\n<p align=\"center\">\n  <a href=\"https://flakiness.io/Laputa/kesha-voice-kit\"><img src=\"https://img.shields.io/endpoint?url=https%3A%2F%2Fflakiness.io%2Fapi%2Fbadge%3Finput%3D%257B%2522badgeToken%2522%253A%2522badge-2IKMRRqUxh9P3w8Ym3Szf0%2522%257D\" alt=\"Tests\"></a>\n  <a href=\"https://www.npmjs.com/package/@drakulavich/kesha-voice-kit\"><img src=\"https://img.shields.io/npm/v/@drakulavich/kesha-voice-kit\" alt=\"npm version\"></a>\n  <a href=\"https://opensource.org/licenses/MIT\"><img src=\"https://img.shields.io/badge/License-MIT-blue.svg\" alt=\"License: MIT\"></a>\n  <a href=\"https://bun.sh\"><img src=\"https://img.shields.io/badge/runtime-Bun-f9f1e1?logo=bun\" alt=\"Bun\"></a>\n</p>\n\n<p align=\"center\"><b>Give your local tools and LLM agents a voice.</b><br>Fast speech-to-text, text-to-speech, voice-activity detection, and language detection in one local-first CLI: Apple Silicon CoreML first, ONNX fallback on supported Linux/Windows builds.</p>\n\n- **Transcribe locally** — [25 languages](docs/languages.md#speech-to-text-25), up to ~19x faster than Whisper on Apple Silicon, ~2.5x on CPU\n- **Speak back** — text-to-speech in [9 languages](docs/languages.md#text-to-speech)\n- **Plug into agents** — ship voice workflows as CLI commands, an MCP server, an <a href=\"docs/openclaw.md\">OpenClaw</a> skill, or a <a href=\"docs/hermes.md\">Hermes</a> agent\n- **Small Rust engine** — single ~60MB binary, no ffmpeg, no Python, no native Node addons\n\n<p align=\"center\">\n  <img src=\"https://github.com/drakulavich/kesha-voice-kit/raw/main/demo.gif\" alt=\"kesha demo — English + Russian transcription with automatic language detection\" width=\"800\">\n</p>\n\n## Quick Start\n\nRuntime: **[Bun](https://bun.sh)** >= 1.3.0 · Platforms: macOS arm64, Linux x64, Windows x64. Linux and Windows run the ONNX engine — everything except microphone capture (`kesha record`), macOS system voices, speaker diarization, and text language detection, which need Apple frameworks.\n\n```bash\n# 1. Install Bun (skip if you have it) — Linux & macOS:\ncurl -fsSL https://bun.sh/install | bash        # or: brew install oven-sh/bun/bun\n# Windows: powershell -c \"irm bun.sh/install.ps1 | iex\"\n# if `bun --version` fails, reload PATH: exec $SHELL -l\n\n# 2. Install Kesha:\nbun add -g @drakulavich/kesha-voice-kit\nkesha --version                                 # confirms `kesha` resolved on PATH\nkesha install --plan                            # preview exact download/disk sizes first — downloads nothing\nkesha install        # ~2.5 GB on Linux/Windows; ~0.6 GB on Apple Silicon, whose CoreML\n                      # engine uses a different, smaller model set. Explicit — never automatic.\n                      # No progress bar during the model step; can take several minutes.\n                      # Prefer a guided wizard? `kesha init` walks through the "},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn70xsptbaknapzrxhhsqepa4x80ynkp\",\n  \"slug\": \"kesha-voice-kit\",\n  \"version\": \"1.6.1\",\n  \"publishedAt\": 1785948561169\n}"},{"path":"skill-card.md","content":"## Description:\n\nKesha Voice Kit lets agents transcribe audio, synthesize speech, perform speaker diarization, and detect language locally through a CLI or MCP server.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[drakulavich](https://clawhub.ai/user/drakulavich)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and agent builders use this skill to add local voice-message transcription, text-to-speech voice-note generation, language detection, and MCP or OpenClaw audio workflows without cloud speech APIs.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The install path uses mutable remote and global installers with broad local code execution authority.\n\nMitigation: Review install commands before execution, prefer package-manager or pinned installation methods, avoid piping remote installers directly into a shell, and run the tool as a normal non-admin user.\n\nRisk: Installation persists a global CLI plus model assets and can change OpenClaw audio or TTS behavior.\n\nMitigation: Preview planned downloads with installation plan commands, verify OpenClaw audio and TTS configuration after install, and keep the CLI route explicit in user configuration.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/drakulavich/skills/kesha-voice-kit)\n- [npm package](https://www.npmjs.com/package/@drakulavich/kesha-voice-kit)\n- [Bun runtime](https://bun.sh)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, JSON, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown with inline shell commands and JSON configuration examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Agent-facing guidance may direct local CLI calls that produce transcripts, timestamped JSON, language metadata, or generated audio files.]\n\n## Skill Version(s):\n\n1.6.1 (source: server release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Local multilingual voice toolkit — speech-to-text (STT), text-to-speech (TTS), speaker diarization, and language detection, over a CLI or an MCP server. Runs entirely offline on Apple Silicon, Linux, and Windows. No API keys, no cloud. NVIDIA Parakeet TDT for STT across 25 European languages, Kokoro-82M + Vosk-TTS for TTS in 9 languages, plus macOS AVSpeechSynthesizer for ~180 system voices with zero install. Skill: kesha-voice-kit Owner: drakulavich Summary: Local multilingual voice toolkit — speech-to-text (STT), text-to-speech (TTS), speaker diarization, and language detection, over a CLI or an MCP server. Runs entirely offline on Apple Silicon, Linux, and Windows. No API keys, no cloud. NVIDIA Parakeet TDT for STT across 25 European languages, Kokoro-82M + Vosk-TTS for TTS in 9 languages, plus macOS AVSpeechSynthesize","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1501,"uniquenessScore":46,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-10T06:05:55.014Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-10T06:05:55.014Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T10:47:54.413Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}