{"id":"05dbcc6f-1877-4122-bfad-7719419b98c5","entityType":"agent","slug":"clawhub-shaharsha-elevenlabs-tts","name":"Elevenlabs Tts","canonicalUrl":"https://www.xpersona.co/agent/clawhub-shaharsha-elevenlabs-tts","canonicalPath":"/agent/clawhub-shaharsha-elevenlabs-tts","generatedAt":"2026-10-10T02:45:36.584Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":null},"description":"ElevenLabs TTS (Text-to-Speech) with emotional audio tags for expressive voice synthesis. WhatsApp-compatible voice messages with Opus conversion. Supports 7... Skill: Elevenlabs Tts Owner: Shaharsha Summary: ElevenLabs TTS (Text-to-Speech) with emotional audio tags for expressive voice synthesis. WhatsApp-compatible voice messages with Opus conversion. Supports 7... Tags: ai-voice:2.1.0, audio:2.1.0, elevenlabs:2.1.0, elevenlabs-tts:1.3.2, hebrew:2.1.0, latest:2.2.0, multilingual:2.1.0, nikud:2.1.0, openclaw:1.3.2, podcast:1.2.1, singing:2.1.0, speech:2.1.0, text-to-speech:","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 4.5K downloads reported by the source. Last updated 4/15/2026.","installCommand":"clawhub skill install kn77700wny92h2kvpav2am1yjx80ewfp:elevenlabs-tts","sourceUrl":"https://clawhub.ai/Shaharsha/elevenlabs-tts","homepage":"https://clawhub.ai/Shaharsha/elevenlabs-tts","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/Shaharsha/elevenlabs-tts","kind":"source"}],"safetyScore":84,"overallRank":62,"popularityScore":67,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"ElevenLabs TTS (Text-to-Speech) with emotional audio tags for expressive voice synthesis. WhatsApp-compatible voice messages with Opus conversion. Supports 7..."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":"No protocol or capability metadata is available."},"protocols":[],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":0,"capabilityMatrix":{"rows":[],"flattenedTokens":""}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":null},"stars":null,"forks":null,"downloads":4468,"packageName":null,"latestVersion":"2.2.0","tractionLabel":"4.5K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-02-28T17:29:02.214Z","emptyReason":null},"lastUpdatedAt":"2026-04-15T00:45:39.800Z","lastCrawledAt":"2026-02-28T17:29:02.214Z","lastIndexedAt":null,"nextCrawlAt":"2026-03-01T17:29:02.215Z","lastVerifiedAt":null,"highlights":[{"version":"2.2.0","createdAt":"2026-02-14T16:49:34.137Z","changelog":"Security scan fixes","fileCount":3,"zipByteSize":8483},{"version":"2.1.0","createdAt":"2026-02-09T13:46:48.695Z","changelog":"Comprehensive Hebrew nikud guide: dagesh (B/V, K/Kh, P/F), gender suffixes, homographs, stress placement, foreign names. Clear principle: only nikud where ambiguity exists.","fileCount":4,"zipByteSize":10870},{"version":"2.0.0","createdAt":"2026-02-09T13:35:12.104Z","changelog":"Major polish: improved description for discoverability, added allowed-tools declaration, fixed stability value in troubleshooting (0.0 not 0.5), selective nikud in Hebrew example, security-clean SKILL.md with lib/audio_convert.py wrapper.","fileCount":null,"zipByteSize":null},{"version":"1.6.0","createdAt":"2026-02-09T13:33:27.653Z","changelog":"Security: moved all ffmpeg shell commands into lib/audio_convert.py wrapper script. SKILL.md no longer contains raw bash commands. Added convert and concat CLI utilities.","fileCount":null,"zipByteSize":null},{"version":"1.5.0","createdAt":"2026-02-09T13:27:37.042Z","changelog":"Fixed stability values (v3 only accepts 0.0/0.5/1.0). Added singing guide with correct format ([singing] on own line). Updated audio-tags reference with singing tips and limitations.","fileCount":null,"zipByteSize":null},{"version":"1.4.0","createdAt":"2026-02-09T12:52:07.278Z","changelog":"Added Hebrew nikud (vowel points) support for accurate pronunciation. Updated Hebrew example with full nikud. Changed config example to use placeholder voiceId.","fileCount":null,"zipByteSize":null},{"version":"1.3.2","createdAt":"2026-02-04T12:14:22.141Z","changelog":"- Updated the skill description in SKILL.md for improved clarity and focus on key ElevenLabs/OpenClaw features. - Revised keywords in tags for better search relevance. - No changes to code or functionality—documentation change only.","fileCount":null,"zipByteSize":null},{"version":"1.3.1","createdAt":"2026-02-04T12:13:02.242Z","changelog":"Version 1.3.1 - No file changes detected in this release. - Documentation and usage instructions remain the same. - No new features, bug fixes, or updates in this version.","fileCount":null,"zipByteSize":null}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install kn77700wny92h2kvpav2am1yjx80ewfp:elevenlabs-tts","setupComplexity":"low","setupSteps":["Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-shaharsha-elevenlabs-tts/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-shaharsha-elevenlabs-tts/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-shaharsha-elevenlabs-tts/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-shaharsha-elevenlabs-tts/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-shaharsha-elevenlabs-tts/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-shaharsha-elevenlabs-tts/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":[]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T02:45:36.582Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-shaharsha-elevenlabs-tts/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-shaharsha-elevenlabs-tts/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-shaharsha-elevenlabs-tts/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-shaharsha-elevenlabs-tts/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":null},"readme":"Skill: Elevenlabs Tts\n\nOwner: Shaharsha\n\nSummary: ElevenLabs TTS (Text-to-Speech) with emotional audio tags for expressive voice synthesis. WhatsApp-compatible voice messages with Opus conversion. Supports 7...\n\nTags: ai-voice:2.1.0, audio:2.1.0, elevenlabs:2.1.0, elevenlabs-tts:1.3.2, hebrew:2.1.0, latest:2.2.0, multilingual:2.1.0, nikud:2.1.0, openclaw:1.3.2, podcast:1.2.1, singing:2.1.0, speech:2.1.0, text-to-speech:2.1.0, tts:2.1.0, voice:2.1.0, whatsapp:2.1.0\n\nVersion history:\n\nv2.2.0 | 2026-02-14T16:49:34.137Z | user\n\nSecurity scan fixes\n\nv2.1.0 | 2026-02-09T13:46:48.695Z | user\n\nComprehensive Hebrew nikud guide: dagesh (B/V, K/Kh, P/F), gender suffixes, homographs, stress placement, foreign names. Clear principle: only nikud where ambiguity exists.\n\nv2.0.0 | 2026-02-09T13:35:12.104Z | user\n\nMajor polish: improved description for discoverability, added allowed-tools declaration, fixed stability value in troubleshooting (0.0 not 0.5), selective nikud in Hebrew example, security-clean SKILL.md with lib/audio_convert.py wrapper.\n\nv1.6.0 | 2026-02-09T13:33:27.653Z | user\n\nSecurity: moved all ffmpeg shell commands into lib/audio_convert.py wrapper script. SKILL.md no longer contains raw bash commands. Added convert and concat CLI utilities.\n\nv1.5.0 | 2026-02-09T13:27:37.042Z | user\n\nFixed stability values (v3 only accepts 0.0/0.5/1.0). Added singing guide with correct format ([singing] on own line). Updated audio-tags reference with singing tips and limitations.\n\nv1.4.0 | 2026-02-09T12:52:07.278Z | user\n\nAdded Hebrew nikud (vowel points) support for accurate pronunciation. Updated Hebrew example with full nikud. Changed config example to use placeholder voiceId.\n\nv1.3.2 | 2026-02-04T12:14:22.141Z | auto\n\n- Updated the skill description in SKILL.md for improved clarity and focus on key ElevenLabs/OpenClaw features.\n- Revised keywords in tags for better search relevance.\n- No changes to code or functionality—documentation change only.\n\nv1.3.1 | 2026-02-04T12:13:02.242Z | auto\n\nVersion 1.3.1\n\n- No file changes detected in this release.\n- Documentation and usage instructions remain the same.\n- No new features, bug fixes, or updates in this version.\n\nv1.2.9 | 2026-02-04T12:12:26.571Z | auto\n\nelevenlabs-tts 1.2.9\n\n- Updated SKILL.md with improved and concise description, adding tags for discoverability.\n- Enhanced feature summary to emphasize WhatsApp, multilingual support, and OpenClaw integration.\n- No code changes; documentation only.\n\nv1.2.8 | 2026-02-03T22:09:57.449Z | user\n\nAdded: WhatsApp transcribe button only works with Opus format\n\nv1.2.7 | 2026-02-03T22:07:12.207Z | user\n\nComplete WhatsApp workflow: generate→convert to Opus→send. MP3 fails on Android, Opus works everywhere.\n\nv1.2.6 | 2026-02-03T22:05:53.551Z | user\n\nAdded WhatsApp sending instructions (message tool with asVoice), audio cutoff fix (add pause at end)\n\nv1.2.5 | 2026-02-03T22:02:39.924Z | user\n\nImproved Quick Start examples with more audio tags demonstrating emotional transitions\n\nv1.3.0 | 2026-02-03T21:59:24.517Z | user\n\nMajor update: Added comprehensive best practices for natural-sounding audio tags - how many to use, where to place them, context tips, regeneration strategies, punctuation effects, and updated examples with emotional progressions\n\nv1.2.4 | 2026-02-03T21:56:25.048Z | user\n\nUpdated examples to show multiple audio tags per message\n\nv1.2.3 | 2026-02-03T18:08:09.266Z | user\n\nFixed display name\n\nv1.2.2 | 2026-02-03T18:07:40.253Z | user\n\nAdded (Text-to-Speech) to title for clarity\n\nv1.2.1 | 2026-02-03T18:03:39.803Z | user\n\nAdded TTS explanation (Text-to-Speech) in description for clarity\n\nv1.2.0 | 2026-02-03T17:54:31.286Z | user\n\nAdded: 3 language examples (EN/HE/ES), OpenClaw config guide, 5 recommended voice IDs with table, voice selection tips, how to get API key\n\nv1.1.0 | 2026-02-03T17:50:29.026Z | user\n\nMajor update: Added multi-speaker dialogue, 50+ new tags, stability modes guide, speed control, punctuation effects, fixed API limit (10K not 5K), 70+ languages support\n\nv1.0.1 | 2026-02-03T17:49:17.446Z | user\n\nAdded references/audio-tags.md\n\nv1.0.0 | 2026-02-03T17:46:32.654Z | auto\n\n- Initial release of ElevenLabs-TTS with integrated audio tag support for expressive voice synthesis.\n- Supports creation of voice messages, podcasts, audiobooks, and other spoken content with emotional expression.\n- Handles WhatsApp voice compatibility, including guidance on Opus conversion.\n- Provides instructions for segmenting and concatenating long-form audio.\n- Includes quick reference for critical audio tags and troubleshooting tips.\n\nArchive index:\n\nArchive v2.2.0: 3 files, 8483 bytes\n\nFiles: references/audio-tags.md (6750b), SKILL.md (10531b), _meta.json (133b)\n\nFile v2.2.0:SKILL.md\n\n---\nname: elevenlabs-tts\ndescription: ElevenLabs TTS - the best ElevenLabs integration for OpenClaw. ElevenLabs Text-to-Speech with emotional audio tags, ElevenLabs voice synthesis for WhatsApp, ElevenLabs multilingual support. Generate realistic AI voices using ElevenLabs API.\ntags: [elevenlabs, tts, voice, text-to-speech, audio, speech, whatsapp, multilingual, ai-voice]\nmetadata: {\"clawdbot\":{\"emoji\":\"🎙️\",\"requires\":{\"env\":[\"ELEVENLABS_API_KEY\"],\"system\":[\"ffmpeg\"]},\"primaryEnv\":\"ELEVENLABS_API_KEY\"}}\nallowed-tools: [exec, tts, message]\n---\n\n# ElevenLabs TTS (Text-to-Speech)\n\nGenerate expressive voice messages using ElevenLabs v3 with audio tags.\n\n## Prerequisites\n\n- **ElevenLabs API Key** (`ELEVENLABS_API_KEY`): Required. Get one at [elevenlabs.io](https://elevenlabs.io) → Profile → API Keys. Configure in `openclaw.json` under `messages.tts.elevenlabs.apiKey`.\n- **ffmpeg**: Required for audio format conversion (MP3 → Opus for WhatsApp compatibility). Must be installed and available on PATH.\n\n## Quick Start Examples\n\n**Storytelling (emotional journey):**\n```\n[soft] It started like any other day... [pause] But something felt different. [nervous] My hands were shaking as I opened the envelope. [gasps] I got in! [excited] I actually got in! [laughs] [happy] This changes everything!\n```\n\n**Horror/Suspense (building dread):**\n```\n[whispers] The house has been empty for years... [pause] At least, that's what they told me. [nervous] But I keep hearing footsteps. [scared] They're getting closer. [gasps] [panicking] The door— it's opening by itself!\n```\n\n**Conversation with reactions:**\n```\n[curious] So what happened at the meeting? [pause] [surprised] Wait, they fired him?! [gasps] [sad] That's terrible... [sighs] He had a family. [thoughtful] I wonder what he'll do now.\n```\n\n**Hebrew (romantic moment):**\n```\n[soft] היא עמדה שם, מול השקיעה... [pause] הלב שלי פעם כל כך חזק. [nervous] לא ידעתי מה להגיד. [hesitates] אני... [breathes] [tender] את יודעת שאני אוהב אותך, נכון?\n```\n\n**Spanish (celebration to reflection):**\n```\n[excited] ¡Lo logramos! [laughs] [happy] No puedo creerlo... [pause] [thoughtful] Fueron tantos años de trabajo. [emotional] [soft] Gracias a todos los que creyeron en mí. [sighs] [content] Valió la pena cada momento.\n```\n\n## Configuration (OpenClaw)\n\nIn `openclaw.json`, configure TTS under `messages.tts`:\n\n```json\n{\n  \"messages\": {\n    \"tts\": {\n      \"provider\": \"elevenlabs\",\n      \"elevenlabs\": {\n        \"apiKey\": \"sk_your_api_key_here\",\n        \"voiceId\": \"pNInz6obpgDQGcFmaJgB\",\n        \"modelId\": \"eleven_v3\",\n        \"languageCode\": \"en\",\n        \"voiceSettings\": {\n          \"stability\": 0.5,\n          \"similarityBoost\": 0.75,\n          \"style\": 0,\n          \"useSpeakerBoost\": true,\n          \"speed\": 1\n        }\n      }\n    }\n  }\n}\n```\n\n**Getting your API Key:**\n1. Go to https://elevenlabs.io\n2. Sign up/login\n3. Click profile → API Keys\n4. Copy your key\n\n## Recommended Voices for v3\n\nThese premade voices are optimized for v3 and work well with audio tags:\n\n| Voice | ID | Gender | Accent | Best For |\n|-------|-----|--------|--------|----------|\n| **Adam** | `pNInz6obpgDQGcFmaJgB` | Male | American | Deep narration, general use |\n| **Rachel** | `21m00Tcm4TlvDq8ikWAM` | Female | American | Calm narration, conversational |\n| **Brian** | `nPczCjzI2devNBz1zQrb` | Male | American | Deep narration, podcasts |\n| **Charlotte** | `XB0fDUnXU5powFXDhCwa` | Female | English-Swedish | Expressive, video games |\n| **George** | `JBFqnCBsd6RMkjVDRZzb` | Male | British | Raspy narration, storytelling |\n\n**Finding more voices:**\n- Browse: https://elevenlabs.io/voice-library\n- v3-optimized collection: https://elevenlabs.io/app/voice-library/collections/aF6JALq9R6tXwCczjhKH\n- API: `GET https://api.elevenlabs.io/v1/voices`\n\n**Voice selection tips:**\n- Use IVC (Instant Voice Clone) or premade voices - PVC not optimized for v3 yet\n- Match voice character to your use case (whispering voice won't shout well)\n- For expressive IVCs, include varied emotional tones in training samples\n\n## Model Settings\n\n- **Model**: `eleven_v3` (alpha) - ONLY model supporting audio tags\n- **Languages**: 70+ supported with full audio tag control\n\n### Stability Modes\n\n| Mode | Stability | Description |\n|------|-----------|-------------|\n| **Creative** | 0.3-0.5 | More emotional/expressive, may hallucinate |\n| **Natural** | 0.5-0.7 | Balanced, closest to original voice |\n| **Robust** | 0.7-1.0 | Highly stable, less responsive to tags |\n\nFor audio tags, use **Creative** (0.5) or **Natural**. Higher stability reduces tag responsiveness.\n\n### Speed Control\n\nRange: 0.7 (slow) to 1.2 (fast), default 1.0\n\nExtreme values affect quality. For pacing, prefer audio tags like `[rushed]` or `[drawn out]`.\n\n## Critical Rules\n\n### Length Limits\n- **Optimal**: <800 characters per segment (best quality)\n- **Maximum**: 10,000 characters (API hard limit)\n- **Quality degrades** with longer text - voice becomes inconsistent\n\n### Audio Tags - Best Practices for Natural Sound\n\n**How many tags to use:**\n- 1-2 tags per sentence or phrase (not more!)\n- Tags persist until the next tag - no need to repeat\n- Overusing tags sounds unnatural and robotic\n\n**Where to place tags:**\n- At emotional transition points\n- Before key dramatic moments\n- When energy/pace changes\n\n**Context matters:**\n- Write text that *matches* the tag emotion\n- Longer text with context = better interpretation\n- Example: `[nervous] I... I'm not sure about this. What if it doesn't work?` works better than `[nervous] Hello.`\n\n**Combine tags for nuance:**\n- `[nervously][whispers]` = nervous whispering\n- `[excited][laughs]` = excited laughter\n- Keep combinations to 2 tags max\n\n**Regenerate for best results:**\n- v3 is non-deterministic - same text = different outputs\n- Generate 3+ versions, pick the best\n- Small text tweaks can improve results\n\n**Match tag to voice:**\n- Don't use `[shouts]` on a whispering voice\n- Don't use `[whispers]` on a loud/energetic voice\n- Test tags with your chosen voice\n\n### SSML Not Supported\nv3 does NOT support SSML break tags. Use audio tags and punctuation instead.\n\n### Punctuation Effects (use with tags!)\n\nPunctuation enhances audio tags:\n- **Ellipses (...)** → dramatic pauses: `[nervous] I... I don't know...`\n- **CAPS** → emphasis: `[excited] That's AMAZING!`\n- **Dashes (—)** → interruptions: `[explaining] So what you do is— [interrupting] Wait!`\n- **Question marks** → uncertainty: `[nervous] Are you sure about this?`\n- **Exclamation!** → energy boost: `[happy] We did it!`\n\nCombine tags + punctuation for maximum effect:\n```\n[tired] It was a long day... [sighs] Nobody listens anymore.\n```\n\n## WhatsApp Voice Messages\n\n### Complete Workflow\n\n1. **Generate** with `tts` tool (returns MP3)\n2. **Convert** to Opus (required for Android!)\n3. **Send** with `message` tool\n\n### Step-by-Step\n\n**1. Generate TTS (add [pause] at end to prevent cutoff):**\n```\ntts text=\"[excited] This is amazing! [pause]\" channel=whatsapp\n```\nReturns: `MEDIA:/tmp/tts-xxx/voice-123.mp3`\n\n**2. Convert MP3 → Opus:**\n```bash\nffmpeg -i /tmp/tts-xxx/voice-123.mp3 -c:a libopus -b:a 64k -vbr on -application voip /tmp/tts-xxx/voice-123.ogg\n```\n\n**3. Send the Opus file:**\n\n> **Note:** The `message` field below contains a Unicode Left-to-Right Mark (U+200E) between the quotes.\n> This is intentional — WhatsApp requires a non-empty message body to send voice notes.\n> The LTR mark is invisible but satisfies this requirement without displaying any text.\n\n```\nmessage action=send channel=whatsapp target=\"+972...\" filePath=\"/tmp/tts-xxx/voice-123.ogg\" asVoice=true message=\"‎\"\n```\n\n### Why Opus?\n\n| Format | iOS | Android | Transcribe |\n|--------|-----|---------|------------|\n| MP3 | ✅ Works | ❌ May fail | ❌ No |\n| Opus (.ogg) | ✅ Works | ✅ Works | ✅ Yes |\n\n**Always convert to Opus** - it's the only format that:\n- Works on all devices (iOS + Android)\n- Supports WhatsApp's transcribe button\n\n### Audio Cutoff Fix\n\nElevenLabs sometimes cuts off the last word. **Always add `[pause]` or `...` at the end:**\n```\n[excited] This is amazing! [pause]\n```\n\n## Long-Form Audio (Podcasts)\n\nFor content >800 chars:\n\n1. Split into short segments (<800 chars each)\n2. Generate each with `tts` tool\n3. Concatenate with ffmpeg:\n   ```bash\n   cat > list.txt << EOF\n   file '/path/file1.mp3'\n   file '/path/file2.mp3'\n   EOF\n   ffmpeg -f concat -safe 0 -i list.txt -c copy final.mp3\n   ```\n4. Convert to Opus for WhatsApp\n5. Send as single voice message\n\n**Important**: Don't mention \"part 2\" or \"chapter\" - keep it seamless.\n\n## Multi-Speaker Dialogue\n\nv3 can handle multiple characters in one generation:\n\n```\nJessica: [whispers] Did you hear that?\nChris: [interrupting] —I heard it too!\nJessica: [panicking] We need to hide!\n```\n\n**Dialogue tags**: `[interrupting]`, `[overlapping]`, `[cuts in]`, `[interjecting]`\n\n## Audio Tags Quick Reference\n\n| Category | Tags | When to Use |\n|----------|------|-------------|\n| **Emotions** | [excited], [happy], [sad], [angry], [nervous], [curious] | Main emotional state - use 1 per section |\n| **Delivery** | [whispers], [shouts], [soft], [rushed], [drawn out] | Volume/speed changes |\n| **Reactions** | [laughs], [sighs], [gasps], [clears throat], [gulps] | Natural human moments - sprinkle sparingly |\n| **Pacing** | [pause], [hesitates], [stammers], [breathes] | Dramatic timing |\n| **Character** | [French accent], [British accent], [robotic tone] | Character voice shifts |\n| **Dialogue** | [interrupting], [overlapping], [cuts in] | Multi-speaker conversations |\n\n**Most effective tags** (reliable results):\n- Emotions: `[excited]`, `[nervous]`, `[sad]`, `[happy]`\n- Reactions: `[laughs]`, `[sighs]`, `[whispers]`\n- Pacing: `[pause]`\n\n**Less reliable** (test and regenerate):\n- Sound effects: `[explosion]`, `[gunshot]`\n- Accents: results vary by voice\n\n**Full tag list**: See [references/audio-tags.md](references/audio-tags.md)\n\n## Troubleshooting\n\n**Tags read aloud?**\n- Verify using `eleven_v3` model\n- Use IVC/premade voices, not PVC\n- Simplify tags (no \"tone\" suffix)\n- Increase text length (250+ chars)\n\n**Voice inconsistent?**\n- Segment is too long - split at <800 chars\n- Regenerate (v3 is non-deterministic)\n- Try lower stability setting\n\n**WhatsApp won't play?**\n- Convert to Opus format (see above)\n\n**No emotion despite tags?**\n- Voice may not match tag style\n- Try Creative stability mode (0.5)\n- Add more context around the tag\n\nFile v2.2.0:_meta.json\n\n{\n  \"ownerId\": \"kn77700wny92h2kvpav2am1yjx80ewfp\",\n  \"slug\": \"elevenlabs-tts\",\n  \"version\": \"2.2.0\",\n  \"publishedAt\": 1771087774137\n}\n\nFile v2.2.0:references/audio-tags.md\n\n# Audio Tags Reference\n\nComplete guide to ElevenLabs v3 audio tags.\n\n## Prerequisites\n\n- **Model**: `eleven_v3` (alpha) - ONLY this model supports audio tags\n- **Voice Type**: IVC (Instant Voice Clone) or designed voices - PVC not optimized yet\n- **Prompt Length**: 250+ characters for consistent results\n- **Stability**: Creative or Natural mode (Robust reduces tag responsiveness)\n\n## Core Principle\n\nWrite NATURAL sentences that tags modify, NOT explanations.\n\n❌ WRONG: `[excited] אני מתרגש!`\n✅ RIGHT: `[excited] זה ממש מדהים מה שעשינו היום!`\n\n---\n\n## Tag Categories\n\n### Emotions (High Reliability)\n\n| Tag | Description |\n|-----|-------------|\n| `[excited]` | Energy, enthusiasm |\n| `[happy]` | Joy, cheerfulness |\n| `[happily]` | Speaking with happiness |\n| `[sad]` | Sadness, melancholy |\n| `[sorrowful]` | Deep sadness |\n| `[angry]` | Anger, intensity |\n| `[curious]` | Curiosity, interest |\n| `[nervous]` | Nervousness, anxiety |\n| `[sarcastic]` | Sarcasm, irony |\n| `[tired]` | Fatigue, weariness |\n| `[serious]` | Seriousness |\n| `[confident]` | Confidence |\n| `[frustrated]` | Frustration |\n| `[mischievous]` | Playful mischief |\n| `[awe]` | Wonder, amazement |\n| `[resigned]` | Acceptance, giving up |\n| `[flustered]` | Confused embarrassment |\n| `[casual]` | Relaxed, informal |\n| `[annoyed]` | Irritation |\n\n### Delivery & Volume (High Reliability)\n\n| Tag | Description |\n|-----|-------------|\n| `[whispers]` | Quiet, intimate |\n| `[shouts]` | Loud, intense |\n| `[dramatic tone]` | Theatrical |\n| `[dramatic]` | Dramatic delivery |\n| `[matter-of-fact]` | Plain, factual |\n| `[whiny]` | Complaining tone |\n| `[flatly]` | No emotion |\n| `[quietly]` | Soft voice |\n| `[suspiciously]` | Suspicious tone |\n\n### Pacing & Timing (High Reliability)\n\n| Tag | Description |\n|-----|-------------|\n| `[pause]` | Brief silence |\n| `[breathes]` | Breathing sound |\n| `[continues after a beat]` | Pause then continue |\n| `[rushed]` | Fast, urgent |\n| `[slows down]` | Decreasing speed |\n| `[deliberate]` | Careful, intentional |\n| `[rapid-fire]` | Very fast |\n| `[drawn out]` | Stretched, slow |\n| `[stammers]` | Stuttering |\n| `[hesitates]` | Uncertainty |\n| `[timidly]` | Shy, tentative |\n| `[repeats]` | Repetition |\n\n### Emphasis (Medium Reliability)\n\n| Tag | Description |\n|-----|-------------|\n| `[emphasized]` | Strong emphasis |\n| `[stress on next word]` | Emphasize following word |\n| `[understated]` | Downplayed delivery |\n\n### Reactions & Sounds (Very High Reliability)\n\n| Tag | Description |\n|-----|-------------|\n| `[laughs]` | Laughter |\n| `[laughs softly]` | Gentle laugh |\n| `[laughs harder]` | Increasing laughter |\n| `[starts laughing]` | Beginning to laugh |\n| `[nervous laugh]` | Anxious laughter |\n| `[giggles]` | Small laugh |\n| `[wheezing]` | Breathless laugh |\n| `[sighs]` | Exhale of emotion |\n| `[sigh]` | Single sigh |\n| `[gasps]` | Sharp intake |\n| `[exhales]` | Breathing out |\n| `[clears throat]` | Throat clearing |\n| `[gulps]` | Swallowing |\n| `[swallows]` | Swallowing sound |\n| `[snorts]` | Snorting sound |\n| `[crying]` | Sobbing |\n\n### Character & Accents (Medium Reliability)\n\n| Tag | Description |\n|-----|-------------|\n| `[French accent]` | French accent |\n| `[American accent]` | American accent |\n| `[British accent]` | British accent |\n| `[Australian accent]` | Australian accent |\n| `[Southern US accent]` | Southern American |\n| `[strong X accent]` | Replace X with accent |\n| `[pirate voice]` | Pirate character |\n| `[evil scientist voice]` | Mad scientist |\n| `[childlike tone]` | Child-like voice |\n| `[robotic tone]` | Robot voice |\n| `[deep voice]` | Lower pitch |\n\n### Narrative & Genre (Medium Reliability)\n\n| Tag | Description |\n|-----|-------------|\n| `[storytelling tone]` | Narrator voice |\n| `[voice-over style]` | Documentary style |\n| `[fantasy narrator]` | Epic fantasy |\n| `[sci-fi AI voice]` | Futuristic AI |\n| `[classic film noir]` | 1940s detective |\n| `[epic build-up]` | Building intensity |\n| `[narrative flourish]` | Dramatic narration |\n\n### Multi-Speaker Dialogue\n\n| Tag | Description |\n|-----|-------------|\n| `[interrupting]` | Cutting off speaker |\n| `[overlapping]` | Speaking over |\n| `[cuts in]` | Interjecting |\n| `[interjecting]` | Jumping in |\n| `[fast-paced]` | Quick exchange |\n\n### Sound Effects (Low-Medium Reliability)\n\n| Tag | Description |\n|-----|-------------|\n| `[gunshot]` | Gun sound |\n| `[clapping]` | Applause |\n| `[applause]` | Audience clapping |\n| `[explosion]` | Blast sound |\n| `[thunder]` | Thunder |\n\n### Experimental (Test First)\n\n| Tag | Description |\n|-----|-------------|\n| `[sings]` | Singing |\n| `[woo]` | Exclamation |\n| `[fart]` | Sound effect |\n| `[panicked]` | Panic |\n| `[trembling]` | Shaking voice |\n\n---\n\n## Usage Guidelines\n\n### ✅ DO:\n- Use simple tags: `[excited]` not `[excited tone]`\n- Write natural sentences that work without tags\n- Use 2-4 tags per paragraph max\n- Place tags at sentence start or key moment\n- Match tags to voice character\n- Test and regenerate (v3 is non-deterministic)\n- Combine tags: `[whispering][pause] Did you hear that?`\n\n### ❌ DON'T:\n- Don't add \"tone\" suffix: `[serious tone]` ❌\n- Don't overload with tags\n- Don't explain what the tag does\n- Don't use incompatible combos (whisper voice + shout tag)\n- Don't expect consistency (regenerate if needed)\n\n---\n\n## Examples\n\n### Emotional Monologue\n```\n[sighs] I've been thinking about what you said. [pause] \nAnd you're right. [sadly] I should have listened earlier.\n[determined] But I'm going to fix this. Starting now.\n```\n\n### Multi-Character Dialogue\n```\nSarah: [whispers] I think someone's coming.\nMike: [interrupting] —I heard it too! [panicked] Hide!\nSarah: [annoyed] I was TRYING to tell you that!\n```\n\n### Comedic Timing\n```\n[confident] So I walked up to the boss and said... \n[pause] [nervous laugh] Actually, I didn't say anything. \n[sighs] I just stood there. [laughs] Classic me.\n```\n\n### Accent Performance\n```\n[British accent] Terribly sorry, but I must insist.\n[switches to Southern US accent] Well now, that's mighty kind of y'all.\n[French accent] Mon ami, you simply must try ze croissant!\n```\n\n---\n\n## Troubleshooting\n\n**Tags being read aloud?**\n- Check you're using `eleven_v3` (not turbo_v3 or v2.5)\n- Use IVC/designed voices, not PVC\n- Simplify tags (remove \"tone\", \"sound\", etc.)\n- Increase prompt length (250+ chars)\n\n**Tags not working?**\n- Generate multiple times (v3 is variable)\n- Use Creative or Natural stability (not Robust)\n- Add surrounding context text\n- Try different tag placement\n- Voice may not match tag style\n\n**Multi-speaker not distinct?**\n- Add character cues: `[deep voice]`, `[higher pitch]`\n- Use accent tags for differentiation\n- Add emotional contrast between speakers\n\nArchive v2.1.0: 4 files, 10870 bytes\n\nFiles: lib/audio_convert.py (4003b), references/audio-tags.md (7739b), SKILL.md (11540b), _meta.json (133b)\n\nFile v2.1.0:SKILL.md\n\n---\nname: elevenlabs-tts\ndescription: ElevenLabs TTS (Text-to-Speech) with emotional audio tags for expressive voice synthesis. WhatsApp-compatible voice messages with Opus conversion. Supports 70+ languages, Hebrew with selective nikud, multi-speaker dialogue, and singing. Includes audio converter utility.\ntags: [elevenlabs, tts, voice, text-to-speech, audio, speech, whatsapp, multilingual, ai-voice, hebrew, nikud, singing]\nallowed-tools: [tts, message, exec]\n---\n\n# ElevenLabs TTS (Text-to-Speech)\n\nGenerate expressive voice messages using ElevenLabs v3 with audio tags.\n\n## Quick Start Examples\n\n**Storytelling (emotional journey):**\n```\n[soft] It started like any other day... [pause] But something felt different. [nervous] My hands were shaking as I opened the envelope. [gasps] I got in! [excited] I actually got in! [laughs] [happy] This changes everything!\n```\n\n**Horror/Suspense (building dread):**\n```\n[whispers] The house has been empty for years... [pause] At least, that's what they told me. [nervous] But I keep hearing footsteps. [scared] They're getting closer. [gasps] [panicking] The door— it's opening by itself!\n```\n\n**Conversation with reactions:**\n```\n[curious] So what happened at the meeting? [pause] [surprised] Wait, they fired him?! [gasps] [sad] That's terrible... [sighs] He had a family. [thoughtful] I wonder what he'll do now.\n```\n\n**Hebrew (romantic moment - selective nikud only where needed):**\n```\n[soft] היא עמדה שם, מול השקיעה... [pause] הלב שלי פעם כל כך חזק. [nervous] לא ידעתי מה להגיד. [hesitates] אני... [breathes] [tender] אַתְּ יודעת שאני אוהב אותָךְ, נכון?\n```\n\n**Spanish (celebration to reflection):**\n```\n[excited] ¡Lo logramos! [laughs] [happy] No puedo creerlo... [pause] [thoughtful] Fueron tantos años de trabajo. [emotional] [soft] Gracias a todos los que creyeron en mí. [sighs] [content] Valió la pena cada momento.\n```\n\n## Configuration (OpenClaw)\n\nIn `openclaw.json`, configure TTS under `messages.tts`:\n\n```json\n{\n  \"messages\": {\n    \"tts\": {\n      \"provider\": \"elevenlabs\",\n      \"elevenlabs\": {\n        \"apiKey\": \"sk_your_api_key_here\",\n        \"voiceId\": \"YOUR_VOICE_ID\",\n        \"modelId\": \"eleven_v3\",\n        \"languageCode\": \"en\",\n        \"voiceSettings\": {\n          \"stability\": 0.5,\n          \"similarityBoost\": 0.75,\n          \"style\": 0,\n          \"useSpeakerBoost\": true,\n          \"speed\": 1\n        }\n      }\n    }\n  }\n}\n```\n\n**Getting your API Key:**\n1. Go to https://elevenlabs.io\n2. Sign up/login\n3. Click profile → API Keys\n4. Copy your key\n\n## Recommended Voices for v3\n\nThese premade voices are optimized for v3 and work well with audio tags:\n\n| Voice | ID | Gender | Accent | Best For |\n|-------|-----|--------|--------|----------|\n| **Adam** | `pNInz6obpgDQGcFmaJgB` | Male | American | Deep narration, general use |\n| **Rachel** | `21m00Tcm4TlvDq8ikWAM` | Female | American | Calm narration, conversational |\n| **Brian** | `nPczCjzI2devNBz1zQrb` | Male | American | Deep narration, podcasts |\n| **Charlotte** | `XB0fDUnXU5powFXDhCwa` | Female | English-Swedish | Expressive, video games |\n| **George** | `JBFqnCBsd6RMkjVDRZzb` | Male | British | Raspy narration, storytelling |\n\n**Finding more voices:**\n- Browse: https://elevenlabs.io/voice-library\n- v3-optimized collection: https://elevenlabs.io/app/voice-library/collections/aF6JALq9R6tXwCczjhKH\n- API: `GET https://api.elevenlabs.io/v1/voices`\n\n**Voice selection tips:**\n- Use IVC (Instant Voice Clone) or premade voices - PVC not optimized for v3 yet\n- Match voice character to your use case (whispering voice won't shout well)\n- For expressive IVCs, include varied emotional tones in training samples\n\n## Model Settings\n\n- **Model**: `eleven_v3` (alpha) - ONLY model supporting audio tags\n- **Languages**: 70+ supported with full audio tag control\n\n### Stability Modes\n\nv3 only accepts three values: 0.0, 0.5, 1.0\n\n| Mode | Value | Description |\n|------|-------|-------------|\n| **Creative** | 0.0 | Most emotional/expressive, best for singing, may hallucinate |\n| **Natural** | 0.5 | Balanced, closest to original voice |\n| **Robust** | 1.0 | Highly stable, less responsive to tags |\n\nFor audio tags, use **Creative** (0.0) or **Natural** (0.5). Robust reduces tag responsiveness.\n\n### Speed Control\n\nRange: 0.7 (slow) to 1.2 (fast), default 1.0\n\nExtreme values affect quality. For pacing, prefer audio tags like `[rushed]` or `[drawn out]`.\n\n## Hebrew Nikud (Vowel Points)\n\nUse nikud **selectively** - only on words where pronunciation is ambiguous. Full nikud on every word can degrade quality.\n\n**The rule: only add nikud where the model might guess wrong.**\n\nCommon cases where nikud helps:\n1. **Gender suffixes** - שלומֵךְ (f) vs שלומְךָ (m), לָךְ (f) vs לְךָ (m), אותָךְ (f) vs אותְךָ (m)\n2. **Dagesh (hard/soft consonants)** - letters בכפ change sound with dagesh:\n   - פּ = P, פ = F: פִּיצה (pizza), פִּייר (Pierre)\n   - בּ = B, ב = V: בְּרָכָה (brakha), בְּדִיוּק (bediyuk)\n   - כּ = K, כ = Kh: כּוֹס (kos), כַּמָּה (kama)\n3. **Homographs** - same spelling, different meaning/pronunciation:\n   - בּוֹקֶר (morning) vs בּוֹקֵר (cowboy)\n   - עוֹלָם (world) vs עוֹלֵם (concealing)\n   - סֵפֶר (book) vs סָפַר (counted)\n4. **Foreign names and loanwords** - the model often guesses wrong\n5. **Stress placement** - when it changes meaning or sounds unnatural\n\n**When NOT to add nikud:**\n- Common words with only one pronunciation (מה, יש, הרבה, שלום, אני, הוא, etc.)\n- Context makes pronunciation obvious\n- Most of the sentence - keep it clean\n\n**Example:**\n```\n❌ Full nikud: מַה שְׁלוֹמְךָ? יֵשׁ לְךָ הַרְבֵּה כֶּסֶף.\n✅ Selective: מה שלומְךָ? יש לְךָ הרבה כסף.\n✅ Dagesh: ז'אן-פִּייר אפה פִּיצה מושלמת.\n```\n\n**Principle:** If you read the word and there's only one way to say it - skip the nikud. If there's ambiguity - add it.\n\n## Critical Rules\n\n### Length Limits\n- **Optimal**: <800 characters per segment (best quality)\n- **Maximum**: 10,000 characters (API hard limit)\n- **Quality degrades** with longer text - voice becomes inconsistent\n\n### Audio Tags - Best Practices for Natural Sound\n\n**How many tags to use:**\n- 1-2 tags per sentence or phrase (not more!)\n- Tags persist until the next tag - no need to repeat\n- Overusing tags sounds unnatural and robotic\n\n**Where to place tags:**\n- At emotional transition points\n- Before key dramatic moments\n- When energy/pace changes\n\n**Context matters:**\n- Write text that *matches* the tag emotion\n- Longer text with context = better interpretation\n- Example: `[nervous] I... I'm not sure about this. What if it doesn't work?` works better than `[nervous] Hello.`\n\n**Combine tags for nuance:**\n- `[nervously][whispers]` = nervous whispering\n- `[excited][laughs]` = excited laughter\n- Keep combinations to 2 tags max\n\n**Regenerate for best results:**\n- v3 is non-deterministic - same text = different outputs\n- Generate 3+ versions, pick the best\n- Small text tweaks can improve results\n\n**Match tag to voice:**\n- Don't use `[shouts]` on a whispering voice\n- Don't use `[whispers]` on a loud/energetic voice\n- Test tags with your chosen voice\n\n### SSML Not Supported\nv3 does NOT support SSML break tags. Use audio tags and punctuation instead.\n\n### Punctuation Effects (use with tags!)\n\nPunctuation enhances audio tags:\n- **Ellipses (...)** → dramatic pauses: `[nervous] I... I don't know...`\n- **CAPS** → emphasis: `[excited] That's AMAZING!`\n- **Dashes (—)** → interruptions: `[explaining] So what you do is— [interrupting] Wait!`\n- **Question marks** → uncertainty: `[nervous] Are you sure about this?`\n- **Exclamation!** → energy boost: `[happy] We did it!`\n\nCombine tags + punctuation for maximum effect:\n```\n[tired] It was a long day... [sighs] Nobody listens anymore.\n```\n\n## WhatsApp Voice Messages\n\n### Complete Workflow\n\n1. **Generate** with `tts` tool (returns MP3)\n2. **Convert** to Opus (required for Android!)\n3. **Send** with `message` tool\n\n### Step-by-Step\n\n**1. Generate TTS (add [pause] at end to prevent cutoff):**\n```\ntts text=\"[excited] This is amazing! [pause]\" channel=whatsapp\n```\nReturns: `MEDIA:/tmp/tts-xxx/voice-123.mp3`\n\n**2. Convert MP3 → Opus using the included converter:**\n```\npython3 lib/audio_convert.py convert /tmp/tts-xxx/voice-123.mp3 /tmp/tts-xxx/voice-123.ogg\n```\n\n**3. Send the Opus file:**\n```\nmessage action=send channel=whatsapp target=\"+972...\" filePath=\"/tmp/tts-xxx/voice-123.ogg\" asVoice=true message=\"‎\"\n```\n\n### Why Opus?\n\n| Format | iOS | Android | Transcribe |\n|--------|-----|---------|------------|\n| MP3 | ✅ Works | ❌ May fail | ❌ No |\n| Opus (.ogg) | ✅ Works | ✅ Works | ✅ Yes |\n\n**Always convert to Opus** - it's the only format that:\n- Works on all devices (iOS + Android)\n- Supports WhatsApp's transcribe button\n\n### Audio Cutoff Fix\n\nElevenLabs sometimes cuts off the last word. **Always add `[pause]` or `...` at the end:**\n```\n[excited] This is amazing! [pause]\n```\n\n## Long-Form Audio (Podcasts)\n\nFor content >800 chars:\n\n1. Split into short segments (<800 chars each)\n2. Generate each with `tts` tool\n3. Concatenate using the included converter:\n   ```\n   python3 lib/audio_convert.py concat /tmp/final.mp3 /tmp/part1.mp3 /tmp/part2.mp3\n   ```\n4. Convert to Opus for WhatsApp:\n   ```\n   python3 lib/audio_convert.py convert /tmp/final.mp3 /tmp/final.ogg\n   ```\n5. Send as single voice message\n\n**Important**: Don't mention \"part 2\" or \"chapter\" - keep it seamless.\n\n## Multi-Speaker Dialogue\n\nv3 can handle multiple characters in one generation:\n\n```\nJessica: [whispers] Did you hear that?\nChris: [interrupting] —I heard it too!\nJessica: [panicking] We need to hide!\n```\n\n**Dialogue tags**: `[interrupting]`, `[overlapping]`, `[cuts in]`, `[interjecting]`\n\n## Audio Tags Quick Reference\n\n| Category | Tags | When to Use |\n|----------|------|-------------|\n| **Emotions** | [excited], [happy], [sad], [angry], [nervous], [curious] | Main emotional state - use 1 per section |\n| **Delivery** | [whispers], [shouts], [soft], [rushed], [drawn out] | Volume/speed changes |\n| **Reactions** | [laughs], [sighs], [gasps], [clears throat], [gulps] | Natural human moments - sprinkle sparingly |\n| **Pacing** | [pause], [hesitates], [stammers], [breathes] | Dramatic timing |\n| **Character** | [French accent], [British accent], [robotic tone] | Character voice shifts |\n| **Dialogue** | [interrupting], [overlapping], [cuts in] | Multi-speaker conversations |\n\n**Most effective tags** (reliable results):\n- Emotions: `[excited]`, `[nervous]`, `[sad]`, `[happy]`\n- Reactions: `[laughs]`, `[sighs]`, `[whispers]`\n- Pacing: `[pause]`\n\n**Less reliable** (test and regenerate):\n- Sound effects: `[explosion]`, `[gunshot]`\n- Accents: results vary by voice\n\n**Full tag list**: See [references/audio-tags.md](references/audio-tags.md)\n\n## Troubleshooting\n\n**Tags read aloud?**\n- Verify using `eleven_v3` model\n- Use IVC/premade voices, not PVC\n- Simplify tags (no \"tone\" suffix)\n- Increase text length (250+ chars)\n\n**Voice inconsistent?**\n- Segment is too long - split at <800 chars\n- Regenerate (v3 is non-deterministic)\n- Try lower stability setting\n\n**WhatsApp won't play?**\n- Convert to Opus format (see above)\n\n**No emotion despite tags?**\n- Voice may not match tag style\n- Try Creative stability mode (0.0)\n- Add more context around the tag\n\nFile v2.1.0:_meta.json\n\n{\n  \"ownerId\": \"kn77700wny92h2kvpav2am1yjx80ewfp\",\n  \"slug\": \"elevenlabs-tts\",\n  \"version\": \"2.1.0\",\n  \"publishedAt\": 1770644808695\n}\n\nFile v2.1.0:references/audio-tags.md\n\n# Audio Tags Reference\n\nComplete guide to ElevenLabs v3 audio tags.\n\n## Prerequisites\n\n- **Model**: `eleven_v3` (alpha) - ONLY this model supports audio tags\n- **Voice Type**: IVC (Instant Voice Clone) or designed voices - PVC not optimized yet\n- **Prompt Length**: 250+ characters for consistent results\n- **Stability**: Creative or Natural mode (Robust reduces tag responsiveness)\n\n## Core Principle\n\nWrite NATURAL sentences that tags modify, NOT explanations.\n\n❌ WRONG: `[excited] אני מתרגש!`\n✅ RIGHT: `[excited] זה ממש מדהים מה שעשינו היום!`\n\n---\n\n## Tag Categories\n\n### Emotions (High Reliability)\n\n| Tag | Description |\n|-----|-------------|\n| `[excited]` | Energy, enthusiasm |\n| `[happy]` | Joy, cheerfulness |\n| `[happily]` | Speaking with happiness |\n| `[sad]` | Sadness, melancholy |\n| `[sorrowful]` | Deep sadness |\n| `[angry]` | Anger, intensity |\n| `[curious]` | Curiosity, interest |\n| `[nervous]` | Nervousness, anxiety |\n| `[sarcastic]` | Sarcasm, irony |\n| `[tired]` | Fatigue, weariness |\n| `[serious]` | Seriousness |\n| `[confident]` | Confidence |\n| `[frustrated]` | Frustration |\n| `[mischievous]` | Playful mischief |\n| `[awe]` | Wonder, amazement |\n| `[resigned]` | Acceptance, giving up |\n| `[flustered]` | Confused embarrassment |\n| `[casual]` | Relaxed, informal |\n| `[annoyed]` | Irritation |\n\n### Delivery & Volume (High Reliability)\n\n| Tag | Description |\n|-----|-------------|\n| `[whispers]` | Quiet, intimate |\n| `[shouts]` | Loud, intense |\n| `[dramatic tone]` | Theatrical |\n| `[dramatic]` | Dramatic delivery |\n| `[matter-of-fact]` | Plain, factual |\n| `[whiny]` | Complaining tone |\n| `[flatly]` | No emotion |\n| `[quietly]` | Soft voice |\n| `[suspiciously]` | Suspicious tone |\n\n### Pacing & Timing (High Reliability)\n\n| Tag | Description |\n|-----|-------------|\n| `[pause]` | Brief silence |\n| `[breathes]` | Breathing sound |\n| `[continues after a beat]` | Pause then continue |\n| `[rushed]` | Fast, urgent |\n| `[slows down]` | Decreasing speed |\n| `[deliberate]` | Careful, intentional |\n| `[rapid-fire]` | Very fast |\n| `[drawn out]` | Stretched, slow |\n| `[stammers]` | Stuttering |\n| `[hesitates]` | Uncertainty |\n| `[timidly]` | Shy, tentative |\n| `[repeats]` | Repetition |\n\n### Emphasis (Medium Reliability)\n\n| Tag | Description |\n|-----|-------------|\n| `[emphasized]` | Strong emphasis |\n| `[stress on next word]` | Emphasize following word |\n| `[understated]` | Downplayed delivery |\n\n### Reactions & Sounds (Very High Reliability)\n\n| Tag | Description |\n|-----|-------------|\n| `[laughs]` | Laughter |\n| `[laughs softly]` | Gentle laugh |\n| `[laughs harder]` | Increasing laughter |\n| `[starts laughing]` | Beginning to laugh |\n| `[nervous laugh]` | Anxious laughter |\n| `[giggles]` | Small laugh |\n| `[wheezing]` | Breathless laugh |\n| `[sighs]` | Exhale of emotion |\n| `[sigh]` | Single sigh |\n| `[gasps]` | Sharp intake |\n| `[exhales]` | Breathing out |\n| `[clears throat]` | Throat clearing |\n| `[gulps]` | Swallowing |\n| `[swallows]` | Swallowing sound |\n| `[snorts]` | Snorting sound |\n| `[crying]` | Sobbing |\n\n### Character & Accents (Medium Reliability)\n\n| Tag | Description |\n|-----|-------------|\n| `[French accent]` | French accent |\n| `[American accent]` | American accent |\n| `[British accent]` | British accent |\n| `[Australian accent]` | Australian accent |\n| `[Southern US accent]` | Southern American |\n| `[strong X accent]` | Replace X with accent |\n| `[pirate voice]` | Pirate character |\n| `[evil scientist voice]` | Mad scientist |\n| `[childlike tone]` | Child-like voice |\n| `[robotic tone]` | Robot voice |\n| `[deep voice]` | Lower pitch |\n\n### Narrative & Genre (Medium Reliability)\n\n| Tag | Description |\n|-----|-------------|\n| `[storytelling tone]` | Narrator voice |\n| `[voice-over style]` | Documentary style |\n| `[fantasy narrator]` | Epic fantasy |\n| `[sci-fi AI voice]` | Futuristic AI |\n| `[classic film noir]` | 1940s detective |\n| `[epic build-up]` | Building intensity |\n| `[narrative flourish]` | Dramatic narration |\n\n### Multi-Speaker Dialogue\n\n| Tag | Description |\n|-----|-------------|\n| `[interrupting]` | Cutting off speaker |\n| `[overlapping]` | Speaking over |\n| `[cuts in]` | Interjecting |\n| `[interjecting]` | Jumping in |\n| `[fast-paced]` | Quick exchange |\n\n### Sound Effects (Low-Medium Reliability)\n\n| Tag | Description |\n|-----|-------------|\n| `[gunshot]` | Gun sound |\n| `[clapping]` | Applause |\n| `[applause]` | Audience clapping |\n| `[explosion]` | Blast sound |\n| `[thunder]` | Thunder |\n\n### Experimental (Test First)\n\n| Tag | Description |\n|-----|-------------|\n| `[sings]` | Singing |\n| `[woo]` | Exclamation |\n| `[fart]` | Sound effect |\n| `[panicked]` | Panic |\n| `[trembling]` | Shaking voice |\n\n---\n\n## Usage Guidelines\n\n### ✅ DO:\n- Use simple tags: `[excited]` not `[excited tone]`\n- Write natural sentences that work without tags\n- Use 2-4 tags per paragraph max\n- Place tags at sentence start or key moment\n- Match tags to voice character\n- Test and regenerate (v3 is non-deterministic)\n- Combine tags: `[whispering][pause] Did you hear that?`\n\n### ❌ DON'T:\n- Don't add \"tone\" suffix: `[serious tone]` ❌\n- Don't overload with tags\n- Don't explain what the tag does\n- Don't use incompatible combos (whisper voice + shout tag)\n- Don't expect consistency (regenerate if needed)\n\n---\n\n## Examples\n\n### Emotional Monologue\n```\n[sighs] I've been thinking about what you said. [pause] \nAnd you're right. [sadly] I should have listened earlier.\n[determined] But I'm going to fix this. Starting now.\n```\n\n### Multi-Character Dialogue\n```\nSarah: [whispers] I think someone's coming.\nMike: [interrupting] —I heard it too! [panicked] Hide!\nSarah: [annoyed] I was TRYING to tell you that!\n```\n\n### Comedic Timing\n```\n[confident] So I walked up to the boss and said... \n[pause] [nervous laugh] Actually, I didn't say anything. \n[sighs] I just stood there. [laughs] Classic me.\n```\n\n### Accent Performance\n```\n[British accent] Terribly sorry, but I must insist.\n[switches to Southern US accent] Well now, that's mighty kind of y'all.\n[French accent] Mon ami, you simply must try ze croissant!\n```\n\n---\n\n## Singing\n\nThe `[singing]` tag can produce melodic intonation. Results are inconsistent - v3 is a TTS model, not a music model.\n\n**Format** (tag on its own line before lyrics):\n```\n\n[singing]\nOh Tommy boy, the pipes the pipes are calling,\nfrom glen to glen and down the mountain side.\n```\n\n**Best settings for singing:**\n- **Stability**: Creative (0.0) - most expressive, best for singing\n- **Voice**: Use v3-optimized premade voices (Adam, Charlotte, etc.)\n- **Language**: English works best; Hebrew is less reliable for singing\n- **Non-deterministic**: Generate multiple times - each result is different\n\n**Tips:**\n- Put `[singing]` on its own line before lyrics\n- Use known songs the model might recognize\n- Stack with emotion: `[happy]\\n[singing]\\nlyrics...`\n- Keep lyrics short per generation\n\n**Limitations:**\n- Not real singing with full melody - more like melodic speech\n- Results vary heavily by voice and generation\n- For actual music generation, use **Suno** or **Udio**\n\n---\n\n## Troubleshooting\n\n**Tags being read aloud?**\n- Check you're using `eleven_v3` (not turbo_v3 or v2.5)\n- Use IVC/designed voices, not PVC\n- Simplify tags (remove \"tone\", \"sound\", etc.)\n- Increase prompt length (250+ chars)\n\n**Tags not working?**\n- Generate multiple times (v3 is variable)\n- Use Creative or Natural stability (not Robust)\n- Add surrounding context text\n- Try different tag placement\n- Voice may not match tag style\n\n**Multi-speaker not distinct?**\n- Add character cues: `[deep voice]`, `[higher pitch]`\n- Use accent tags for differentiation\n- Add emotional contrast between speakers","readmeExcerpt":"Skill: Elevenlabs Tts Owner: Shaharsha Summary: ElevenLabs TTS (Text-to-Speech) with emotional audio tags for expressive voice synthesis. WhatsApp-compatible voice messages with Opus conversion. Supports 7... Tags: ai-voice:2.1.0, audio:2.1.0, elevenlabs:2.1.0, elevenlabs-tts:1.3.2, hebrew:2.1.0, latest:2.2.0, multilingual:2.1.0, nikud:2.1.0, openclaw:1.3.2, podcast:1.2.1, singing:2.1.0, speech:2.1.0, text-to-speech:","codeSnippets":[],"executableExamples":[{"language":"text","snippet":"[soft] It started like any other day... [pause] But something felt different. [nervous] My hands were shaking as I opened the envelope. [gasps] I got in! [excited] I actually got in! [laughs] [happy] This changes everything!"},{"language":"text","snippet":"[whispers] The house has been empty for years... [pause] At least, that's what they told me. [nervous] But I keep hearing footsteps. [scared] They're getting closer. [gasps] [panicking] The door— it's opening by itself!"},{"language":"text","snippet":"[curious] So what happened at the meeting? [pause] [surprised] Wait, they fired him?! [gasps] [sad] That's terrible... [sighs] He had a family. [thoughtful] I wonder what he'll do now."},{"language":"text","snippet":"[soft] היא עמדה שם, מול השקיעה... [pause] הלב שלי פעם כל כך חזק. [nervous] לא ידעתי מה להגיד. [hesitates] אני... [breathes] [tender] את יודעת שאני אוהב אותך, נכון?"},{"language":"text","snippet":"[excited] ¡Lo logramos! [laughs] [happy] No puedo creerlo... [pause] [thoughtful] Fueron tantos años de trabajo. [emotional] [soft] Gracias a todos los que creyeron en mí. [sighs] [content] Valió la pena cada momento."},{"language":"json","snippet":"{\n  \"messages\": {\n    \"tts\": {\n      \"provider\": \"elevenlabs\",\n      \"elevenlabs\": {\n        \"apiKey\": \"sk_your_api_key_here\",\n        \"voiceId\": \"pNInz6obpgDQGcFmaJgB\",\n        \"modelId\": \"eleven_v3\",\n        \"languageCode\": \"en\",\n        \"voiceSettings\": {\n          \"stability\": 0.5,\n          \"similarityBoost\": 0.75,\n          \"style\": 0,\n          \"useSpeakerBoost\": true,\n          \"speed\": 1\n        }\n      }\n    }\n  }\n}"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: elevenlabs-tts\ndescription: ElevenLabs TTS - the best ElevenLabs integration for OpenClaw. ElevenLabs Text-to-Speech with emotional audio tags, ElevenLabs voice synthesis for WhatsApp, ElevenLabs multilingual support. Generate realistic AI voices using ElevenLabs API.\ntags: [elevenlabs, tts, voice, text-to-speech, audio, speech, whatsapp, multilingual, ai-voice]\nmetadata: {\"clawdbot\":{\"emoji\":\"🎙️\",\"requires\":{\"env\":[\"ELEVENLABS_API_KEY\"],\"system\":[\"ffmpeg\"]},\"primaryEnv\":\"ELEVENLABS_API_KEY\"}}\nallowed-tools: [exec, tts, message]\n---\n\n# ElevenLabs TTS (Text-to-Speech)\n\nGenerate expressive voice messages using ElevenLabs v3 with audio tags.\n\n## Prerequisites\n\n- **ElevenLabs API Key** (`ELEVENLABS_API_KEY`): Required. Get one at [elevenlabs.io](https://elevenlabs.io) → Profile → API Keys. Configure in `openclaw.json` under `messages.tts.elevenlabs.apiKey`.\n- **ffmpeg**: Required for audio format conversion (MP3 → Opus for WhatsApp compatibility). Must be installed and available on PATH.\n\n## Quick Start Examples\n\n**Storytelling (emotional journey):**\n```\n[soft] It started like any other day... [pause] But something felt different. [nervous] My hands were shaking as I opened the envelope. [gasps] I got in! [excited] I actually got in! [laughs] [happy] This changes everything!\n```\n\n**Horror/Suspense (building dread):**\n```\n[whispers] The house has been empty for years... [pause] At least, that's what they told me. [nervous] But I keep hearing footsteps. [scared] They're getting closer. [gasps] [panicking] The door— it's opening by itself!\n```\n\n**Conversation with reactions:**\n```\n[curious] So what happened at the meeting? [pause] [surprised] Wait, they fired him?! [gasps] [sad] That's terrible... [sighs] He had a family. [thoughtful] I wonder what he'll do now.\n```\n\n**Hebrew (romantic moment):**\n```\n[soft] היא עמדה שם, מול השקיעה... [pause] הלב שלי פעם כל כך חזק. [nervous] לא ידעתי מה להגיד. [hesitates] אני... [breathes] [tender] את יודעת שאני אוהב אותך, נכון?\n```\n\n**Spanish (celebration to reflection):**\n```\n[excited] ¡Lo logramos! [laughs] [happy] No puedo creerlo... [pause] [thoughtful] Fueron tantos años de trabajo. [emotional] [soft] Gracias a todos los que creyeron en mí. [sighs] [content] Valió la pena cada momento.\n```\n\n## Configuration (OpenClaw)\n\nIn `openclaw.json`, configure TTS under `messages.tts`:\n\n```json\n{\n  \"messages\": {\n    \"tts\": {\n      \"provider\": \"elevenlabs\",\n      \"elevenlabs\": {\n        \"apiKey\": \"sk_your_api_key_here\",\n        \"voiceId\": \"pNInz6obpgDQGcFmaJgB\",\n        \"modelId\": \"eleven_v3\",\n        \"languageCode\": \"en\",\n        \"voiceSettings\": {\n          \"stability\": 0.5,\n          \"similarityBoost\": 0.75,\n          \"style\": 0,\n          \"useSpeakerBoost\": true,\n          \"speed\": 1\n        }\n      }\n    }\n  }\n}\n```\n\n**Getting your API Key:**\n1. Go to https://elevenlabs.io\n2. Sign up/login\n3. Click profile → API Keys\n4. Copy your key\n\n## Recommended Voices for v3\n\nThese premade voices are optimized for v3 and wo"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn77700wny92h2kvpav2am1yjx80ewfp\",\n  \"slug\": \"elevenlabs-tts\",\n  \"version\": \"2.2.0\",\n  \"publishedAt\": 1771087774137\n}"},{"path":"references/audio-tags.md","content":"# Audio Tags Reference\n\nComplete guide to ElevenLabs v3 audio tags.\n\n## Prerequisites\n\n- **Model**: `eleven_v3` (alpha) - ONLY this model supports audio tags\n- **Voice Type**: IVC (Instant Voice Clone) or designed voices - PVC not optimized yet\n- **Prompt Length**: 250+ characters for consistent results\n- **Stability**: Creative or Natural mode (Robust reduces tag responsiveness)\n\n## Core Principle\n\nWrite NATURAL sentences that tags modify, NOT explanations.\n\n❌ WRONG: `[excited] אני מתרגש!`\n✅ RIGHT: `[excited] זה ממש מדהים מה שעשינו היום!`\n\n---\n\n## Tag Categories\n\n### Emotions (High Reliability)\n\n| Tag | Description |\n|-----|-------------|\n| `[excited]` | Energy, enthusiasm |\n| `[happy]` | Joy, cheerfulness |\n| `[happily]` | Speaking with happiness |\n| `[sad]` | Sadness, melancholy |\n| `[sorrowful]` | Deep sadness |\n| `[angry]` | Anger, intensity |\n| `[curious]` | Curiosity, interest |\n| `[nervous]` | Nervousness, anxiety |\n| `[sarcastic]` | Sarcasm, irony |\n| `[tired]` | Fatigue, weariness |\n| `[serious]` | Seriousness |\n| `[confident]` | Confidence |\n| `[frustrated]` | Frustration |\n| `[mischievous]` | Playful mischief |\n| `[awe]` | Wonder, amazement |\n| `[resigned]` | Acceptance, giving up |\n| `[flustered]` | Confused embarrassment |\n| `[casual]` | Relaxed, informal |\n| `[annoyed]` | Irritation |\n\n### Delivery & Volume (High Reliability)\n\n| Tag | Description |\n|-----|-------------|\n| `[whispers]` | Quiet, intimate |\n| `[shouts]` | Loud, intense |\n| `[dramatic tone]` | Theatrical |\n| `[dramatic]` | Dramatic delivery |\n| `[matter-of-fact]` | Plain, factual |\n| `[whiny]` | Complaining tone |\n| `[flatly]` | No emotion |\n| `[quietly]` | Soft voice |\n| `[suspiciously]` | Suspicious tone |\n\n### Pacing & Timing (High Reliability)\n\n| Tag | Description |\n|-----|-------------|\n| `[pause]` | Brief silence |\n| `[breathes]` | Breathing sound |\n| `[continues after a beat]` | Pause then continue |\n| `[rushed]` | Fast, urgent |\n| `[slows down]` | Decreasing speed |\n| `[deliberate]` | Careful, intentional |\n| `[rapid-fire]` | Very fast |\n| `[drawn out]` | Stretched, slow |\n| `[stammers]` | Stuttering |\n| `[hesitates]` | Uncertainty |\n| `[timidly]` | Shy, tentative |\n| `[repeats]` | Repetition |\n\n### Emphasis (Medium Reliability)\n\n| Tag | Description |\n|-----|-------------|\n| `[emphasized]` | Strong emphasis |\n| `[stress on next word]` | Emphasize following word |\n| `[understated]` | Downplayed delivery |\n\n### Reactions & Sounds (Very High Reliability)\n\n| Tag | Description |\n|-----|-------------|\n| `[laughs]` | Laughter |\n| `[laughs softly]` | Gentle laugh |\n| `[laughs harder]` | Increasing laughter |\n| `[starts laughing]` | Beginning to laugh |\n| `[nervous laugh]` | Anxious laughter |\n| `[giggles]` | Small laugh |\n| `[wheezing]` | Breathless laugh |\n| `[sighs]` | Exhale of emotion |\n| `[sigh]` | Single sigh |\n| `[gasps]` | Sharp intake |\n| `[exhales]` | Breathing out |\n| `[clears throat]` | Throat clearing |\n| `[gulps]` | Swallowing |\n| `[swallows]` | Swallowin"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"ElevenLabs TTS (Text-to-Speech) with emotional audio tags for expressive voice synthesis. WhatsApp-compatible voice messages with Opus conversion. Supports 7... Skill: Elevenlabs Tts Owner: Shaharsha Summary: ElevenLabs TTS (Text-to-Speech) with emotional audio tags for expressive voice synthesis. WhatsApp-compatible voice messages with Opus conversion. Supports 7... Tags: ai-voice:2.1.0, audio:2.1.0, elevenlabs:2.1.0, elevenlabs-tts:1.3.2, hebrew:2.1.0, latest:2.2.0, multilingual:2.1.0, nikud:2.1.0, openclaw:1.3.2, podcast:1.2.1, singing:2.1.0, speech:2.1.0, text-to-speech:","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1043,"uniquenessScore":63,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"agent-directory","verified":false,"confidence":"low","updatedAt":"2026-10-10T02:45:36.584Z","emptyReason":"No close protocol neighbors were found."},"items":[],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[]}}}