{"id":"db0e52b3-c692-4033-9531-2cd9731ea51b","entityType":"agent","slug":"clawhub-eftalyurtseven-eachlabs-voice-audio","name":"Eachlabs Voice Audio","canonicalUrl":"https://www.xpersona.co/agent/clawhub-eftalyurtseven-eachlabs-voice-audio","canonicalPath":"/agent/clawhub-eftalyurtseven-eachlabs-voice-audio","generatedAt":"2026-10-09T12:50:22.942Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":null},"description":"Text-to-speech, speech-to-text, voice conversion, and audio processing using EachLabs AI models. Supports ElevenLabs TTS, Whisper transcription with diarization, and RVC voice conversion. Use when the user needs TTS, transcription, or voice conversion.","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 622 downloads reported by the source. Last updated 4/15/2026.","installCommand":"clawhub skill install kn7daknf647gdfdwzpyq5scnvn80tq7d:eachlabs-voice-audio","sourceUrl":"https://clawhub.ai/eftalyurtseven/eachlabs-voice-audio","homepage":"https://clawhub.ai/eftalyurtseven/eachlabs-voice-audio","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/eftalyurtseven/eachlabs-voice-audio","kind":"source"}],"safetyScore":84,"overallRank":62,"popularityScore":56,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Eachlabs Voice Audio technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":null},"stars":null,"forks":null,"downloads":622,"packageName":null,"latestVersion":"0.1.0","tractionLabel":"622 downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-03-01T03:53:00.746Z","emptyReason":null},"lastUpdatedAt":"2026-04-15T00:45:39.800Z","lastCrawledAt":"2026-03-01T03:53:00.746Z","lastIndexedAt":null,"nextCrawlAt":"2026-03-02T03:53:00.746Z","lastVerifiedAt":null,"highlights":[{"version":"0.1.0","createdAt":"2026-02-09T02:19:25.919Z","changelog":"Initial release of EachLabs Voice & Audio skill. - Provides text-to-speech, speech-to-text, voice conversion, and audio processing via EachLabs AI models. - Supports ElevenLabs TTS, Whisper transcription (including diarization), RVC voice conversion, and audio utilities. - Includes detailed model list, API usage instructions, and authentication steps. - Sample curl commands demonstrate predictions for TTS, transcription, diarization, voice conversion, and merging audio with video. - Lists available ElevenLabs voice IDs and references for model parameters.","fileCount":3,"zipByteSize":10160}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install kn7daknf647gdfdwzpyq5scnvn80tq7d:eachlabs-voice-audio","setupComplexity":"low","setupSteps":["Install using `clawhub skill install kn7daknf647gdfdwzpyq5scnvn80tq7d:eachlabs-voice-audio` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/eftalyurtseven/eachlabs-voice-audio before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-eftalyurtseven-eachlabs-voice-audio/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-eftalyurtseven-eachlabs-voice-audio/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-eftalyurtseven-eachlabs-voice-audio/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-eftalyurtseven-eachlabs-voice-audio/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-eftalyurtseven-eachlabs-voice-audio/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-eftalyurtseven-eachlabs-voice-audio/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-09T12:50:22.941Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-eftalyurtseven-eachlabs-voice-audio/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-eftalyurtseven-eachlabs-voice-audio/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-eftalyurtseven-eachlabs-voice-audio/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-eftalyurtseven-eachlabs-voice-audio/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":null},"readme":"Skill: Eachlabs Voice Audio\n\nOwner: eftalyurtseven\n\nSummary: Text-to-speech, speech-to-text, voice conversion, and audio processing using EachLabs AI models. Supports ElevenLabs TTS, Whisper transcription with diarization, and RVC voice conversion. Use when the user needs TTS, transcription, or voice conversion.\n\nTags: latest:0.1.0\n\nVersion history:\n\nv0.1.0 | 2026-02-09T02:19:25.919Z | auto\n\nInitial release of EachLabs Voice & Audio skill.\n\n- Provides text-to-speech, speech-to-text, voice conversion, and audio processing via EachLabs AI models.\n- Supports ElevenLabs TTS, Whisper transcription (including diarization), RVC voice conversion, and audio utilities.\n- Includes detailed model list, API usage instructions, and authentication steps.\n- Sample curl commands demonstrate predictions for TTS, transcription, diarization, voice conversion, and merging audio with video.\n- Lists available ElevenLabs voice IDs and references for model parameters.\n\nArchive index:\n\nArchive v0.1.0: 3 files, 10160 bytes\n\nFiles: references/MODELS.md (32175b), SKILL.md (6891b), _meta.json (139b)\n\nFile v0.1.0:SKILL.md\n\n---\nname: eachlabs-voice-audio\ndescription: Text-to-speech, speech-to-text, voice conversion, and audio processing using EachLabs AI models. Supports ElevenLabs TTS, Whisper transcription with diarization, and RVC voice conversion. Use when the user needs TTS, transcription, or voice conversion.\nmetadata:\n  author: eachlabs\n  version: \"1.0\"\n---\n\n# EachLabs Voice & Audio\n\nText-to-speech, speech-to-text transcription, voice conversion, and audio utilities via the EachLabs Predictions API.\n\n## Authentication\n\n```\nHeader: X-API-Key: <your-api-key>\n```\n\nSet the `EACHLABS_API_KEY` environment variable. Get your key at [eachlabs.ai](https://eachlabs.ai).\n\n## Available Models\n\n### Text-to-Speech\n\n| Model | Slug | Best For |\n|-------|------|----------|\n| ElevenLabs TTS | `elevenlabs-text-to-speech` | High quality TTS |\n| ElevenLabs TTS w/ Timestamps | `elevenlabs-text-to-speech-with-timestamp` | TTS with word timing |\n| ElevenLabs Text to Dialogue | `elevenlabs-text-to-dialogue` | Multi-speaker dialogue |\n| ElevenLabs Sound Effects | `elevenlabs-sound-effects` | Sound effect generation |\n| ElevenLabs Voice Design v2 | `elevenlabs-voice-design-v2` | Custom voice design |\n| Kling V1 TTS | `kling-v1-tts` | Kling text-to-speech |\n| Kokoro 82M | `kokoro-82m` | Lightweight TTS |\n| Play AI Dialog | `play-ai-text-to-speech-dialog` | Dialog TTS |\n| Stable Audio 2.5 | `stable-audio-2-5-text-to-audio` | Text to audio |\n\n### Speech-to-Text\n\n| Model | Slug | Best For |\n|-------|------|----------|\n| ElevenLabs Scribe v2 | `elevenlabs-speech-to-text-scribe-v2` | Best quality transcription |\n| ElevenLabs STT | `elevenlabs-speech-to-text` | Standard transcription |\n| Wizper with Timestamp | `wizper-with-timestamp` | Timestamped transcription |\n| Wizper | `wizper` | Basic transcription |\n| Whisper | `whisper` | Open-source transcription |\n| Whisper Diarization | `whisper-diarization` | Speaker identification |\n| Incredibly Fast Whisper | `incredibly-fast-whisper` | Fastest transcription |\n\n### Voice Conversion & Cloning\n\n| Model | Slug | Best For |\n|-------|------|----------|\n| RVC v2 | `rvc-v2` | Voice conversion |\n| Train RVC | `train-rvc` | Train custom voice model |\n| ElevenLabs Voice Clone | `elevenlabs-voice-clone` | Voice cloning |\n| ElevenLabs Voice Changer | `elevenlabs-voice-changer` | Voice transformation |\n| ElevenLabs Voice Design v3 | `elevenlabs-voice-design-v3` | Advanced voice design |\n| ElevenLabs Dubbing | `elevenlabs-dubbing` | Video dubbing |\n| Chatterbox S2S | `chatterbox-speech-to-speech` | Speech to speech |\n| Open Voice | `openvoice` | Open-source voice clone |\n| XTTS v2 | `xtts-v2` | Multi-language voice clone |\n| Stable Audio 2.5 Inpaint | `stable-audio-2-5-inpaint` | Audio inpainting |\n| Stable Audio 2.5 A2A | `stable-audio-2-5-audio-to-audio` | Audio transformation |\n| Audio Trimmer | `audio-trimmer-with-fade` | Audio trimming with fade |\n\n### Audio Utilities\n\n| Model | Slug | Best For |\n|-------|------|----------|\n| FFmpeg Merge Audio Video | `ffmpeg-api-merge-audio-video` | Merge audio with video |\n| Toolkit Video Convert | `toolkit` | Video/audio conversion |\n\n## Prediction Flow\n\n1. **Check model** `GET https://api.eachlabs.ai/v1/model?slug=<slug>` — validates the model exists and returns the `request_schema` with exact input parameters. Always do this before creating a prediction to ensure correct inputs.\n2. **POST** `https://api.eachlabs.ai/v1/prediction` with model slug, version `\"0.0.1\"`, and input matching the schema\n3. **Poll** `GET https://api.eachlabs.ai/v1/prediction/{id}` until status is `\"success\"` or `\"failed\"`\n4. **Extract** the output from the response\n\n## Examples\n\n### Text-to-Speech with ElevenLabs\n\n```bash\ncurl -X POST https://api.eachlabs.ai/v1/prediction \\\n  -H \"Content-Type: application/json\" \\\n  -H \"X-API-Key: $EACHLABS_API_KEY\" \\\n  -d '{\n    \"model\": \"elevenlabs-text-to-speech\",\n    \"version\": \"0.0.1\",\n    \"input\": {\n      \"text\": \"Welcome to our product demo. Today we will walk through the key features.\",\n      \"voice_id\": \"EXAVITQu4vr4xnSDxMaL\",\n      \"model_id\": \"eleven_v3\",\n      \"stability\": 0.5,\n      \"similarity_boost\": 0.7\n    }\n  }'\n```\n\n### Transcription with ElevenLabs Scribe\n\n```bash\ncurl -X POST https://api.eachlabs.ai/v1/prediction \\\n  -H \"Content-Type: application/json\" \\\n  -H \"X-API-Key: $EACHLABS_API_KEY\" \\\n  -d '{\n    \"model\": \"elevenlabs-speech-to-text-scribe-v2\",\n    \"version\": \"0.0.1\",\n    \"input\": {\n      \"media_url\": \"https://example.com/recording.mp3\",\n      \"diarize\": true,\n      \"timestamps_granularity\": \"word\"\n    }\n  }'\n```\n\n### Transcription with Wizper (Whisper)\n\n```bash\ncurl -X POST https://api.eachlabs.ai/v1/prediction \\\n  -H \"Content-Type: application/json\" \\\n  -H \"X-API-Key: $EACHLABS_API_KEY\" \\\n  -d '{\n    \"model\": \"wizper-with-timestamp\",\n    \"version\": \"0.0.1\",\n    \"input\": {\n      \"audio_url\": \"https://example.com/audio.mp3\",\n      \"language\": \"en\",\n      \"task\": \"transcribe\",\n      \"chunk_level\": \"segment\"\n    }\n  }'\n```\n\n### Speaker Diarization with Whisper\n\n```bash\ncurl -X POST https://api.eachlabs.ai/v1/prediction \\\n  -H \"Content-Type: application/json\" \\\n  -H \"X-API-Key: $EACHLABS_API_KEY\" \\\n  -d '{\n    \"model\": \"whisper-diarization\",\n    \"version\": \"0.0.1\",\n    \"input\": {\n      \"file_url\": \"https://example.com/meeting.mp3\",\n      \"num_speakers\": 3,\n      \"language\": \"en\",\n      \"group_segments\": true\n    }\n  }'\n```\n\n### Voice Conversion with RVC v2\n\n```bash\ncurl -X POST https://api.eachlabs.ai/v1/prediction \\\n  -H \"Content-Type: application/json\" \\\n  -H \"X-API-Key: $EACHLABS_API_KEY\" \\\n  -d '{\n    \"model\": \"rvc-v2\",\n    \"version\": \"0.0.1\",\n    \"input\": {\n      \"input_audio\": \"https://example.com/vocals.wav\",\n      \"rvc_model\": \"CUSTOM\",\n      \"custom_rvc_model_download_url\": \"https://example.com/my-voice-model.zip\",\n      \"pitch_change\": 0,\n      \"output_format\": \"wav\"\n    }\n  }'\n```\n\n### Merge Audio with Video\n\n```bash\ncurl -X POST https://api.eachlabs.ai/v1/prediction \\\n  -H \"Content-Type: application/json\" \\\n  -H \"X-API-Key: $EACHLABS_API_KEY\" \\\n  -d '{\n    \"model\": \"ffmpeg-api-merge-audio-video\",\n    \"version\": \"0.0.1\",\n    \"input\": {\n      \"video_url\": \"https://example.com/video.mp4\",\n      \"audio_url\": \"https://example.com/narration.mp3\",\n      \"start_offset\": 0\n    }\n  }'\n```\n\n## ElevenLabs Voice IDs\n\nThe `elevenlabs-text-to-speech` model supports these voice IDs. Pass the raw ID string:\n\n| Voice ID | Notes |\n|----------|-------|\n| `EXAVITQu4vr4xnSDxMaL` | Default voice |\n| `9BWtsMINqrJLrRacOk9x` | — |\n| `CwhRBWXzGAHq8TQ4Fs17` | — |\n| `FGY2WhTYpPnrIDTdsKH5` | — |\n| `JBFqnCBsd6RMkjVDRZzb` | — |\n| `N2lVS1w4EtoT3dr4eOWO` | — |\n| `TX3LPaxmHKxFdv7VOQHJ` | — |\n| `XB0fDUnXU5powFXDhCwa` | — |\n| `onwK4e9ZLuTAKqWW03F9` | — |\n| `pFZP5JQG7iQjIQuC4Bku` | — |\n\n## Parameter Reference\n\nSee [references/MODELS.md](references/MODELS.md) for complete parameter details for each model.\n\nFile v0.1.0:_meta.json\n\n{\n  \"ownerId\": \"kn7daknf647gdfdwzpyq5scnvn80tq7d\",\n  \"slug\": \"eachlabs-voice-audio\",\n  \"version\": \"0.1.0\",\n  \"publishedAt\": 1770603565919\n}\n\nFile v0.1.0:references/MODELS.md\n\n# Voice & Audio Models Reference\n\nComplete parameter reference for all voice and audio models. All models use version `0.0.1`.\n\n# Text To Voice\n\n---\n\n## Mureka | Create Podcast\n\n**Slug:** `mureka-create-podcast`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `conversations` | array | No | — | — | Speaker limit: This model supports exactly 2 speakers (two Voice IDs). Requests with more than two speakers are not s... |\n\n---\n\n## Mureka | Stem Song\n\n**Slug:** `mureka-stem-song`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `url` | string | Yes | — | — | — |\n\n---\n\n## ElevenLabs | Text to Speech with Timestamp\n\n**Slug:** `elevenlabs-text-to-speech-with-timestamp`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `text` | string | Yes | — | — | — |\n| `model_id` | string | No | `\"eleven_multilingual_v2\"` | eleven_multilingual_v2,eleven_flash_v2_5,eleven_turbo_v2_5,eleven_turbo_v2,el... | — |\n| `voice_id` | string | Yes | — | {\"9BWtsMINqrJLrRacOk9x\":{\"title\":\"Aria\",\"audio\":\"https://storage.googleapis.c... | Select a voice while using the web/ui, or send only the raw ElevenLabs voice_id string (e.g. \"EXAVITQu4vr4xnSDxMaL\") ... |\n| `language_code` | string | No | — | — | Language code (ISO 639-1) used to enforce a language for the model and text normalization. If the model does not supp... |\n| `stability` | number | No | `\"0.5\"` | — | — |\n| `use_speaker_boost` | boolean | No | — | — | — |\n| `similarity_boost` | number | No | `\"0.7\"` | — | — |\n| `style` | number | No | — | — | — |\n| `speed` | number | No | — | — | — |\n| `seed` | integer | No | — | — | — |\n| `previous_text` | string | No | — | — | — |\n| `next_text` | string | No | — | — | — |\n| `apply_text_normalization` | string | No | `\"auto\"` | auto,on,off | — |\n| `apply_language_text_normalization` | boolean | No | — | — | — |\n\n---\n\n## Minimax Music v2\n\n**Slug:** `minimax-music-v2`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `audio_setting` | object | No | — | — | — |\n| `prompt` | string | Yes | — | — | A description of the music, specifying style, mood, and scenario. 10-300 characters. |\n| `lyrics_prompt` | string | Yes | — | — | Lyrics of the song. Use n to separate lines. You may add structure tags like [Intro], [Verse], [Chorus], [Bridge], [O... |\n\n---\n\n## Elevenlabs Text to Dialogue\n\n**Slug:** `elevenlabs-text-to-dialogue`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `inputs` | array | Yes | — | — | — |\n| `model_id` | string | No | `\"eleven_v3\"` | eleven_v3 | — |\n| `stability` | number | No | `\"0.5\"` | — | — |\n| `language_code` | string | No | — | — | Language code (ISO 639-1) used to enforce a language for the model and text normalization. If the model does not supp... |\n| `seed` | integer | No | — | — | If specified, our system will make a best effort to sample deterministically, such that repeated requests with the sa... |\n| `apply_text_normalization` | string | No | `\"auto\"` | auto,on,off | This parameter controls text normalization with three modes: ‘auto’, ‘on’, and ‘off’. When set to ‘auto’, the system ... |\n\n---\n\n## Elevenlabs Voice Design V2\n\n**Slug:** `elevenlabs-voice-design-v2`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `voice_description` | string | Yes | — | — | Description to use for the created voice. |\n| `text` | string | No | — | — | Text to generate, text length has to be between 100 and 1000. |\n| `auto_generate_text` | boolean | No | `\"false\"` | — | Whether to automatically generate a text suitable for the voice description. |\n| `model_id` | string | No | `\"eleven_multilingual_ttv_v2\"` | eleven_multilingual_ttv_v2 | — |\n| `loudness` | number | No | `\"0.5\"` | — | Controls the volume level of the generated voice. -1 is quietest, 1 is loudest, 0 corresponds to roughly -24 LUFS. |\n| `seed` | integer | No | — | — | Random number that controls the voice generation. Same seed with same inputs produces same voice. max: 2147483646 |\n| `guidance_scale` | integer | No | `\"5\"` | — | Controls how closely the AI follows the prompt. Lower numbers give the AI more freedom to be creative, while higher n... |\n| `quality` | number | No | `\"0\"` | — | Higher quality results in better voice output but less variety. |\n\n---\n\n## Kling V1 | Text to Speech\n\n**Slug:** `kling-v1-tts`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `text` | string | Yes | — | — | The text to be converted to speech Max 120 character |\n| `voice_id` | string | No | `\"genshin_vindi2\"` | genshin_vindi2,zhinen_xuesheng,AOT,ai_shatang,genshin_klee2,genshin_kirara,ai... | The voice ID to use for speech synthesis |\n| `voice_speed` | number | No | `\"1\"` | — | Rate of speech |\n\n---\n\n## Minimax Music | V1.5\n\n**Slug:** `minimax-music-v1-5`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `prompt` | string | Yes | — | — | Lyrics, supports [intro][verse][chorus][bridge][outro] sections. 10-600 characters. |\n| `lyrics_prompt` | string | Yes | — | — | Control music generation. 10-3000 characters. |\n\n---\n\n## Stable Audio 2.5 | Text to Audio\n\n**Slug:** `stable-audio-2-5-text-to-audio`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `prompt` | string | Yes | — | — | The prompt to generate audio from |\n| `seconds_total` | integer | No | `\"190\"` | — | The duration of the audio clip to generate |\n| `num_inference_steps` | integer | No | `\"8\"` | — | The number of steps to denoise the audio for |\n| `guidance_scale` | integer | No | `\"1\"` | — | How strictly the diffusion process adheres to the prompt text (higher values make your audio closer to your prompt). |\n| `seed` | integer | No | — | — | — |\n\n---\n\n## Play AI | Text to Speech | Dialog\n\n**Slug:** `play-ai-text-to-speech-dialog`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `input` | string | No | — | — | The dialogue text with turn prefixes to distinguish speakers. |\n| `voices` | array | Yes | — | — | The unique ID of a PlayHT or Cloned Voice, or a name from the available presets. |\n| `response_format` | string | No | `\"url\"` | url | The format of the response. |\n| `seed` | integer | No | — | — | An integer number greater than or equal to 0. If equal to null or not provided, a random seed will be used. Useful to... |\n\n---\n\n## ElevenLabs | Sound Effects\n\n**Slug:** `elevenlabs-sound-effects`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `text` | string | Yes | — | — | — |\n| `duration_seconds` | integer | No | `\"1\"` | — | The duration of the sound which will be generated in seconds. Must be at least 0.5 and at most 22. If set to None we ... |\n| `prompt_influence` | number | No | `\"0.3\"` | — | A higher prompt influence makes your generation follow the prompt more closely while also making generations less var... |\n\n---\n\n## ElevenLabs | Text to Speech\n\n**Slug:** `elevenlabs-text-to-speech`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `text` | string | Yes | — | — | — |\n| `model_id` | string | Yes | `\"eleven_v3\"` | eleven_multilingual_v2,eleven_flash_v2_5,eleven_turbo_v2_5,eleven_turbo_v2,el... | — |\n| `voice_id` | string | Yes | — | {\"9BWtsMINqrJLrRacOk9x\":{\"title\":\"Aria\",\"audio\":\"https://storage.googleapis.c... | Select a voice while using the web/ui, or send only the raw ElevenLabs voice_id string (e.g. \"EXAVITQu4vr4xnSDxMaL\") ... |\n| `use_speaker_boost` | boolean | No | `\"false\"` | — | Boost the similarity of the synthesized speech and the voice at the cost of some generation speed. |\n| `style` | number | No | `\"0\"` | — | High values are recommended if the style of the speech should be exaggerated compared to the uploaded audio. Higher v... |\n| `similarity_boost` | number | No | `\"0.7\"` | — | High enhancement boosts overall voice clarity and target speaker similarity. Very high values can cause artifacts, so... |\n| `stability` | number | No | `\"0.5\"` | — | Increasing stability will make the voice more consistent between re-generations, but it can also make it sounds a bit... |\n| `seed` | integer | No | — | — | If specified, our system will make a best effort to sample deterministically, such that repeated requests with the sa... |\n\n---\n\n## Kokoro 82M\n\n**Slug:** `kokoro-82m`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `voice` | string | No | `\"af\"` | af,af_bella,af_sarah,am_adam,am_michael,bf_emma,bf_isabella,bm_george,bm_lewi... | An enumeration. |\n| `speed` | number | No | `\"1\"` | — | Speech speed multiplier (0.5 = half speed, 2.0 = double speed) |\n| `text` | string | Yes | — | — | Text input (long text is automatically split into smaller chunks) |\n\n---\n\n# Voice To Text\n\n---\n\n## ElevenLabs | Speech to Text Scribe V2\n\n**Slug:** `elevenlabs-speech-to-text-scribe-v2`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `media_url` | string | Yes | — | — | — |\n| `language_code` | string | No | — | — | An ISO-639-1 or ISO-639-3 language_code corresponding to the language of the audio file. Can sometimes improve transc... |\n| `tag_audio_events` | boolean | No | `\"false\"` | — | Whether to tag audio events like (laughter), (footsteps), etc. in the transcription. |\n| `num_speakers` | integer | No | — | — | The maximum amount of speakers talking in the uploaded file. Can help with predicting who speaks when. The maximum am... |\n| `timestamps_granularity` | string | No | `\"word\"` | none,word,character | The granularity of the timestamps in the transcription. ‘word’ provides word-level timestamps and ‘character’ provide... |\n| `diarize` | boolean | No | `\"false\"` | — | Whether to annotate which speaker is currently talking in the uploaded file. |\n| `diarization_threshold` | number | No | — | — | Diarization threshold to apply during speaker diarization. A higher value means there will be a lower chance of one s... |\n| `temperature` | number | No | — | — | Controls the randomness of the transcription output. Accepts values between 0.0 and 2.0, where higher values result i... |\n| `seed` | integer | No | — | — | — |\n| `use_multi_channel` | boolean | No | `\"false\"` | — | Whether the audio file contains multiple channels where each channel contains a single speaker. When enabled, each ch... |\n\n---\n\n## Wizper with Timestamp\n\n**Slug:** `wizper-with-timestamp`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `audio_url` | string | Yes | — | — | URL of the audio file to transcribe. Supported formats: mp3, mp4, mpeg, mpga, m4a, wav or webm. |\n| `task` | string | No | `\"transcribe\"` | transcribe, translate | Task to perform on the audio file. Either transcribe or translate. |\n| `language` | string | No | — | af,am,ar,as,az,ba,be,bg,bn,bo,br,bs,ca,cs,cy,da,de,el,en,es,et,eu,fa,fi,fo,fr... | Language of the audio file. If translate is selected as the task, the audio will be translated to English, regardless... |\n| `chunk_level` | string | No | `\"segment\"` | — | Level of the chunks to return. |\n| `max_segment_len` | integer | No | `\"29\"` | — | Maximum speech segment duration in seconds before splitting. |\n| `merge_chunks` | boolean | No | `\"True\"` | — | Whether to merge consecutive chunks. When enabled, chunks are merged if their combined duration does not exceed max_s... |\n| `version` | string | No | `\"3\"` | — | Version of the model to use. All of the models are the Whisper large variant. |\n\n---\n\n## Whisper Diarization\n\n**Slug:** `whisper-diarization`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `file_string` | string | No | — | — | Either provide: Base64 encoded audio file, |\n| `file_url` | string | No | — | — | Or provide: A direct audio file URL |\n| `file` | string | Yes | — | — | Or an audio file |\n| `group_segments` | boolean | No | `\"True\"` | — | Group segments of same speaker shorter apart than 2 seconds |\n| `num_speakers` | integer | No | `\"2\"` | — | Number of speakers, leave empty to autodetect. |\n| `translate` | boolean | No | `\"false\"` | — | Translate the speech into English. |\n| `language` | string | No | `\"en\"` | — | Language of the spoken words as a language code like 'en'. Leave empty to auto detect language. |\n| `prompt` | string | No | — | — | Vocabulary: provide names, acronyms and loanwords in a list. Use punctuation for best accuracy. |\n\n---\n\n## Whisper\n\n**Slug:** `whisper`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `audio_url` | string | Yes | — | — | URL of the audio file to transcribe. Supported formats: mp3, mp4, mpeg, mpga, m4a, wav or webm. |\n| `task` | string | No | `\"transcribe\"` | transcribe,translate | Task to perform on the audio file. Either transcribe or translate. |\n| `language` | string | No | — | af,am,ar,as,az,ba,be,bg,bn,bo,br,bs,ca,cs,cy,da,de,el,en,es,et,eu,fa,fi,fo,fr... | Language of the audio file. If set to null, the language will be automatically detected. Defaults to null. If transla... |\n| `diarize` | boolean | No | `\"False\"` | — | Whether to diarize the audio file. Defaults to false. Setting to true will add costs proportional to diarization infe... |\n| `chunk_level` | string | No | `\"segment\"` | none,segment,word | Level of the chunks to return. Either none, segment or word. `none` would imply that all of the audio will be transcr... |\n| `version` | string | No | `\"3\"` | 3 | Version of the model to use. All of the models are the Whisper large variant. |\n| `batch_size` | integer | No | `\"64\"` | — | — |\n| `prompt` | string | No | — | — | Prompt to use for generation. Defaults to an empty string. |\n| `num_speakers` | integer | No | — | — | Number of speakers in the audio file. Defaults to null. If not provided, the number of speakers will be automatically... |\n\n---\n\n## Wizper\n\n**Slug:** `wizper`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `audio_url` | string | Yes | — | — | URL of the audio file to transcribe. Supported formats: mp3, mp4, mpeg, mpga, m4a, wav or webm. |\n| `task` | string | No | `\"transcribe\"` | transcribe,translate | Task to perform on the audio file. Either transcribe or translate. |\n| `language` | string | No | — | af,am,ar,as,az,ba,be,bg,bn,bo,br,bs,ca,cs,cy,da,de,el,en,es,et,eu,fa,fi,fo,fr... | Language of the audio file. If translate is selected as the task, the audio will be translated to English, regardless... |\n| `chunk_level` | string | No | `\"segment\"` | — | Level of the chunks to return. |\n| `version` | string | No | `\"3\"` | — | Version of the model to use. All of the models are the Whisper large variant. |\n\n---\n\n## ElevenLabs | Speech to Text\n\n**Slug:** `elevenlabs-speech-to-text`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `diarization_threshold` | number | No | `\"0.22\"` | — | Diarization threshold to apply during speaker diarization. A higher value means there will be a lower chance of one s... |\n| `audio_url` | string | Yes | — | — | — |\n| `model_id` | string | Yes | `\"scribe_v1\"` | scribe_v1,scribe_v1_experimental | — |\n| `language_code` | string | No | — | — | An ISO-639-1 or ISO-639-3 language_code corresponding to the language of the audio file. Can sometimes improve transc... |\n| `tag_audio_events` | boolean | No | — | — | — |\n| `num_speakers` | integer | No | — | — | — |\n| `timestamp_granularity` | string | No | `\"none\"` | none,word,character | — |\n| `diarize` | boolean | No | `\"false\"` | — | Whether to annotate which speaker is currently talking in the uploaded file. |\n\n---\n\n## Incredibly Fast Whisper\n\n**Slug:** `incredibly-fast-whisper`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `audio` | string | Yes | `\"https://storage.googleapis.com/magicpoint/inputs/fast-whisper-input.wav\"` | — | Audio refers to the sound or recording that is being analyzed or processed. |\n| `task` | string | No | `\"transcribe\"` | transcribe | Task defines the specific operation or activity the model is required to perform on the input audio. |\n| `language` | string | No | `\"None\"` | None,afrikaans,amharic,arabic,azerbaijani,belarusian,bosnian,breton,bulgarian... | Language refers to the specific language in which the input audio is provided. |\n| `batch_size` | integer | No | `\"24\"` | — | Batch size is the number of samples processed together in one iteration. |\n| `timestamp` | string | No | `\"chunk\"` | chunk,word | Timestamp denotes the specific time at which an event occurs in the audio. |\n| `diarise_audio` | boolean | No | — | — | Diarise audio involves splitting a conversation into segments based on who is speaking. |\n| `hf_token` | string | No | — | — | HF token is a special key used to authenticate and access resources on the Hugging Face platform. |\n\n---\n\n# Voice To Voice\n\n---\n\n## Mureka | Extend Song\n\n**Slug:** `mureka-extend-song`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `lyrics` | string | Yes | — | — | The lyrics to be extended. |\n| `extend_at` | integer | Yes | — | — | Extending start time (milliseconds). If greater than song duration, defaults to song duration. Valid range: [8000,420... |\n| `upload_audio_id` | string | No | — | — | Upload ID of the song to be extended, generated by the files/upload API (purpose: audio). Only supports songs generat... |\n| `song_id` | string | No | — | — | Song ID for extending, generated by the song/generate API. Mutually exclusive with the upload_audio_id parameter. |\n\n---\n\n## Rvc v2\n\n**Slug:** `rvc-v2`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `input_audio` | string | Yes | — | — | Upload your audio file here. |\n| `rvc_model` | string | No | `\"CUSTOM\"` | Obama,Trump,Sandy,Rogan,CUSTOM | An enumeration. |\n| `custom_rvc_model_download_url` | string | No | — | — | URL to download a custom RVC model. If provided, the model will be downloaded (if it doesn't already exist) and used ... |\n| `pitch_change` | number | No | `\"0\"` | — | Adjust pitch of AI vocals in semitones. Use positive values to increase pitch, negative to decrease. |\n| `index_rate` | number | No | `\"0.5\"` | — | Control how much of the AI's accent to leave in the vocals. |\n| `filter_radius` | integer | No | `\"3\"` | — | If >=3: apply median filtering to the harvested pitch results. |\n| `rms_mix_rate` | number | No | `\"0.25\"` | — | Control how much to use the original vocal's loudness (0) or a fixed loudness (1). |\n| `f0_method` | string | No | `\"rmvpe\"` | rmvpe,mangio-crepe | An enumeration. |\n| `crepe_hop_length` | integer | No | `\"128\"` | — | When `f0_method` is set to `mangio-crepe`, this controls how often it checks for pitch changes in milliseconds. |\n| `protect` | number | No | `\"0.33\"` | — | Control how much of the original vocals' breath and voiceless consonants to leave in the AI vocals. Set 0.5 to disable. |\n| `output_format` | string | No | `\"wav\"` | mp3,wav | An enumeration. |\n\n---\n\n## Elevenlabs Voice Design V3\n\n**Slug:** `elevenlabs-voice-design-v3`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `voice_description` | string | Yes | — | — | Description to use for the created voice. |\n| `reference_audio_url` | string | Yes | — | — | — |\n| `prompt_strength` | number | No | `\"0\"` | — | Controls the balance of prompt versus reference audio when generating voice samples. 0 means almost no prompt influen... |\n| `text` | string | No | — | — | Text to generate, text length has to be between 100 and 1000. |\n| `auto_generate_text` | boolean | No | `\"false\"` | — | Whether to automatically generate a text suitable for the voice description. |\n| `guidance_scale` | integer | No | `\"5\"` | — | Controls how closely the AI follows the prompt. Lower numbers give the AI more freedom to be creative, while higher n... |\n| `loudness` | number | No | `\"0.5\"` | — | Controls the volume level of the generated voice. -1 is quietest, 1 is loudest, 0 corresponds to roughly -24 LUFS. |\n| `seed` | integer | No | — | — | Random number that controls the voice generation. Same seed with same inputs produces same voice. max: 2147483646 |\n| `model_id` | string | No | `\"eleven_ttv_v3\"` | eleven_ttv_v3 | — |\n\n---\n\n## Chatterbox | Speech to Speech\n\n**Slug:** `chatterbox-speech-to-speech`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `source_audio_url` | string | Yes | — | — | — |\n| `target_voice_audio_url` | string | No | — | — | Optional URL to an audio file to use as a reference for the generated speech. If provided, the model will try to matc... |\n\n---\n\n## Stable Audio 2.5 | Audio to Audio\n\n**Slug:** `stable-audio-2-5-audio-to-audio`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `prompt` | string | Yes | — | — | The prompt to guide the audio generation |\n| `audio_url` | string | Yes | — | — | The audio clip to transform |\n| `strength` | number | No | `\"0.8\"` | — | Sometimes referred to as denoising, this parameter controls how much influence the `audio_url` parameter has on the g... |\n| `num_inference_steps` | integer | No | `\"8\"` | — | The number of steps to denoise the audio for |\n| `total_seconds` | integer | No | — | — | The duration of the audio clip to generate. If not provided, it will be set to the duration of the input audio. |\n| `guidance_scale` | integer | No | `\"1\"` | — | How strictly the diffusion process adheres to the prompt text (higher values make your audio closer to your prompt). |\n| `seed` | integer | No | — | — | — |\n\n---\n\n## Stable Audio 2.5 | Inpaint\n\n**Slug:** `stable-audio-2-5-inpaint`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `prompt` | string | Yes | — | — | The prompt to guide the audio generation |\n| `audio_url` | string | Yes | — | — | The audio clip to inpaint |\n| `seconds_total` | integer | No | `\"190\"` | — | The duration of the audio clip to generate. If not provided, it will be set to the duration of the input audio. |\n| `guidance_scale` | integer | No | `\"1\"` | — | How strictly the diffusion process adheres to the prompt text (higher values make your audio closer to your prompt). |\n| `mask_start` | integer | No | `\"30\"` | — | The start point of the audio mask |\n| `mask_end` | integer | No | `\"190\"` | — | The end point of the audio mask |\n| `num_inference_steps` | integer | No | `\"8\"` | — | The number of steps to denoise the audio for |\n| `seed` | integer | No | — | — | — |\n\n---\n\n## Elevenlabs Voice Clone\n\n**Slug:** `elevenlabs-voice-clone`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `name` | string | No | — | — | Enter voice name |\n| `files` | array | No | — | — | Upload your files |\n| `remove_background_noise` | boolean | No | — | — | — |\n| `description` | string | No | — | — | Defines the tone, style, and personality of the generated voice. |\n\n---\n\n## Audio Trimmer\n\n**Slug:** `audio-trimmer-with-fade`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `audio_file` | string | Yes | — | — | Input MP3 file |\n| `start_time` | integer | Yes | — | — | Start time in MMSS format, e.g., 130 for 1 minute and 30 seconds |\n| `end_time` | integer | Yes | — | — | End time in MMSS format, e.g., 625 for 6 minutes and 25 seconds |\n| `fade_out` | boolean | No | `\"false\"` | — | Apply fade out effect to the last 2.5 seconds |\n\n---\n\n## ElevenLabs | Voice Changer\n\n**Slug:** `elevenlabs-voice-changer`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `audio_url` | string | Yes | — | — | — |\n| `model_id` | string | No | `\"eleven_english_sts_v2\"` | eleven_english_sts_v2 | — |\n| `voice_id` | string | Yes | — | {\"9BWtsMINqrJLrRacOk9x\":{\"title\":\"Aria\",\"audio\":\"https://storage.googleapis.c... | Select a voice while using the web/ui, or send only the raw ElevenLabs voice_id string (e.g. \"EXAVITQu4vr4xnSDxMaL\") ... |\n\n---\n\n## ElevenLabs | Dubbing\n\n**Slug:** `elevenlabs-dubbing`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `source_url` | string | Yes | — | — | URL of the source video/audio file. |\n| `source_lang` | string | No | — | — | — |\n| `target_lang` | string | Yes | — | — | — |\n\n---\n\n## XTTS\n\n**Slug:** `xtts-v2`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `text` | string | No | `\"Hello, you are now at Eachlabs AI. If you need any support, just contact us.\"` | — | This is the written input that you want to be converted into spoken words. |\n| `speaker` | string | Yes | — | — | This determines the specific voice or persona that will speak the provided text. |\n| `language` | string | No | `\"en\"` | en,es,fr,de,it,pt,pl,tr,ru,nl,cs,ar,zh,hu,ko,hi | This refers to the choice of language for the text-to-speech synthesis. |\n| `cleanup_voice` | boolean | No | `\"true\"` | — | This option helps in refining and improving the quality of the generated speech. |\n\n---\n\n## Open Voice\n\n**Slug:** `openvoice`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `audio` | string | Yes | `\"https://cdn.eachlabs.ai/ipfs/3FA9ck5a0woAEJulGzeg3CJkcPVBcLCfdixBNrAxA8ediDrlA/out.wav\"` | — | Input reference audio |\n| `text` | string | No | `\"Did you ever hear a folk tale about a giant turtle?\"` | — | Input text |\n| `language` | string | No | `\"EN_NEWEST\"` | EN_NEWEST,EN,ES,FR,ZH,JP,KR | An enumeration. |\n| `speed` | number | No | `\"1\"` | — | Set speed scale of the output audio |\n\n---\n\n## Voice Changer\n\n**Slug:** `realistic-voice-cloning`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `song_input` | string | Yes | — | — | The original song file provided as input. |\n| `rvc_model` | string | No | `\"Squidward\"` | Squidward,MrKrabs,Plankton,Drake,Vader,Trump,Biden,Obama,Guitar,Voilin,CUSTOM | The specific RVC model used for the function. |\n| `custom_rvc_model_download_url` | string | No | — | — | URL to download a custom RVC model. |\n| `pitch_change` | string | No | `\"no-change\"` | no-change,male-to-female,female-to-male | Alteration in the pitch of the audio. |\n| `index_rate` | number | No | `\"0.5\"` | — | The frequency at which indexing is performed. |\n| `filter_radius` | integer | No | `\"3\"` | — | Range of frequencies affected by the filter. |\n| `rms_mix_rate` | number | No | `\"0.25\"` | — | Ratio of root mean square levels for mixing. |\n| `pitch_detection_algorithm` | string | No | `\"rmvpe\"` | rmvpe,mangio-crepe | The method used to detect the pitch of the vocals and instruments. |\n| `crepe_hop_length` | integer | No | `\"128\"` | — | The step size for the pitch detection process using the CREPE algorithm. |\n| `protect` | number | No | `\"0.33\"` | — | Safety or backup mechanism for the original audio. |\n| `main_vocals_volume_change` | number | No | `\"0\"` | — | Adjustment of the main vocals' volume. |\n| `backup_vocals_volume_change` | number | No | `\"0\"` | — | Adjustment of the backup vocals' volume. |\n| `instrumental_volume_change` | number | No | `\"0\"` | — | Change in the volume of the instrumental part of the song. |\n| `pitch_change_all` | number | No | `\"0\"` | — | Modification of the pitch for all elements of the song. |\n| `reverb_size` | number | No | `\"0.15\"` | — | The perceived size of the reverb effect location. |\n| `reverb_wetness` | number | No | `\"0.2\"` | — | The amount of the reverb effect applied (wet signal). |\n| `reverb_dryness` | number | No | `\"0.8\"` | — | The degree to which the direct sound is present without reverb (dry signal). |\n| `reverb_damping` | number | No | `\"0.7\"` | — | The reduction of high frequencies in the reverb effect. |\n| `output_format` | string | No | `\"mp3\"` | mp3,wav | The format of the resulting audio file. |\n\n---\n\n## Toolkit - Video Convert\n\n**Slug:** `toolkit`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `task` | string | Yes | — | convert_input_to_mp4,convert_input_to_gif,extract_video_audio_as_mp3,zipped_f... | An enumeration. |\n| `input_file` | string | Yes | — | — | File – zip, image or video to process |\n| `fps` | integer | No | `\"1\"` | — | frames per second, if relevant. Use 0 to keep original fps (or use default). Converting to GIF defaults to 12fps |\n\n---\n\n## Ffmpeg Api | Merge Audio Video\n\n**Slug:** `ffmpeg-api-merge-audio-video`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `video_url` | string | Yes | — | — | URL of the video file to use as the video track |\n| `audio_url` | string | Yes | — | — | URL of the audio file to use as the audio track |\n| `start_offset` | number | No | `\"0\"` | — | Offset in seconds for when the audio should start relative to the video |\n\n---","readmeExcerpt":"Skill: Eachlabs Voice Audio Owner: eftalyurtseven Summary: Text-to-speech, speech-to-text, voice conversion, and audio processing using EachLabs AI models. Supports ElevenLabs TTS, Whisper transcription with diarization, and RVC voice conversion. Use when the user needs TTS, transcription, or voice conversion. Tags: latest:0.1.0 Version history: v0.1.0 | 2026-02-09T02:19:25.919Z | auto Initial release of EachLabs Voi","codeSnippets":[],"executableExamples":[{"language":"text","snippet":"Header: X-API-Key: <your-api-key>"},{"language":"bash","snippet":"curl -X POST https://api.eachlabs.ai/v1/prediction \\\n  -H \"Content-Type: application/json\" \\\n  -H \"X-API-Key: $EACHLABS_API_KEY\" \\\n  -d '{"},{"language":"bash","snippet":"curl -X POST https://api.eachlabs.ai/v1/prediction \\\n  -H \"Content-Type: application/json\" \\\n  -H \"X-API-Key: $EACHLABS_API_KEY\" \\\n  -d '{\n    \"model\": \"elevenlabs-text-to-speech\",\n    \"version\": \"0.0.1\",\n    \"input\": {\n      \"text\": \"Welcome to our product demo. Today we will walk through the key features.\",\n      \"voice_id\": \"EXAVITQu4vr4xnSDxMaL\",\n      \"model_id\": \"eleven_v3\",\n      \"stability\": 0.5,\n      \"similarity_boost\": 0.7\n    }\n  }'"},{"language":"bash","snippet":"curl -X POST https://api.eachlabs.ai/v1/prediction \\\n  -H \"Content-Type: application/json\" \\\n  -H \"X-API-Key: $EACHLABS_API_KEY\" \\\n  -d '{"},{"language":"bash","snippet":"curl -X POST https://api.eachlabs.ai/v1/prediction \\\n  -H \"Content-Type: application/json\" \\\n  -H \"X-API-Key: $EACHLABS_API_KEY\" \\\n  -d '{\n    \"model\": \"elevenlabs-speech-to-text-scribe-v2\",\n    \"version\": \"0.0.1\",\n    \"input\": {\n      \"media_url\": \"https://example.com/recording.mp3\",\n      \"diarize\": true,\n      \"timestamps_granularity\": \"word\"\n    }\n  }'"},{"language":"bash","snippet":"curl -X POST https://api.eachlabs.ai/v1/prediction \\\n  -H \"Content-Type: application/json\" \\\n  -H \"X-API-Key: $EACHLABS_API_KEY\" \\\n  -d '{"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: eachlabs-voice-audio\ndescription: Text-to-speech, speech-to-text, voice conversion, and audio processing using EachLabs AI models. Supports ElevenLabs TTS, Whisper transcription with diarization, and RVC voice conversion. Use when the user needs TTS, transcription, or voice conversion.\nmetadata:\n  author: eachlabs\n  version: \"1.0\"\n---\n\n# EachLabs Voice & Audio\n\nText-to-speech, speech-to-text transcription, voice conversion, and audio utilities via the EachLabs Predictions API.\n\n## Authentication\n\n```\nHeader: X-API-Key: <your-api-key>\n```\n\nSet the `EACHLABS_API_KEY` environment variable. Get your key at [eachlabs.ai](https://eachlabs.ai).\n\n## Available Models\n\n### Text-to-Speech\n\n| Model | Slug | Best For |\n|-------|------|----------|\n| ElevenLabs TTS | `elevenlabs-text-to-speech` | High quality TTS |\n| ElevenLabs TTS w/ Timestamps | `elevenlabs-text-to-speech-with-timestamp` | TTS with word timing |\n| ElevenLabs Text to Dialogue | `elevenlabs-text-to-dialogue` | Multi-speaker dialogue |\n| ElevenLabs Sound Effects | `elevenlabs-sound-effects` | Sound effect generation |\n| ElevenLabs Voice Design v2 | `elevenlabs-voice-design-v2` | Custom voice design |\n| Kling V1 TTS | `kling-v1-tts` | Kling text-to-speech |\n| Kokoro 82M | `kokoro-82m` | Lightweight TTS |\n| Play AI Dialog | `play-ai-text-to-speech-dialog` | Dialog TTS |\n| Stable Audio 2.5 | `stable-audio-2-5-text-to-audio` | Text to audio |\n\n### Speech-to-Text\n\n| Model | Slug | Best For |\n|-------|------|----------|\n| ElevenLabs Scribe v2 | `elevenlabs-speech-to-text-scribe-v2` | Best quality transcription |\n| ElevenLabs STT | `elevenlabs-speech-to-text` | Standard transcription |\n| Wizper with Timestamp | `wizper-with-timestamp` | Timestamped transcription |\n| Wizper | `wizper` | Basic transcription |\n| Whisper | `whisper` | Open-source transcription |\n| Whisper Diarization | `whisper-diarization` | Speaker identification |\n| Incredibly Fast Whisper | `incredibly-fast-whisper` | Fastest transcription |\n\n### Voice Conversion & Cloning\n\n| Model | Slug | Best For |\n|-------|------|----------|\n| RVC v2 | `rvc-v2` | Voice conversion |\n| Train RVC | `train-rvc` | Train custom voice model |\n| ElevenLabs Voice Clone | `elevenlabs-voice-clone` | Voice cloning |\n| ElevenLabs Voice Changer | `elevenlabs-voice-changer` | Voice transformation |\n| ElevenLabs Voice Design v3 | `elevenlabs-voice-design-v3` | Advanced voice design |\n| ElevenLabs Dubbing | `elevenlabs-dubbing` | Video dubbing |\n| Chatterbox S2S | `chatterbox-speech-to-speech` | Speech to speech |\n| Open Voice | `openvoice` | Open-source voice clone |\n| XTTS v2 | `xtts-v2` | Multi-language voice clone |\n| Stable Audio 2.5 Inpaint | `stable-audio-2-5-inpaint` | Audio inpainting |\n| Stable Audio 2.5 A2A | `stable-audio-2-5-audio-to-audio` | Audio transformation |\n| Audio Trimmer | `audio-trimmer-with-fade` | Audio trimming with fade |\n\n### Audio Utilities\n\n| Model | Slug | Best For |\n|-------|------|----------|\n| FFmpeg Merge Audio Video | "},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7daknf647gdfdwzpyq5scnvn80tq7d\",\n  \"slug\": \"eachlabs-voice-audio\",\n  \"version\": \"0.1.0\",\n  \"publishedAt\": 1770603565919\n}"},{"path":"references/MODELS.md","content":"# Voice & Audio Models Reference\n\nComplete parameter reference for all voice and audio models. All models use version `0.0.1`.\n\n# Text To Voice\n\n---\n\n## Mureka | Create Podcast\n\n**Slug:** `mureka-create-podcast`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `conversations` | array | No | — | — | Speaker limit: This model supports exactly 2 speakers (two Voice IDs). Requests with more than two speakers are not s... |\n\n---\n\n## Mureka | Stem Song\n\n**Slug:** `mureka-stem-song`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `url` | string | Yes | — | — | — |\n\n---\n\n## ElevenLabs | Text to Speech with Timestamp\n\n**Slug:** `elevenlabs-text-to-speech-with-timestamp`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `text` | string | Yes | — | — | — |\n| `model_id` | string | No | `\"eleven_multilingual_v2\"` | eleven_multilingual_v2,eleven_flash_v2_5,eleven_turbo_v2_5,eleven_turbo_v2,el... | — |\n| `voice_id` | string | Yes | — | {\"9BWtsMINqrJLrRacOk9x\":{\"title\":\"Aria\",\"audio\":\"https://storage.googleapis.c... | Select a voice while using the web/ui, or send only the raw ElevenLabs voice_id string (e.g. \"EXAVITQu4vr4xnSDxMaL\") ... |\n| `language_code` | string | No | — | — | Language code (ISO 639-1) used to enforce a language for the model and text normalization. If the model does not supp... |\n| `stability` | number | No | `\"0.5\"` | — | — |\n| `use_speaker_boost` | boolean | No | — | — | — |\n| `similarity_boost` | number | No | `\"0.7\"` | — | — |\n| `style` | number | No | — | — | — |\n| `speed` | number | No | — | — | — |\n| `seed` | integer | No | — | — | — |\n| `previous_text` | string | No | — | — | — |\n| `next_text` | string | No | — | — | — |\n| `apply_text_normalization` | string | No | `\"auto\"` | auto,on,off | — |\n| `apply_language_text_normalization` | boolean | No | — | — | — |\n\n---\n\n## Minimax Music v2\n\n**Slug:** `minimax-music-v2`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `audio_setting` | object | No | — | — | — |\n| `prompt` | string | Yes | — | — | A description of the music, specifying style, mood, and scenario. 10-300 characters. |\n| `lyrics_prompt` | string | Yes | — | — | Lyrics of the song. Use n to separate lines. You may add structure tags like [Intro], [Verse], [Chorus], [Bridge], [O... |\n\n---\n\n## Elevenlabs Text to Dialogue\n\n**Slug:** `elevenlabs-text-to-dialogue`\n\n| Parameter | Type | Required | Default | Options / Constraints | Description |\n|-----------|------|----------|---------|----------------------|-------------|\n| `inputs` | array | Yes | — | — | — |\n| `model_id` | string | No | `"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":1133,"uniquenessScore":42,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T12:50:22.942Z","emptyReason":null},"items":[{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-10T18:48:31.762Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}