{"id":"aa7b0343-9777-4b5c-a5c6-045eab1e3c3b","entityType":"agent","slug":"clawhub-gladiaio-gladia-documentation-auto","name":"gladia-documentation-auto","canonicalUrl":"https://www.xpersona.co/agent/clawhub-gladiaio-gladia-documentation-auto","canonicalPath":"/agent/clawhub-gladiaio-gladia-documentation-auto","generatedAt":"2026-10-10T21:43:13.555Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T15:44:38.092Z","emptyReason":null},"description":"Comprehensive Gladia speech-to-text reference auto-synced from docs.gladia.io. Use as a general-purpose fallback when other specialized skills don't match, or when the user needs a broad overview of Gladia capabilities, endpoints, decision guidance, or workflows. Always prefer the official SDK; fall back to raw REST/WebSocket only when SDK cannot satisfy the requirement.","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.4K downloads reported by the source. Last updated 10/10/2026.","installCommand":"clawhub skill install s170451ybbxgxcc4es3g96b4fs8823td:gladia-documentation-auto","sourceUrl":"https://clawhub.ai/gladiaio/gladia-documentation-auto","homepage":"https://clawhub.ai/gladiaio/skills/gladia-documentation-auto","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/gladiaio/gladia-documentation-auto","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/gladiaio/skills/gladia-documentation-auto","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":63,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"gladia-documentation-auto technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-10T15:44:38.092Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T15:44:38.092Z","emptyReason":null},"stars":null,"forks":null,"downloads":1354,"packageName":null,"latestVersion":"1.0.6","tractionLabel":"1.4K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T15:44:38.092Z","emptyReason":null},"lastUpdatedAt":"2026-10-10T15:44:38.092Z","lastCrawledAt":"2026-10-10T15:44:38.092Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-11T15:44:38.092Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.6","createdAt":"2026-10-02T15:08:46.054Z","changelog":"gladia-documentation-auto 1.0.6 - Updated core documentation (SKILL.md) with refreshed product summary, usage scenarios, API/SDK/CLI command examples, model/feature tables, authentication methods, and up-to-date limits and decision guidance. - Synchronized references and product details with the latest docs.gladia.io (metadata and data updated to 2026-10-02). - Removed the outdated skill-card.md file for clarity and cleanup.","fileCount":3,"zipByteSize":7236},{"version":"1.0.5","createdAt":"2026-08-24T22:05:07.084Z","changelog":"- Updated documentation sync to latest Gladia auto-reference as of 2026-08-24. - Expanded decision guides for model selection, polling vs. webhook usage, and diarization scenarios. - Added CLI installation and usage instructions for quick terminal-based transcription. - Improved tables for model capabilities, use cases, and feature parameters. - Revised product summary and usage sections for clarity and up-to-date language support. - Removed deprecated file (skill-card.md) and restructured quick reference/workflow sections.","fileCount":3,"zipByteSize":6795},{"version":"1.0.4","createdAt":"2026-07-29T15:39:36.748Z","changelog":"gladia-documentation-auto 1.0.4 - Updated synced documentation to latest (2026-07-29 digest). - Expanded and clarified \"When to use\", decision guidance, and use-case scenarios. - Added/remodeled quick reference sections (authentication, supported formats, models, config keys). - Made workflow examples more concrete for both SDK and REST API (pre-recorded and live). - Removed skill-card.md (no longer used). - Improved organization and conciseness throughout the documentation.","fileCount":3,"zipByteSize":6346},{"version":"1.0.3","createdAt":"2026-07-09T14:07:27.808Z","changelog":"gladia-documentation-auto 1.0.3 - Updated SKILL.md to clarify model options, configuration parameters, and supported features. - Expanded information on usage scenarios, decision guidance, and workflow examples. - Revised tables listing supported formats, quotas, and API/SDK selection tips. - Removed the obsolete skill-card.md file. - Improved descriptions and guidance for when to use pre-recorded vs live, SDK vs API, and configuring diarization, vocabulary, and language detection.","fileCount":3,"zipByteSize":6677},{"version":"1.0.2","createdAt":"2026-06-09T11:42:13.972Z","changelog":"Version 1.0.2 of gladia-documentation-auto - Updated SKILL.md with improved workflow instructions, revised quick reference, and clarified decision criteria for common Gladia use cases. - Enhanced product summary and use case guidance to better assist agents with transcription, audio intelligence features, and integration options. - Removed skill-card.md (no longer maintained). - Updated audio formats, feature comparison tables, and instructions for both SDK and raw API usage. - Improved organization and readability of the documentation for broader and faster agent adoption.","fileCount":3,"zipByteSize":6774},{"version":"1.0.1","createdAt":"2026-06-08T18:17:06.069Z","changelog":"- No functional changes; documentation only. - SKILL.md file updated for expanded content and clarification. - Reference, guidance, and workflow sections remain consistent; guidance tables and feature lists unchanged. - No new features or breaking changes in this version.","fileCount":3,"zipByteSize":7159},{"version":"1.0.0","createdAt":"2026-06-08T18:01:23.795Z","changelog":"Gladia Documentation Auto v1.0.0 - Initial release of a comprehensive, auto-synced Gladia speech-to-text reference, based on docs.gladia.io. - Summarizes all major API capabilities: pre-recorded and live transcription, audio/video formats, intelligence features (translation, diarization, summarization, speaker detection, sentiment analysis, etc.). - Provides decision guidance on when to use pre-recorded vs. live, SDK vs. raw API, diarization vs. multi-channel, and custom vocabulary vs. custom spelling. - Includes usage limits, quick authentication and workflow examples in JavaScript SDK, and best-practice integration notes. - Designed as a fallback or broad overview skill; always prefer official SDK integration where possible.","fileCount":3,"zipByteSize":6953}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s170451ybbxgxcc4es3g96b4fs8823td:gladia-documentation-auto","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s170451ybbxgxcc4es3g96b4fs8823td:gladia-documentation-auto` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/gladiaio/gladia-documentation-auto before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-gladiaio-gladia-documentation-auto/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-gladiaio-gladia-documentation-auto/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-gladiaio-gladia-documentation-auto/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-gladiaio-gladia-documentation-auto/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-gladiaio-gladia-documentation-auto/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-gladiaio-gladia-documentation-auto/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T21:43:13.550Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-gladiaio-gladia-documentation-auto/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-gladiaio-gladia-documentation-auto/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-gladiaio-gladia-documentation-auto/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-gladiaio-gladia-documentation-auto/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T15:44:38.092Z","emptyReason":null},"readme":"Skill: gladia-documentation-auto\n\nOwner: gladiaio\n\nSummary: Comprehensive Gladia speech-to-text reference auto-synced from docs.gladia.io. Use as a general-purpose fallback when other specialized skills don't match, or when the user needs a broad overview of Gladia capabilities, endpoints, decision guidance, or workflows. Always prefer the official SDK; fall back to raw REST/WebSocket only when SDK cannot satisfy the requirement.\n\nTags: latest:1.0.6\n\nVersion history:\n\nv1.0.6 | 2026-10-02T15:08:46.054Z | auto\n\ngladia-documentation-auto 1.0.6\n\n- Updated core documentation (SKILL.md) with refreshed product summary, usage scenarios, API/SDK/CLI command examples, model/feature tables, authentication methods, and up-to-date limits and decision guidance.\n- Synchronized references and product details with the latest docs.gladia.io (metadata and data updated to 2026-10-02).\n- Removed the outdated skill-card.md file for clarity and cleanup.\n\nv1.0.5 | 2026-08-24T22:05:07.084Z | auto\n\n- Updated documentation sync to latest Gladia auto-reference as of 2026-08-24.\n- Expanded decision guides for model selection, polling vs. webhook usage, and diarization scenarios.\n- Added CLI installation and usage instructions for quick terminal-based transcription.\n- Improved tables for model capabilities, use cases, and feature parameters.\n- Revised product summary and usage sections for clarity and up-to-date language support.\n- Removed deprecated file (skill-card.md) and restructured quick reference/workflow sections.\n\nv1.0.4 | 2026-07-29T15:39:36.748Z | auto\n\ngladia-documentation-auto 1.0.4\n\n- Updated synced documentation to latest (2026-07-29 digest).\n- Expanded and clarified \"When to use\", decision guidance, and use-case scenarios.\n- Added/remodeled quick reference sections (authentication, supported formats, models, config keys).\n- Made workflow examples more concrete for both SDK and REST API (pre-recorded and live).\n- Removed skill-card.md (no longer used).\n- Improved organization and conciseness throughout the documentation.\n\nv1.0.3 | 2026-07-09T14:07:27.808Z | auto\n\ngladia-documentation-auto 1.0.3\n\n- Updated SKILL.md to clarify model options, configuration parameters, and supported features.\n- Expanded information on usage scenarios, decision guidance, and workflow examples.\n- Revised tables listing supported formats, quotas, and API/SDK selection tips.\n- Removed the obsolete skill-card.md file.\n- Improved descriptions and guidance for when to use pre-recorded vs live, SDK vs API, and configuring diarization, vocabulary, and language detection.\n\nv1.0.2 | 2026-06-09T11:42:13.972Z | auto\n\nVersion 1.0.2 of gladia-documentation-auto\n\n- Updated SKILL.md with improved workflow instructions, revised quick reference, and clarified decision criteria for common Gladia use cases.\n- Enhanced product summary and use case guidance to better assist agents with transcription, audio intelligence features, and integration options.\n- Removed skill-card.md (no longer maintained).\n- Updated audio formats, feature comparison tables, and instructions for both SDK and raw API usage.\n- Improved organization and readability of the documentation for broader and faster agent adoption.\n\nv1.0.1 | 2026-06-08T18:17:06.069Z | auto\n\n- No functional changes; documentation only.\n- SKILL.md file updated for expanded content and clarification.\n- Reference, guidance, and workflow sections remain consistent; guidance tables and feature lists unchanged.\n- No new features or breaking changes in this version.\n\nv1.0.0 | 2026-06-08T18:01:23.795Z | auto\n\nGladia Documentation Auto v1.0.0\n\n- Initial release of a comprehensive, auto-synced Gladia speech-to-text reference, based on docs.gladia.io.\n- Summarizes all major API capabilities: pre-recorded and live transcription, audio/video formats, intelligence features (translation, diarization, summarization, speaker detection, sentiment analysis, etc.).\n- Provides decision guidance on when to use pre-recorded vs. live, SDK vs. raw API, diarization vs. multi-channel, and custom vocabulary vs. custom spelling.\n- Includes usage limits, quick authentication and workflow examples in JavaScript SDK, and best-practice integration notes.\n- Designed as a fallback or broad overview skill; always prefer official SDK integration where possible.\n\nArchive index:\n\nArchive v1.0.6: 3 files, 7236 bytes\n\nFiles: skill-card.md (2013b), SKILL.md (14535b), _meta.json (144b)\n\nFile v1.0.6:SKILL.md\n\n---\nname: gladia-documentation-auto\ndescription: Comprehensive Gladia speech-to-text reference auto-synced from docs.gladia.io. Use as a general-purpose fallback when other specialized skills don't match, or when the user needs a broad overview of Gladia capabilities, endpoints, decision guidance, or workflows. Always prefer the official SDK; fall back to raw REST/WebSocket only when SDK cannot satisfy the requirement.\nlicense: MIT\nmetadata:\n  source: https://docs.gladia.io/.well-known/agent-skills/gladia/skill.md\n  digest: sha256:fce0f1bdb678fca35d434a7f9f589187478d079847f0b9fcd6b846c032d1365c\n  synced: \"2026-10-02\"\n---\n\n> **SDK-first**: always use the official SDK — see [gladia-sdk-integration](../gladia-sdk-integration/SKILL.md) for policy, setup, and fallback criteria.\n\n## References\n\nConsult these sibling skills as needed:\n\n- ../gladia-sdk-integration/SKILL.md -- SDK setup, client initialization, error handling, and SDK vs raw API decision guide\n- ../gladia-sdk-integration/references/sdk-versions.md -- Current SDK versions (auto-synced by CI)\n- ../gladia-troubleshooting/SKILL.md -- Common errors, gotchas, and verification checklist\n- ../gladia-live-transcription/SKILL.md -- Live streaming transcription\n- ../gladia-pre-recorded-transcription/SKILL.md -- Pre-recorded file transcription\n\n---\nname: gladia\ndescription: Use when transcribing audio or video to text, building real-time voice applications, extracting insights from speech (diarization, translation, sentiment), or integrating speech-to-text into voice agents, meeting recorders, or multilingual applications. Supports both pre-recorded (async) and live (streaming) transcription with 100+ languages.\nmetadata:\n    mintlify-proj: gladia\n    version: \"1.0\"\n---\n\n# Gladia Skill\n\n## Product summary\n\nGladia is a speech-to-text (STT) API for transcribing audio and video to text. It supports two modes: **pre-recorded** (async file upload) and **live** (real-time WebSocket streaming). The API includes audio intelligence features (diarization, translation, sentiment analysis, PII redaction, summarization) and two models: **Solaria-3** (highest accuracy on European audio, pre-recorded only, 5 languages) and **Solaria-1** (default, 100+ languages, live + async, code switching).\n\n**Key files and endpoints:**\n- API key: Get from https://app.gladia.io/apikeys\n- Pre-recorded: `POST /v2/pre-recorded` (create job), `GET /v2/pre-recorded/:id` (poll result)\n- Live: `POST /v2/live` (init session), WebSocket connection for streaming\n- Authentication: Header `x-gladia-key: YOUR_API_KEY`\n- SDKs: `@gladiaio/sdk` (JavaScript/TypeScript), `gladiaio-sdk` (Python)\n- CLI: `gladia transcribe <file>` for terminal use\n\n**Primary docs:** https://docs.gladia.io\n\n## When to use\n\nReach for Gladia when:\n- Transcribing pre-recorded audio/video files (MP3, WAV, M4A, etc.) asynchronously\n- Building real-time voice applications (live captions, voice agents, meeting recorders)\n- Extracting structured data from speech (who spoke when, sentiment, entities, translations)\n- Handling multilingual audio (100+ languages, code switching)\n- Needing high accuracy on European business audio (Solaria-3)\n- Integrating with Pipecat, LiveKit, Vapi, Twilio, or other voice platforms\n- Running transcription from the terminal (CLI)\n\nDo not use Gladia for:\n- Audio longer than 135 minutes in a single pre-recorded request (use enterprise plan for 4h15)\n- Live sessions exceeding 3 hours (start a new session before the limit)\n- Solaria-3 with code switching or languages outside EN/FR/DE/ES/IT\n- Storing audio indefinitely (configure data retention policy)\n\n## Quick reference\n\n### Authentication\n```bash\n# Set API key in environment\nexport GLADIA_API_KEY=your_key\n\n# Or pass per request\ncurl -H \"x-gladia-key: your_key\" https://api.gladia.io/v2/pre-recorded\n```\n\n### Pre-recorded transcription (SDK)\n```javascript\nconst gladia = new GladiaClient({ apiKey: \"YOUR_KEY\" });\nconst result = await gladia.preRecorded().transcribe(\"audio.mp3\");\n```\n\n```python\ngladia = GladiaClient(api_key=\"YOUR_KEY\").prerecorded()\nresult = gladia.transcribe(\"audio.mp3\")\n```\n\n### Live transcription (SDK)\n```javascript\nconst session = gladia.liveV2().startSession({\n  encoding: \"wav/pcm\",\n  sample_rate: 16000,\n  bit_depth: 16,\n  channels: 1,\n});\nsession.on(\"message\", (msg) => {\n  if (msg.type === \"transcript\" && msg.data.is_final) {\n    console.log(msg.data.utterance.text);\n  }\n});\nsession.sendAudio(audioChunk);\nsession.stopRecording();\n```\n\n### CLI\n```bash\ngladia auth set your_key\ngladia transcribe meeting.wav                    # text output\ngladia transcribe podcast.mp3 -o json            # JSON output\ngladia transcribe call.wav --diarize -o srt      # subtitles with speakers\ngladia transcribe mixed.mp3 --code-switching     # mixed languages\ngladia transcribe audio.mp3 --model solaria-3 --language en\n```\n\n### Model selection\n| Model | Best for | Modes | Languages | Code switching |\n|-------|----------|-------|-----------|---|\n| **Solaria-3** | European real-world audio, highest accuracy | Pre-recorded only | EN, FR, DE, ES, IT | No |\n| **Solaria-1** | Default, global coverage, live streaming | Pre-recorded + live | 100+ | Yes |\n\n### Audio Intelligence features\n- **Diarization**: Identify speakers (`diarization: true`)\n- **Translation**: Translate to 100+ languages (`translation: true`, set `target_languages`)\n- **Sentiment analysis**: Extract emotion and tone (`sentiment_analysis: true`)\n- **PII redaction**: Anonymize sensitive data (`pii_redaction: true`)\n- **Summarization**: Generate summaries (`summarization: true`, set `type: \"general\" | \"bullet_points\" | \"concise\"`)\n- **Named Entity Recognition**: Extract entities (`named_entity_recognition: true`)\n- **Custom vocabulary**: Boost accuracy for domain terms (`custom_vocabulary: true`, provide `vocabulary` list)\n- **Subtitles**: Generate SRT/VTT (`subtitles: true`, set `formats`)\n\n### Limits\n| Limit | Value |\n|-------|-------|\n| Pre-recorded max duration | 135 minutes (enterprise: 4h15) |\n| Live session max duration | 3 hours |\n| Pre-recorded concurrency | 25 parallel + 300 queued (paid) |\n| Live concurrency | 30 concurrent sessions |\n| File size | 1000 MB max |\n| Channels (pre-recorded) | 2 (mono/stereo) |\n| Channels (live) | 8 |\n\n## Decision guidance\n\n### When to use Solaria-3 vs Solaria-1\n\n| Condition | Use Solaria-3 | Use Solaria-1 |\n|-----------|---|---|\n| Pre-recorded audio | ✓ | ✓ |\n| Live/streaming | ✗ | ✓ |\n| European business audio (calls, meetings) | ✓ | — |\n| 100+ languages needed | ✗ | ✓ |\n| Code switching (mixed languages) | ✗ | ✓ |\n| EN, FR, DE, ES, IT only | ✓ | ✓ |\n| Clean, formal speech | — | ✓ |\n\n### When to use pre-recorded vs live\n\n| Scenario | Pre-recorded | Live |\n|----------|---|---|\n| Transcribe uploaded file | ✓ | ✗ |\n| Real-time captions | ✗ | ✓ |\n| Voice agent / IVR | ✗ | ✓ |\n| Meeting recording | ✓ | ✓ (stream during call) |\n| Batch processing | ✓ | ✗ |\n| Async job with polling | ✓ | ✗ |\n| WebSocket streaming | ✗ | ✓ |\n\n### When to use SDK vs API vs CLI\n\n| Tool | Best for | Complexity |\n|------|----------|---|\n| **SDK** (JavaScript/Python) | Application integration, error handling, retries | Low |\n| **API** (REST/WebSocket) | Custom workflows, non-SDK languages | Medium |\n| **CLI** | Terminal, scripts, CI/CD, one-off transcriptions | Very low |\n\n### Result retrieval: polling vs webhooks vs callbacks\n\n| Method | Use when | Trade-off |\n|--------|----------|-----------|\n| **Polling** (SDK `.poll()`) | Simple, synchronous flow | Blocks, wastes requests |\n| **Webhooks** | Server-to-server, configured in dashboard | Setup required, less flexible |\n| **Callbacks** | Per-job notification, no polling | Must expose HTTP endpoint |\n\n## Workflow\n\n### Pre-recorded transcription (typical flow)\n\n1. **Get API key** from https://app.gladia.io/apikeys and set `GLADIA_API_KEY` environment variable.\n\n2. **Choose model and features**: Decide between Solaria-3 (accuracy, 5 languages) or Solaria-1 (coverage, 100+ languages). List required audio intelligence features (diarization, translation, etc.).\n\n3. **Prepare audio**: Ensure file is under 135 minutes, under 1000 MB, and in a supported format (MP3, WAV, M4A, FLAC, OGG, etc.).\n\n4. **Upload and transcribe** (SDK):\n   ```javascript\n   const result = await gladia.preRecorded().transcribe(\"audio.mp3\", {\n     model: \"solaria-3\",\n     language_config: { languages: [\"en\"] },\n     diarization: true,\n   });\n   ```\n\n5. **Poll or wait for callback**: SDK `.transcribe()` polls automatically. For raw API, use `.poll()` or configure a callback URL.\n\n6. **Extract results**: Access `result.transcription.full_transcript`, `result.transcription.utterances`, `result.diarization`, `result.translation`, etc.\n\n7. **Handle errors**: Check `result.status` for \"done\" or \"error\". Inspect error details before retrying (transient vs. input issues).\n\n### Live transcription (typical flow)\n\n1. **Initialize session** (backend):\n   ```javascript\n   const response = await fetch(\"https://api.gladia.io/v2/live\", {\n     method: \"POST\",\n     headers: { \"x-gladia-key\": \"YOUR_KEY\", \"Content-Type\": \"application/json\" },\n     body: JSON.stringify({\n       encoding: \"wav/pcm\",\n       sample_rate: 16000,\n       bit_depth: 16,\n       channels: 1,\n       language_config: { languages: [\"en\"] },\n     }),\n   });\n   const { id, url } = await response.json();\n   ```\n\n2. **Return secure URL to client**: Pass `url` (contains temporary token) to frontend/mobile app. Keep API key on backend.\n\n3. **Client connects to WebSocket** and sends audio chunks as they arrive.\n\n4. **Listen for messages**: Handle `transcript` (partial/final), `speech_start`, `speech_end`, `sentiment_analysis`, etc.\n\n5. **Stop recording**: Send `stop_recording` message or close WebSocket with code 1000. Server processes remaining audio and post-processing.\n\n6. **Retrieve final results**: Call `GET /v2/live/:id` to fetch complete transcript, diarization, translation, etc.\n\n## Common gotchas\n\n- **Solaria-3 with multiple languages**: Solaria-3 does not support code switching. Pass exactly one language in `language_config.languages` (e.g., `[\"fr\"]`), not multiple. Use Solaria-1 for mixed-language audio.\n\n- **Live session timeout**: A single WebSocket session cannot exceed 3 hours. For longer events, start a new session before reaching the limit. Track session duration and reconnect proactively.\n\n- **Resubmitting the same audio**: If a POST already returned 200 or you received a `transcription.created` webhook, do not resubmit the same audio. Wait on that job ID. Resubmitting creates a new billable job.\n\n- **Polling without backoff**: Polling too aggressively wastes API quota. Use exponential backoff (start at 1s, cap at 10s) or switch to callbacks/webhooks.\n\n- **Audio format mismatch**: Ensure `encoding`, `sample_rate`, `bit_depth`, and `channels` match your actual audio. Mismatches cause silent failures or garbled output.\n\n- **Missing language specification**: If you know the language, set `language_config.languages` to skip auto-detection and reduce latency. Auto-detection adds 1–2 seconds.\n\n- **Callback URL not reachable**: If using callbacks, ensure your endpoint is publicly accessible and returns 2xx within a reasonable timeout. Gladia retries failed callbacks.\n\n- **Concurrency limits**: Free tier: 3 pre-recorded concurrent, 1 live. Paid: 25 pre-recorded concurrent, 30 live. Requests beyond the limit queue. Monitor queue time during high load.\n\n- **Data retention**: By default, audio and transcripts are retained. If GDPR/privacy is a concern, enable Zero Data Retention (results delivered only via callbacks, no retrieval).\n\n- **Partial transcripts accuracy**: Partial transcripts use a faster, smaller model than finals. Accuracy degrades with multiple languages or code switching. Use for UX only, not for final output.\n\n## Verification checklist\n\nBefore submitting work:\n\n- [ ] API key is set in environment or passed securely (never hardcoded in client code).\n- [ ] Model choice matches use case (Solaria-3 for European accuracy, Solaria-1 for live or 100+ languages).\n- [ ] Language configuration is correct: single language for Solaria-3, multiple allowed for Solaria-1 with code switching.\n- [ ] Audio file is under 135 minutes and 1000 MB (or enterprise plan for longer).\n- [ ] Audio format, encoding, sample rate, and channels are specified correctly.\n- [ ] Audio Intelligence features are enabled only if needed (diarization, translation, etc.).\n- [ ] Callback or webhook URL is configured if using async notification (not polling).\n- [ ] Error handling is in place: check job status, inspect error details, retry only on transient failures.\n- [ ] Live session duration is monitored; new session started before 3-hour limit.\n- [ ] Results are extracted from the correct fields: `transcription.full_transcript`, `transcription.utterances`, `diarization.speakers`, `translation.results`, etc.\n- [ ] Concurrency limits are respected; requests queue gracefully when limits are hit.\n- [ ] Data retention policy is configured (default: stored; set to zero-retention if required).\n\n## Resources\n\n**Comprehensive page-by-page navigation:**\nhttps://docs.gladia.io/llms.txt\n\n**Critical documentation pages:**\n1. [Pre-recorded STT Quickstart](https://docs.gladia.io/chapters/pre-recorded-stt/quickstart) — Upload, create job, poll results, configure features\n2. [Live STT Quickstart](https://docs.gladia.io/chapters/live-stt/quickstart) — WebSocket init, streaming, message handling\n3. [Models](https://docs.gladia.io/chapters/introduction/models) — Solaria-3 vs Solaria-1 comparison and selection guide\n4. [Audio Intelligence](https://docs.gladia.io/chapters/audio-intelligence/) — Diarization, translation, sentiment, PII redaction, summarization\n5. [Limits & Specifications](https://docs.gladia.io/chapters/limits-and-specifications/concurrency) — Concurrency, duration, file size, rate limits\n6. [CLI](https://docs.gladia.io/chapters/developer-tools/gladia-cli) — Terminal transcription without code\n7. [API Reference](https://docs.gladia.io/api-reference/) — Full endpoint documentation, request/response schemas\n\n---\n\n> For additional documentation and navigation, see: https://docs.gladia.io/llms.txt\n---\n\n> This file is auto-synced from https://docs.gladia.io/.well-known/agent-skills/gladia/skill.md\n> Do not edit manually — changes will be overwritten by CI.\n> For additional documentation and navigation, see: https://docs.gladia.io/llms.txt\n\nFile v1.0.6:_meta.json\n\n{\n  \"ownerId\": \"kn7fs6cmj9hqg7232sbkacf31d882wp2\",\n  \"slug\": \"gladia-documentation-auto\",\n  \"version\": \"1.0.6\",\n  \"publishedAt\": 1790953726054\n}\n\nFile v1.0.6:skill-card.md\n\n## Description:\n\nProvides Gladia speech-to-text integration guidance for pre-recorded and live transcription, including SDK examples, API workflows, and feature selection.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[gladiaio](https://clawhub.ai/user/gladiaio)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers use this reference to choose Gladia transcription modes and features, integrate the SDK or API, and handle audio transcription results.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Transcription can send sensitive audio or video to Gladia and retain audio and transcripts.\n\nMitigation: Obtain consent before uploading; configure zero data retention when privacy or compliance requires it.\n\nRisk: Exposing the Gladia API key in client code can allow unauthorized use.\n\nMitigation: Keep the API key on the backend or in a protected environment variable; never hardcode it in client code.\n\n## Reference(s):\n\n- [Gladia skill documentation](https://docs.gladia.io/.well-known/agent-skills/gladia/skill.md)\n- [Gladia documentation](https://docs.gladia.io/llms.txt)\n- [Pre-recorded transcription quickstart](https://docs.gladia.io/chapters/pre-recorded-stt/quickstart)\n- [Live transcription quickstart](https://docs.gladia.io/chapters/live-stt/quickstart)\n- [Gladia API reference](https://docs.gladia.io/api-reference/)\n\n## Skill Output:\n\n**Output Type(s):** [Guidance, Code, Shell commands, Configuration instructions]\n\n**Output Format:** [Markdown with code examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [SDK-first guidance for pre-recorded and live transcription.]\n\n## Skill Version(s):\n\n1.0.6 (source: ClawHub release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v1.0.5: 3 files, 6795 bytes\n\nFiles: skill-card.md (2299b), SKILL.md (13347b), _meta.json (144b)\n\nFile v1.0.5:SKILL.md\n\n---\nname: gladia-documentation-auto\ndescription: Comprehensive Gladia speech-to-text reference auto-synced from docs.gladia.io. Use as a general-purpose fallback when other specialized skills don't match, or when the user needs a broad overview of Gladia capabilities, endpoints, decision guidance, or workflows. Always prefer the official SDK; fall back to raw REST/WebSocket only when SDK cannot satisfy the requirement.\nlicense: MIT\nmetadata:\n  source: https://docs.gladia.io/.well-known/agent-skills/gladia/skill.md\n  digest: sha256:1d7371c8d1abec687444f241e8bdf70df72389225c2665e17c56ab0a730d8a4d\n  synced: \"2026-08-24\"\n---\n\n> **SDK-first**: always use the official SDK — see [gladia-sdk-integration](../gladia-sdk-integration/SKILL.md) for policy, setup, and fallback criteria.\n\n## References\n\nConsult these sibling skills as needed:\n\n- ../gladia-sdk-integration/SKILL.md -- SDK setup, client initialization, error handling, and SDK vs raw API decision guide\n- ../gladia-sdk-integration/references/sdk-versions.md -- Current SDK versions (auto-synced by CI)\n- ../gladia-troubleshooting/SKILL.md -- Common errors, gotchas, and verification checklist\n- ../gladia-live-transcription/SKILL.md -- Live streaming transcription\n- ../gladia-pre-recorded-transcription/SKILL.md -- Pre-recorded file transcription\n\n---\nname: Gladia\ndescription: Use when building speech-to-text transcription features, processing audio or video files, implementing real-time transcription, extracting speaker information, translating transcripts, or adding audio intelligence features like summarization, sentiment analysis, and entity recognition.\nmetadata:\n    mintlify-proj: gladia\n    version: \"1.0\"\n---\n\n# Gladia Skill\n\n## Product summary\n\nGladia is a speech-to-text (STT) API that transcribes audio and video files asynchronously (pre-recorded) or in real-time (live). It supports 100+ languages, speaker diarization, translation, and audio intelligence features like summarization, sentiment analysis, and entity recognition. Use the JavaScript/Python SDKs (`@gladiaio/sdk` or `gladiaio-sdk`), REST API endpoints, or the CLI (`gladia` command) to transcribe. Authentication uses the `x-gladia-key` header. Primary docs: https://docs.gladia.io\n\n**Key endpoints:**\n- Pre-recorded: `POST /v2/pre-recorded` (create job), `GET /v2/pre-recorded/{id}` (get result), `POST /v2/upload` (upload file)\n- Live: `POST /v2/live` (init session), WebSocket connection for streaming audio\n- Authentication: `x-gladia-key` header on all requests\n\n## When to use\n\nReach for this skill when:\n- A user asks to transcribe audio or video files (meetings, calls, podcasts, interviews)\n- Building real-time transcription for live calls, meetings, or voice agents\n- Extracting speaker information from multi-speaker audio (diarization)\n- Translating transcripts to multiple languages\n- Generating summaries, detecting sentiment, or extracting entities from audio\n- Processing audio from URLs or local files\n- Implementing webhooks or callbacks for async job completion\n- Migrating from Deepgram or AssemblyAI to Gladia\n- Using the CLI for quick terminal-based transcription\n\n## Quick reference\n\n### Models\n\n| Model | Use case | Languages | Code switching | Live support |\n|-------|----------|-----------|-----------------|--------------|\n| `solaria-3` | Highest accuracy on European audio | en, fr, de, es, it | No | No (pre-recorded only) |\n| `solaria-1` | Generalist, maximum coverage | 100+ languages | Yes | Yes |\n\n### Pre-recorded workflow\n\n1. **Upload** (optional): `POST /v2/upload` with multipart form-data → get `audio_url`\n2. **Create job**: `POST /v2/pre-recorded` with `audio_url` and options → get `id`\n3. **Get result**: Poll `GET /v2/pre-recorded/{id}` until `status: \"done\"` or use webhooks/callbacks\n\n### Live workflow\n\n1. **Init session**: `POST /v2/live` with audio config → get WebSocket `url` and `id`\n2. **Connect**: Open WebSocket, send audio chunks as binary or base64\n3. **Read messages**: Receive transcript, translation, sentiment, etc. via WebSocket\n4. **Stop**: Send `stop_recording` message or close with code 1000\n\n### Audio Intelligence features (pre-recorded & live)\n\n| Feature | Parameter | Use case |\n|---------|-----------|----------|\n| Speaker diarization | `diarization: true` | Identify who spoke when |\n| Translation | `translation: true` | Translate to multiple languages |\n| Summarization | `summarization: true` | Generate summaries or bullet points |\n| Sentiment analysis | `sentiment_analysis: true` | Extract emotions and tone |\n| Named entity recognition | `named_entity_recognition: true` | Detect people, organizations, dates |\n| PII redaction | `pii_redaction: true` | Anonymize sensitive data |\n| Custom vocabulary | `custom_vocabulary: true` | Boost accuracy for domain terms |\n| Subtitles | `subtitles: true` | Generate SRT or VTT files |\n\n### SDK installation\n\n```bash\n# JavaScript\nnpm install @gladiaio/sdk\n\n# Python\npip install gladiaio-sdk\n# or\nuv add gladiaio-sdk\n```\n\n### CLI installation\n\n```bash\n# macOS & Linux\ncurl -fsSL https://github.com/gladiaio/gladia-cli/releases/latest/download/install.sh | sh\n\n# Windows (PowerShell)\npowershell -c \"irm https://github.com/gladiaio/gladia-cli/releases/latest/download/install.ps1 | iex\"\n```\n\n### CLI quick commands\n\n```bash\ngladia auth set YOUR_API_KEY\ngladia transcribe audio.wav                    # plain text\ngladia transcribe audio.wav -o json            # JSON output\ngladia transcribe audio.wav --diarize          # with speaker labels\ngladia transcribe audio.wav --language en,fr   # constrain languages\ngladia transcribe audio.wav --model solaria-3  # use specific model\n```\n\n## Decision guidance\n\n### When to use pre-recorded vs. live\n\n| Scenario | Use |\n|----------|-----|\n| User uploads a file, you transcribe later | Pre-recorded (`/v2/pre-recorded`) |\n| Real-time call, meeting, or voice agent | Live (`/v2/live` + WebSocket) |\n| Batch processing many files | Pre-recorded with polling or webhooks |\n| Need results immediately as user speaks | Live with `receive_partial_transcripts: true` |\n\n### When to use polling vs. webhooks vs. callbacks\n\n| Approach | Best for | Trade-off |\n|----------|----------|-----------|\n| Polling | Small jobs, quick feedback | Keeps connection open, higher latency |\n| Webhooks | Production, many jobs | Configure once at https://app.gladia.io/webhooks |\n| Callbacks | Per-job control | Set `callback_config` in request body |\n\n### When to use solaria-1 vs. solaria-3\n\n| Condition | Choose |\n|-----------|--------|\n| European audio, high accuracy needed | `solaria-3` |\n| Multiple languages or code switching | `solaria-1` |\n| Live transcription required | `solaria-1` (only option) |\n| Unknown language or 100+ language support | `solaria-1` |\n\n### When to enable diarization\n\n| Scenario | Enable |\n|----------|--------|\n| Multi-speaker call or meeting | Yes |\n| Single speaker or known channels | No (use `channel` field instead) |\n| Want speaker labels in output | Yes, set `diarization: true` |\n| Know exact speaker count | Set `diarization_config.number_of_speakers` |\n| Speaker count varies | Set `diarization_config.min_speakers` and `max_speakers` |\n\n## Workflow\n\n### Pre-recorded transcription (SDK)\n\n1. **Initialize client**: Create `GladiaClient` with API key\n2. **Transcribe in one call**: Use `.transcribe(audioPath, options)` for end-to-end flow\n   - Pass local file path, URL, or binary data\n   - SDK handles upload, job creation, polling automatically\n3. **Or use individual steps** for fine control:\n   - Upload: `.uploadFile(path)` → get `audio_url`\n   - Create: `.createUntyped({audio_url, ...options})` → get `id`\n   - Poll: `.get(id)` until `status: \"done\"`\n4. **Access results**: Extract `result.transcription.full_transcript`, utterances, translations, etc.\n\n### Pre-recorded transcription (API)\n\n1. **Upload audio** (if local file): `POST /v2/upload` with multipart form-data\n2. **Create job**: `POST /v2/pre-recorded` with JSON body containing `audio_url` and options\n3. **Poll for result**: `GET /v2/pre-recorded/{id}` every 2–5 seconds until `status: \"done\"`\n   - Or configure webhook at https://app.gladia.io/webhooks\n   - Or set `callback_config` in request to receive POST when done\n4. **Parse result**: Check `result.transcription`, `result.translation`, `result.summarization`, etc.\n\n### Live transcription (SDK)\n\n1. **Initialize session**: Call `.startSession(config)` with audio format (encoding, sample_rate, bit_depth, channels)\n2. **Listen for events**: Register handlers for `message`, `started`, `ended`, `error`\n3. **Send audio chunks**: Call `.sendAudio(chunk)` as audio arrives\n4. **Read transcripts**: On `message` event, check `message.type === 'transcript'` and `is_final` flag\n5. **Stop recording**: Call `.stopRecording()` to finalize and trigger post-processing\n6. **Get final result**: Call `GET /v2/live/{id}` after session ends\n\n### Live transcription (API)\n\n1. **Init session**: `POST /v2/live` with audio config → get WebSocket `url` and `id`\n2. **Connect WebSocket**: Open connection to returned `url`\n3. **Send audio**: Send binary chunks or JSON with base64-encoded audio\n4. **Read messages**: Parse incoming JSON messages; check `type` and `is_final` fields\n5. **Stop**: Send `{\"type\": \"stop_recording\"}` or close with code 1000\n6. **Retrieve final result**: `GET /v2/live/{id}` after WebSocket closes\n\n### CLI transcription\n\n1. **Set API key**: `gladia auth set YOUR_KEY` (or export `GLADIA_API_KEY`)\n2. **Transcribe**: `gladia transcribe <file-or-url> [options]`\n3. **Choose output**: Use `-o text|json|json-full|srt|vtt`\n4. **Add features**: Use `--diarize`, `--language en,fr`, `--code-switching`, `--model solaria-3`\n5. **Pipe output**: Use in scripts: `gladia transcribe audio.wav -o json | jq '.transcription'`\n\n## Common gotchas\n\n- **solaria-3 with multiple languages**: solaria-3 accepts only ONE language in `language_config.languages` — do not pass multiple or enable code switching. Use solaria-1 for multi-language.\n- **Live transcription model**: Live only supports `solaria-1`. Attempting `solaria-3` will fail.\n- **Audio format mismatch**: Specify `encoding`, `sample_rate`, `bit_depth`, and `channels` correctly in live init — mismatches cause garbled output.\n- **3-hour session limit**: Live sessions terminate after 3 hours. Start a new session before hitting the limit.\n- **Pre-recorded duration limit**: Max 135 minutes per request (4h15 for enterprise). Split longer files into ~60-minute chunks.\n- **File size limit**: Max 1000 MB. Larger files are rejected.\n- **Polling without timeout**: Always set a max retry count or timeout when polling — don't loop indefinitely.\n- **API key in client code**: Never expose `x-gladia-key` in frontend code. Generate WebSocket URL on backend and pass only the URL to clients.\n- **Diarization hints are not hard constraints**: `number_of_speakers`, `min_speakers`, `max_speakers` are hints; actual detection may differ.\n- **Custom vocabulary not a spell-checker**: Custom vocabulary boosts recognition but doesn't force exact spelling — use `custom_spelling` for strict spelling control.\n- **Callback URL must be public**: Webhook and callback endpoints must be reachable from Gladia's servers; localhost won't work.\n- **Multi-channel billing**: Transcribing multi-channel audio is billed per channel — a 2-channel stream costs 2x the duration.\n- **Deprecated endpoints**: Avoid `/v2/transcription/*` endpoints (deprecated); use `/v2/pre-recorded/*` instead.\n\n## Verification checklist\n\nBefore submitting transcription work:\n\n- [ ] API key is set and valid (test with a simple request)\n- [ ] Audio file or URL is accessible and in a supported format (mp3, wav, m4a, ogg, flac, etc.)\n- [ ] For pre-recorded: audio duration ≤ 135 minutes (or split into chunks)\n- [ ] For live: session will not exceed 3 hours\n- [ ] Model choice matches use case (solaria-3 for European audio + single language; solaria-1 for multi-language or live)\n- [ ] Language config is correct (solaria-3: exactly one language; solaria-1: one or more, or auto-detect)\n- [ ] Audio Intelligence features are enabled only if needed (diarization, translation, etc.)\n- [ ] Callback/webhook URL is public and reachable (if using async notifications)\n- [ ] Result parsing handles all expected fields (`transcription`, `translation`, `summarization`, etc.)\n- [ ] Error handling covers job failures, network timeouts, and rate limits\n- [ ] For live: WebSocket reconnection logic handles network drops gracefully\n\n## Resources\n\n**Comprehensive page listing**: https://docs.gladia.io/llms.txt\n\n**Critical documentation pages**:\n1. [Pre-recorded STT Quickstart](https://docs.gladia.io/chapters/pre-recorded-stt/quickstart) — Upload, create job, get result\n2. [Live STT Quickstart](https://docs.gladia.io/chapters/live-stt/quickstart) — WebSocket streaming, real-time transcription\n3. [API Reference](https://docs.gladia.io/api-reference) — Full endpoint documentation, request/response schemas\n\n---\n\n> For additional documentation and navigation, see: https://docs.gladia.io/llms.txt\n---\n\n> This file is auto-synced from https://docs.gladia.io/.well-known/agent-skills/gladia/skill.md\n> Do not edit manually — changes will be overwritten by CI.\n> For additional documentation and navigation, see: https://docs.gladia.io/llms.txt\n\nFile v1.0.5:_meta.json\n\n{\n  \"ownerId\": \"kn7fs6cmj9hqg7232sbkacf31d882wp2\",\n  \"slug\": \"gladia-documentation-auto\",\n  \"version\": \"1.0.5\",\n  \"publishedAt\": 1787609107084\n}\n\nFile v1.0.5:skill-card.md\n\n## Description:\n\nComprehensive Gladia speech-to-text reference auto-synced from docs.gladia.io for broad guidance on Gladia capabilities, endpoints, decision guidance, and workflows.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[gladiaio](https://clawhub.ai/user/gladiaio)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and engineers use this skill when building Gladia speech-to-text features, including pre-recorded transcription, live transcription, diarization, translation, audio intelligence, CLI use, and SDK-first implementation guidance.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill includes CLI installation examples that execute mutable remote scripts directly.\n\nMitigation: Prefer official SDKs with pinned versions and lockfiles, or download and verify CLI installers before execution.\n\nRisk: Gladia workflows may involve API keys, webhook endpoints, audio, URLs, or transcripts.\n\nMitigation: Protect API keys and webhook endpoints, and only send audio, URLs, and transcripts that are approved for upload to Gladia.\n\n## Reference(s):\n\n- [Gladia Agent Skill Source](https://docs.gladia.io/.well-known/agent-skills/gladia/skill.md)\n- [Gladia Documentation](https://docs.gladia.io)\n- [Gladia Documentation Index](https://docs.gladia.io/llms.txt)\n- [Pre-recorded STT Quickstart](https://docs.gladia.io/chapters/pre-recorded-stt/quickstart)\n- [Live STT Quickstart](https://docs.gladia.io/chapters/live-stt/quickstart)\n- [Gladia API Reference](https://docs.gladia.io/api-reference)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, code, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown with inline code blocks, API examples, command snippets, and decision tables]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May include SDK-first recommendations, REST and WebSocket workflows, CLI usage, and verification checklists.]\n\n## Skill Version(s):\n\n1.0.5 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v1.0.4: 3 files, 6346 bytes\n\nFiles: skill-card.md (2790b), SKILL.md (12000b), _meta.json (144b)\n\nFile v1.0.4:SKILL.md\n\n---\nname: gladia-documentation-auto\ndescription: Comprehensive Gladia speech-to-text reference auto-synced from docs.gladia.io. Use as a general-purpose fallback when other specialized skills don't match, or when the user needs a broad overview of Gladia capabilities, endpoints, decision guidance, or workflows. Always prefer the official SDK; fall back to raw REST/WebSocket only when SDK cannot satisfy the requirement.\nlicense: MIT\nmetadata:\n  source: https://docs.gladia.io/.well-known/agent-skills/gladia/skill.md\n  digest: sha256:cb36d328ed195df119a37786db62d66aa0a88eda05f8520ddac648e26a00601f\n  synced: \"2026-07-29\"\n---\n\n> **SDK-first**: always use the official SDK — see [gladia-sdk-integration](../gladia-sdk-integration/SKILL.md) for policy, setup, and fallback criteria.\n\n## References\n\nConsult these sibling skills as needed:\n\n- ../gladia-sdk-integration/SKILL.md -- SDK setup, client initialization, error handling, and SDK vs raw API decision guide\n- ../gladia-sdk-integration/references/sdk-versions.md -- Current SDK versions (auto-synced by CI)\n- ../gladia-troubleshooting/SKILL.md -- Common errors, gotchas, and verification checklist\n- ../gladia-live-transcription/SKILL.md -- Live streaming transcription\n- ../gladia-pre-recorded-transcription/SKILL.md -- Pre-recorded file transcription\n\n---\nname: Gladia\ndescription: Use when building speech-to-text transcription features, processing audio or video files, implementing real-time transcription, extracting insights from audio (speaker identification, translation, sentiment, summaries), or integrating transcription into applications via API or SDK.\nmetadata:\n    mintlify-proj: gladia\n    version: \"1.0\"\n---\n\n# Gladia Skill\n\n## Product summary\n\nGladia is a speech-to-text (STT) API that transcribes audio and video files in two modes: **pre-recorded** (asynchronous, for files) and **live** (real-time, for streaming audio). Beyond transcription, it offers audio intelligence features like speaker diarization, translation, sentiment analysis, summarization, and named entity recognition. Use the **JavaScript/TypeScript SDK** (`@gladiaio/sdk`) or **Python SDK** (`gladiaio-sdk`) for simplified integration, or call the REST API directly. Authenticate with the `x-gladia-key` header. Primary docs: https://docs.gladia.io\n\n**Key endpoints:**\n- Pre-recorded: `POST /v2/pre-recorded` (create job), `GET /v2/pre-recorded/:id` (get result)\n- Live: `POST /v2/live` (init session), WebSocket connection for streaming\n- Upload: `POST /v2/upload` (for local files)\n\n**Models:** `solaria-3` (pre-recorded, highest quality), `solaria-1` (live only)\n\n## When to use\n\nReach for Gladia when:\n- **Transcribing files**: User uploads an audio/video file and needs a transcript (meetings, podcasts, interviews, call recordings)\n- **Real-time transcription**: Streaming audio from a microphone, phone call, or live event that needs instant captions or transcripts\n- **Extracting insights**: Need to identify speakers, translate to other languages, detect sentiment, summarize, or extract entities from audio\n- **Multi-language support**: Audio contains multiple languages or needs translation to 100+ target languages\n- **Subtitle generation**: Creating SRT/VTT subtitle files for video content\n- **Integration**: Building a feature into an app or workflow (via SDK or API)\n\nDo **not** use Gladia for: text-only processing, image analysis, or non-audio content.\n\n## Quick reference\n\n### Authentication\n```bash\n# All requests require the x-gladia-key header\ncurl -H \"x-gladia-key: YOUR_API_KEY\" https://api.gladia.io/v2/...\n```\n\n### Pre-recorded workflow (SDK)\n```javascript\nconst { GladiaClient } = require(\"@gladiaio/sdk\");\nconst client = new GladiaClient({ apiKey: \"YOUR_KEY\" });\n\n// One-call transcription\nconst result = await client.preRecorded().transcribe(\"audio.mp3\", {\n  model: \"solaria-3\",\n  language_config: { languages: [\"en\"] },\n  diarization: true,\n  translation: true,\n  translation_config: { target_languages: [\"fr\"] }\n});\n```\n\n### Live workflow (SDK)\n```javascript\nconst config = {\n  model: \"solaria-1\",\n  encoding: \"wav/pcm\",\n  sample_rate: 16000,\n  bit_depth: 16,\n  channels: 1,\n  language_config: { languages: [\"en\"] }\n};\n\nconst session = client.liveV2().startSession(config);\nsession.on(\"message\", (msg) => console.log(msg));\nsession.sendAudio(audioChunk);\nsession.stopRecording();\n```\n\n### Supported audio formats\nMP3, WAV, FLAC, OGG, OPUS, AAC, M4A, AC3, EAC3, MP2\n\n### Supported video formats\nMP4, MOV, AVI, MKV, FLV, 3GP, WMV, M4V\n\n### Language codes\nUse 2-letter ISO 639-1 codes: `en`, `fr`, `de`, `es`, `zh`, `ja`, etc. Full list at `/chapters/language/supported-languages`\n\n### Audio intelligence features\n| Feature | Pre-recorded | Live | Config key |\n|---------|--------------|------|-----------|\n| Speaker diarization | ✓ | ✗ | `diarization` |\n| Translation | ✓ | ✓ | `translation` |\n| Sentiment analysis | ✓ | ✗ | `sentiment_analysis` |\n| Summarization | ✓ | ✗ | `summarization` |\n| Named entity recognition | ✓ | ✓ | `named_entity_recognition` |\n| PII redaction | ✓ | ✗ | `pii_redaction` |\n| Subtitles (SRT/VTT) | ✓ | ✗ | `subtitles` |\n| Chapterization | ✓ | ✗ | `chapterization` |\n| Custom vocabulary | ✓ | ✓ | `custom_vocabulary` |\n\n## Decision guidance\n\n### Pre-recorded vs. Live\n\n| Scenario | Use pre-recorded | Use live |\n|----------|------------------|----------|\n| User uploads a file | ✓ | ✗ |\n| Streaming microphone input | ✗ | ✓ |\n| Phone call transcription | ✗ | ✓ |\n| Podcast/interview processing | ✓ | ✗ |\n| Real-time captions | ✗ | ✓ |\n| Need diarization | ✓ | ✗ (use multi-channel instead) |\n| Need sentiment/summarization | ✓ | ✗ |\n\n### Model selection\n\n| Model | Use case | Availability |\n|-------|----------|--------------|\n| `solaria-3` | Highest accuracy, best for meetings/calls/podcasts | Pre-recorded only |\n| `solaria-1` | Real-time streaming, lower latency | Live only |\n\n### Language detection vs. explicit language\n\n| Approach | When to use |\n|----------|------------|\n| Explicit `language_config.languages: [\"en\"]` | Language is known; avoids detection overhead |\n| Auto-detect (omit languages) | Language unknown or mixed; slower but flexible |\n| Code switching `code_switching: true` | Audio mixes multiple languages; requires language list |\n\n### Diarization vs. multi-channel\n\n| Approach | When to use |\n|----------|------------|\n| `diarization: true` | Single audio stream, multiple speakers (meetings, interviews) |\n| Multi-channel (channels > 1) | Separate audio tracks per speaker (phone calls with distinct channels) |\n\n## Workflow\n\n### Pre-recorded transcription (file upload)\n\n1. **Prepare the file**: Ensure audio/video is in a supported format (MP3, WAV, MP4, etc.) and under 1000 MB. For files >135 minutes, split into ~60-minute chunks.\n\n2. **Upload the file** (if local):\n   ```bash\n   curl -X POST https://api.gladia.io/v2/upload \\\n     -H \"x-gladia-key: YOUR_KEY\" \\\n     -F \"audio=@audio.mp3\"\n   ```\n   Save the returned `audio_url`.\n\n3. **Create transcription job**:\n   ```bash\n   curl -X POST https://api.gladia.io/v2/pre-recorded \\\n     -H \"x-gladia-key: YOUR_KEY\" \\\n     -H \"Content-Type: application/json\" \\\n     -d '{\n       \"audio_url\": \"https://api.gladia.io/file/...\",\n       \"model\": \"solaria-3\",\n       \"language_config\": { \"languages\": [\"en\"] },\n       \"diarization\": true\n     }'\n   ```\n   Save the returned `id`.\n\n4. **Poll for results**:\n   ```bash\n   curl https://api.gladia.io/v2/pre-recorded/ID \\\n     -H \"x-gladia-key: YOUR_KEY\"\n   ```\n   Repeat until `status: \"done\"`. Or use webhooks/callbacks instead of polling.\n\n5. **Extract data**: Parse `result.transcription.utterances` for text, timing, speaker, language, and confidence.\n\n### Live transcription (streaming)\n\n1. **Initialize session**:\n   ```bash\n   curl -X POST https://api.gladia.io/v2/live \\\n     -H \"x-gladia-key: YOUR_KEY\" \\\n     -H \"Content-Type: application/json\" \\\n     -d '{\n       \"encoding\": \"wav/pcm\",\n       \"sample_rate\": 16000,\n       \"bit_depth\": 16,\n       \"channels\": 1\n     }'\n   ```\n   Save the returned WebSocket `url`.\n\n2. **Connect WebSocket**: Open a WebSocket connection to the URL.\n\n3. **Send audio chunks**: Stream audio in the specified encoding/sample rate.\n\n4. **Read messages**: Listen for `transcript`, `translation`, `sentiment_analysis`, etc. messages.\n\n5. **Stop recording**: Send `{ \"type\": \"stop_recording\" }` or close the socket with code 1000.\n\n6. **Retrieve final results**: Call `GET /v2/live/:id` to get the complete result.\n\n## Common gotchas\n\n- **Model mismatch**: `solaria-3` is pre-recorded only; `solaria-1` is live only. Using the wrong model will fail.\n- **Language config with solaria-3**: Must set exactly **one** language in `language_config.languages` (e.g., `[\"fr\"]`). Do not pass multiple languages or enable code switching with solaria-3.\n- **Polling timeout**: Pre-recorded jobs can take minutes. Don't assume instant results; use webhooks or callbacks for production.\n- **File size limits**: Max 1000 MB per file; max 135 minutes per job (enterprise plans support 4h15). Split large files.\n- **Live session duration**: Max 3 hours per WebSocket session. Start a new session before hitting the limit.\n- **Audio format mismatch**: Ensure `encoding`, `sample_rate`, `bit_depth`, and `channels` match your actual audio stream, or transcription will fail silently.\n- **Diarization hints are not constraints**: `number_of_speakers`, `min_speakers`, `max_speakers` are hints, not hard limits. The model may detect a different count.\n- **Translation with code switching**: Do not enable both `code_switching: true` and `translation` on the same request without a constrained language list.\n- **Concurrency limits**: Free tier: 3 concurrent pre-recorded, 1 live. Paid: 25 concurrent pre-recorded, 30 live. Queued requests beyond concurrency limits will wait.\n- **Deprecated V1 API**: Old endpoints like `/audio/text/audio-transcription/` no longer work. Use `/v2/pre-recorded` and `/v2/live`.\n\n## Verification checklist\n\nBefore submitting transcription work:\n\n- [ ] API key is valid and passed in `x-gladia-key` header\n- [ ] Audio file is in a supported format (MP3, WAV, MP4, etc.)\n- [ ] File size is under 1000 MB; duration under 135 minutes (or split into chunks)\n- [ ] For pre-recorded: `model` is `solaria-3`; for live: `model` is `solaria-1`\n- [ ] Language code is valid ISO 639-1 (e.g., `en`, `fr`, `de`)\n- [ ] For solaria-3: exactly one language in `language_config.languages`\n- [ ] Audio encoding, sample rate, bit depth, and channels match the actual stream (live only)\n- [ ] Diarization is enabled if speaker identification is needed\n- [ ] Translation target languages are in the supported list\n- [ ] Webhook/callback URL is reachable (if using async notifications)\n- [ ] Job status is `done` before reading results\n- [ ] Response contains `result.transcription.utterances` with text and timing\n\n## Resources\n\n- **Full page navigation**: https://docs.gladia.io/llms.txt\n- **API Reference**: https://docs.gladia.io/api-reference/\n- **Pre-recorded quickstart**: https://docs.gladia.io/chapters/pre-recorded-stt/quickstart\n- **Live transcription quickstart**: https://docs.gladia.io/chapters/live-stt/quickstart\n- **Audio intelligence features**: https://docs.gladia.io/chapters/audio-intelligence/\n- **Recommended parameters by use case**: https://docs.gladia.io/chapters/pre-recorded-stt/recommended-parameters\n- **Supported languages**: https://docs.gladia.io/chapters/language/supported-languages\n- **SDK (JavaScript)**: https://www.npmjs.com/package/@gladiaio/sdk\n- **SDK (Python)**: https://pypi.org/project/gladiaio-sdk/\n\n---\n\n> For additional documentation and navigation, see: https://docs.gladia.io/llms.txt\n---\n\n> This file is auto-synced from https://docs.gladia.io/.well-known/agent-skills/gladia/skill.md\n> Do not edit manually — changes will be overwritten by CI.\n> For additional documentation and navigation, see: https://docs.gladia.io/llms.txt\n\nFile v1.0.4:_meta.json\n\n{\n  \"ownerId\": \"kn7fs6cmj9hqg7232sbkacf31d882wp2\",\n  \"slug\": \"gladia-documentation-auto\",\n  \"version\": \"1.0.4\",\n  \"publishedAt\": 1785339576748\n}\n\nFile v1.0.4:skill-card.md\n\n## Description: <br>\nComprehensive Gladia speech-to-text reference auto-synced from docs.gladia.io for agents that need broad Gladia capability, endpoint, decision, and workflow guidance. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[gladiaio](https://clawhub.ai/user/gladiaio) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and agents use this skill to plan and implement Gladia speech-to-text workflows for prerecorded files, live transcription, audio intelligence, and SDK-first API integration. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill can guide agents toward sending audio, video, and related metadata to Gladia's external API. <br>\nMitigation: Confirm user intent, privacy and retention requirements, and use of an appropriate Gladia API key before processing recordings. <br>\nRisk: Sensitive or regulated recordings may require additional approval before transcription. <br>\nMitigation: Avoid processing those recordings unless the user's applicable privacy, retention, and approval requirements are satisfied. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/gladiaio/skills/gladia-documentation-auto) <br>\n- [Gladia agent skill source](https://docs.gladia.io/.well-known/agent-skills/gladia/skill.md) <br>\n- [Gladia documentation](https://docs.gladia.io) <br>\n- [Gladia API reference](https://docs.gladia.io/api-reference/) <br>\n- [Pre-recorded transcription quickstart](https://docs.gladia.io/chapters/pre-recorded-stt/quickstart) <br>\n- [Live transcription quickstart](https://docs.gladia.io/chapters/live-stt/quickstart) <br>\n- [Audio intelligence features](https://docs.gladia.io/chapters/audio-intelligence/) <br>\n- [Supported languages](https://docs.gladia.io/chapters/language/supported-languages) <br>\n- [JavaScript SDK](https://www.npmjs.com/package/@gladiaio/sdk) <br>\n- [Python SDK](https://pypi.org/project/gladiaio-sdk/) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, code, shell commands, configuration, guidance] <br>\n**Output Format:** [Markdown guidance with inline code, shell commands, API examples, and configuration snippets] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [SDK-first guidance; raw REST or WebSocket examples are fallback options when the SDK cannot satisfy the requirement.] <br>\n\n## Skill Version(s): <br>\n1.0.4 (source: server release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.0.3: 3 files, 6677 bytes\n\nFiles: skill-card.md (2817b), SKILL.md (12592b), _meta.json (144b)\n\nFile v1.0.3:SKILL.md\n\n---\nname: gladia-documentation-auto\ndescription: Comprehensive Gladia speech-to-text reference auto-synced from docs.gladia.io. Use as a general-purpose fallback when other specialized skills don't match, or when the user needs a broad overview of Gladia capabilities, endpoints, decision guidance, or workflows. Always prefer the official SDK; fall back to raw REST/WebSocket only when SDK cannot satisfy the requirement.\nlicense: MIT\nmetadata:\n  source: https://docs.gladia.io/.well-known/agent-skills/gladia/skill.md\n  digest: sha256:c555f4874bbc15a36a96c434a0d3852c3a94b95868381ae35fe823e80876a71e\n  synced: \"2026-07-09\"\n---\n\n> **SDK-first**: always use the official SDK — see [gladia-sdk-integration](../gladia-sdk-integration/SKILL.md) for policy, setup, and fallback criteria.\n\n## References\n\nConsult these sibling skills as needed:\n\n- ../gladia-sdk-integration/SKILL.md -- SDK setup, client initialization, error handling, and SDK vs raw API decision guide\n- ../gladia-sdk-integration/references/sdk-versions.md -- Current SDK versions (auto-synced by CI)\n- ../gladia-troubleshooting/SKILL.md -- Common errors, gotchas, and verification checklist\n- ../gladia-live-transcription/SKILL.md -- Live streaming transcription\n- ../gladia-pre-recorded-transcription/SKILL.md -- Pre-recorded file transcription\n\n---\nname: Gladia\ndescription: Use when building speech-to-text transcription features, processing audio/video files, implementing real-time transcription, extracting insights from audio (speaker identification, translation, sentiment), or integrating audio intelligence into applications. Agents should reach for this skill when users request transcription, audio analysis, or speech processing capabilities.\nmetadata:\n    mintlify-proj: gladia\n    version: \"1.0\"\n---\n\n# Gladia Skill\n\n## Product summary\n\nGladia is a speech-to-text (STT) API that transcribes audio and video files in both pre-recorded (asynchronous) and live (real-time) modes. It provides transcription plus audio intelligence features: speaker diarization, translation, sentiment analysis, PII redaction, summarization, and more. Agents use Gladia to build transcription workflows, extract data from audio, and integrate speech processing into applications.\n\n**Key entry points:**\n- **Pre-recorded API**: `POST /v2/pre-recorded` (async transcription)\n- **Live API**: `POST /v2/live` (WebSocket-based real-time)\n- **Upload endpoint**: `POST /v2/upload` (for local files)\n- **Authentication**: `x-gladia-key` header with API key\n- **SDKs**: JavaScript/TypeScript (`@gladiaio/sdk`) and Python (`gladiaio-sdk`)\n- **Primary docs**: https://docs.gladia.io\n\n## When to use\n\nReach for this skill when:\n- A user requests transcription of audio or video files (meetings, calls, podcasts, interviews)\n- Building real-time transcription for live events or streaming audio\n- Extracting speaker information, translations, or sentiment from audio\n- Implementing PII redaction for compliance (GDPR, HIPAA, CCPA)\n- Processing multi-language content with language detection or code switching\n- Generating subtitles, summaries, or meeting notes from audio\n- Integrating with platforms like Twilio, LiveKit, Vapi, or Pipecat\n- Optimizing transcription quality with custom vocabulary or domain-specific terms\n\n## Quick reference\n\n### Models\n\n| Model | Use case | Supports |\n|-------|----------|----------|\n| `solaria-3` | Pre-recorded (best quality, slower) | All features except live |\n| `solaria-1` | Live/real-time only | Streaming, partial transcripts |\n\n### Core parameters (pre-recorded)\n\n```json\n{\n  \"audio_url\": \"https://...\",  // or upload first\n  \"model\": \"solaria-3\",\n  \"language_config\": {\n    \"languages\": [\"en\"],  // ISO 639-1 codes\n    \"code_switching\": false\n  },\n  \"diarization\": true,\n  \"diarization_config\": {\n    \"number_of_speakers\": 2,  // or min/max\n    \"min_speakers\": 1,\n    \"max_speakers\": 5\n  }\n}\n```\n\n### Core parameters (live)\n\n```json\n{\n  \"model\": \"solaria-1\",\n  \"encoding\": \"wav/pcm\",\n  \"sample_rate\": 16000,\n  \"bit_depth\": 16,\n  \"channels\": 1,\n  \"language_config\": {\n    \"languages\": [\"en\"],\n    \"code_switching\": false\n  }\n}\n```\n\n### Audio intelligence features\n\n| Feature | Pre-recorded | Live | Use for |\n|---------|--------------|------|---------|\n| Speaker diarization | ✓ | ✓ | Identify who said what |\n| Translation | ✓ | ✓ | Multi-language output |\n| Subtitles | ✓ | ✗ | SRT/VTT files |\n| Custom vocabulary | ✓ | ✓ | Domain-specific terms |\n| Custom spelling | ✓ | ✓ | Normalize misspellings |\n| PII redaction | ✓ | ✗ | GDPR/HIPAA compliance |\n| Sentiment analysis | ✓ | ✓ | Emotion detection |\n| Named entity recognition | ✓ | ✓ | Extract people, places, dates |\n| Summarization | ✓ | ✗ | Auto-generate summaries |\n| Chapterization | ✓ | ✗ | Break into sections |\n\n### Supported formats\n\n**Audio**: aac, ac3, eac3, flac, m4a, mp2, mp3, ogg, opus, wav  \n**Video**: 3g2, 3gp, avi, flv, m4v, matroska, mov, mp4, wmv  \n**Online**: TikTok, Instagram, Facebook, Vimeo, LinkedIn, Dailymotion, Sharechat, Likee\n\n### Limits\n\n| Limit | Free | Paid | Enterprise |\n|-------|------|------|------------|\n| Monthly usage | 10 hours | Unlimited | Unlimited |\n| Pre-recorded concurrency | 3 | 25 | On demand |\n| Live concurrency | 1 | 30 | On demand |\n| Max file duration | 135 min | 135 min | 4h 15m |\n| Max file size | 1000 MB | 1000 MB | 1000 MB |\n| Max live session | 3 hours | 3 hours | 3 hours |\n\n## Decision guidance\n\n### When to use pre-recorded vs. live\n\n| Scenario | Use |\n|----------|-----|\n| User uploads a file, wants transcript later | Pre-recorded (`/v2/pre-recorded`) |\n| Real-time transcription of streaming audio | Live (`/v2/live` WebSocket) |\n| Meeting recording to process after | Pre-recorded |\n| Live event, conference, or call transcription | Live |\n| Need subtitles or summarization | Pre-recorded (live doesn't support these) |\n\n### When to use SDK vs. raw API\n\n| Scenario | Use |\n|----------|-----|\n| Simple end-to-end transcription | SDK (handles upload, polling, retries) |\n| Fine-grained control over each step | Raw API (upload, create job, poll separately) |\n| Integrating into existing HTTP client | Raw API |\n| Building with Node.js or Python | SDK (better DX) |\n\n### Custom vocabulary vs. custom spelling\n\n| Scenario | Use |\n|----------|-----|\n| Model outputs garbled/phonetically wrong text | Custom vocabulary (phoneme-based matching) |\n| Model outputs recognizable but misspelled text | Custom spelling (literal text matching) |\n| Brand names or proper nouns | Custom vocabulary with pronunciations |\n| Normalizing variant spellings | Custom spelling |\n\n### Diarization vs. multi-channel audio\n\n| Scenario | Use |\n|----------|-----|\n| Single audio file, need to identify speakers | Diarization (`diarization: true`) |\n| Multiple separate audio tracks (e.g., Zoom participants) | Multi-channel (merge into one stream, set `channels: N`) |\n| Call recording with agent + customer | Diarization (simpler, set `number_of_speakers: 2`) |\n\n### Language detection vs. explicit language\n\n| Scenario | Use |\n|----------|-----|\n| Language is known in advance | Set `language_config.languages: [\"en\"]` (faster) |\n| Language is unknown | Omit `languages` (auto-detect) |\n| Multiple languages in one file | Set `languages: [\"en\", \"fr\"]` + `code_switching: true` |\n\n## Workflow\n\n### Pre-recorded transcription (SDK)\n\n1. **Initialize client**: `new GladiaClient({ apiKey: \"YOUR_KEY\" })`\n2. **Call transcribe()**: Pass audio URL or local path + options\n3. **Wait for result**: SDK polls until job completes\n4. **Extract data**: Access `transcription`, `translation`, `diarization`, etc. from result\n\n### Pre-recorded transcription (API)\n\n1. **Upload audio** (if local): `POST /v2/upload` → get `audio_url`\n2. **Create job**: `POST /v2/pre-recorded` with `audio_url` + config\n3. **Poll for result**: `GET /v2/pre-recorded/:id` until `status: \"done\"`\n4. **Parse response**: Extract transcription, audio intelligence results\n\n### Live transcription (SDK)\n\n1. **Initialize session**: `gladiaClient.liveV2().startSession(config)`\n2. **Register handlers**: Listen for `message`, `error`, `started`, `ended` events\n3. **Send audio chunks**: `liveSession.sendAudio(chunk)` as data arrives\n4. **Handle callbacks**: Process `transcript`, `translation`, `sentiment` messages in real-time\n5. **Stop recording**: `liveSession.stopRecording()` when done\n6. **Fetch final result** (optional): `GET /v2/live/:id` for complete output\n\n### Live transcription (API)\n\n1. **Initiate session**: `POST /v2/live` with encoding, sample rate, channels → get WebSocket URL\n2. **Connect WebSocket**: Open connection to returned URL\n3. **Send audio**: Send binary or base64-encoded audio chunks\n4. **Read messages**: Parse JSON messages (transcript, translation, etc.)\n5. **Stop recording**: Send `{ \"type\": \"stop_recording\" }` or close with code 1000\n6. **Fetch final result** (optional): `GET /v2/live/:id`\n\n### Adding audio intelligence\n\n1. **Identify features needed**: diarization, translation, sentiment, PII redaction, etc.\n2. **Add to request body**: Enable feature flag + config object\n3. **For pre-recorded**: Add at top level (e.g., `\"diarization\": true`, `\"translation\": true`)\n4. **For live**: Nest under `realtime_processing` (e.g., `\"realtime_processing\": { \"translation\": true }`)\n5. **Parse results**: Features appear in response under their own keys\n\n## Common gotchas\n\n- **solaria-3 with multiple languages**: Set exactly ONE language in `language_config.languages` (e.g., `[\"fr\"]`). Do not pass multiple languages or enable code switching with solaria-3.\n- **Live sessions limited to 3 hours**: After 3 hours, session terminates. Start a new session before hitting the limit.\n- **Polling without backoff**: Don't hammer the API. Implement exponential backoff or use webhooks/callbacks instead.\n- **Custom vocabulary intensity too high**: Start at 0.4–0.6. Higher values cause false positives across unrelated words.\n- **Forgetting to set encoding/sample_rate for live**: These are required to parse audio chunks correctly. Mismatch causes garbled transcription.\n- **Multi-channel audio billing**: Transcribing N-channel audio is billed as N × duration. Merge channels only if necessary.\n- **File size near 1000 MB**: Split into ~60-minute chunks before uploading. Larger files fail silently.\n- **Language detection on code-switched audio**: Don't enable code switching with an empty `languages` list. Specify 3–5 expected languages.\n- **PII redaction only for pre-recorded**: Not available for live transcription. Plan compliance workflows accordingly.\n- **Webhooks vs. callbacks**: Webhooks are configured in the dashboard; callbacks are per-request. Use callbacks for one-off jobs.\n\n## Verification checklist\n\nBefore submitting work with Gladia:\n\n- [ ] API key is set in `x-gladia-key` header (not in body or query params)\n- [ ] Audio URL is valid and accessible (or file uploaded successfully)\n- [ ] Model matches use case: `solaria-3` for pre-recorded, `solaria-1` for live\n- [ ] Language config is set correctly (explicit language or auto-detect, not both)\n- [ ] If using solaria-3 with multiple languages, set exactly one language\n- [ ] Diarization config (min/max speakers) is reasonable for the content\n- [ ] Custom vocabulary entries have pronunciations for phonetically wrong terms\n- [ ] PII redaction entity types match compliance requirements (GDPR, HIPAA, etc.)\n- [ ] Live session encoding/sample_rate/bit_depth match actual audio format\n- [ ] File size is under 1000 MB; duration under 135 min (or 4h 15m for enterprise)\n- [ ] Polling includes backoff or uses webhooks/callbacks\n- [ ] Response parsing handles both success and error states\n- [ ] Transcription quality is acceptable (test with sample audio first)\n\n## Resources\n\n**Comprehensive navigation**: https://docs.gladia.io/llms.txt\n\n**Critical pages**:\n- [Pre-recorded quickstart](https://docs.gladia.io/chapters/pre-recorded-stt/quickstart) — end-to-end transcription workflow\n- [Live quickstart](https://docs.gladia.io/chapters/live-stt/quickstart) — real-time transcription setup\n- [Audio intelligence features](https://docs.gladia.io/chapters/audio-intelligence/) — diarization, translation, sentiment, PII redaction, and more\n\n---\n\n> For additional documentation and navigation, see: https://docs.gladia.io/llms.txt\n---\n\n> This file is auto-synced from https://docs.gladia.io/.well-known/agent-skills/gladia/skill.md\n> Do not edit manually — changes will be overwritten by CI.\n> For additional documentation and navigation, see: https://docs.gladia.io/llms.txt\n\nFile v1.0.3:_meta.json\n\n{\n  \"ownerId\": \"kn7fs6cmj9hqg7232sbkacf31d882wp2\",\n  \"slug\": \"gladia-documentation-auto\",\n  \"version\": \"1.0.3\",\n  \"publishedAt\": 1783606047808\n}\n\nFile v1.0.3:skill-card.md\n\n## Description: <br>\nGladia Documentation Auto gives agents a comprehensive Gladia speech-to-text reference for transcription, audio intelligence, endpoint selection, and SDK-first workflow guidance. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[gladiaio](https://clawhub.ai/user/gladiaio) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and agents use this skill to build Gladia transcription and audio-analysis workflows, choose pre-recorded versus live APIs, configure SDK/API requests, and avoid common implementation errors. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill may guide agents toward Gladia transcription or audio-analysis behavior outside an explicit user request because the security summary notes a routing-scope concern. <br>\nMitigation: Use the skill only for explicit Gladia, transcription, speech-to-text, or audio-analysis tasks until the broad fallback wording is narrowed. <br>\nRisk: Transcription workflows may require Gladia credentials or may send audio to Gladia. <br>\nMitigation: Review whether a task requires a Gladia API key or audio upload, keep credentials out of generated code and logs, and get user confirmation before sending audio to the service. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/gladiaio/skills/gladia-documentation-auto) <br>\n- [Gladia publisher profile](https://clawhub.ai/user/gladiaio) <br>\n- [Server-resolved provenance unavailable](evidence.json#provenance) <br>\n- [Source skill metadata](https://docs.gladia.io/.well-known/agent-skills/gladia/skill.md) <br>\n- [Gladia documentation](https://docs.gladia.io) <br>\n- [Gladia documentation index](https://docs.gladia.io/llms.txt) <br>\n- [Pre-recorded quickstart](https://docs.gladia.io/chapters/pre-recorded-stt/quickstart) <br>\n- [Live quickstart](https://docs.gladia.io/chapters/live-stt/quickstart) <br>\n- [Audio intelligence features](https://docs.gladia.io/chapters/audio-intelligence/) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [guidance, markdown, code, shell commands, configuration] <br>\n**Output Format:** [Markdown guidance with code snippets, API examples, and configuration notes] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [SDK-first recommendations with raw REST/WebSocket fallback guidance when the SDK cannot satisfy the requirement.] <br>\n\n## Skill Version(s): <br>\n1.0.3 (source: server release metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.0.2: 3 files, 6774 bytes\n\nFiles: skill-card.md (2628b), SKILL.md (13202b), _meta.json (144b)\n\nFile v1.0.2:SKILL.md\n\n---\nname: gladia-documentation-auto\ndescription: Comprehensive Gladia speech-to-text reference auto-synced from docs.gladia.io. Use as a general-purpose fallback when other specialized skills don't match, or when the user needs a broad overview of Gladia capabilities, endpoints, decision guidance, or workflows. Always prefer the official SDK; fall back to raw REST/WebSocket only when SDK cannot satisfy the requirement.\nlicense: MIT\nmetadata:\n  source: https://docs.gladia.io/.well-known/agent-skills/gladia/skill.md\n  digest: sha256:cae5af7bbe95a1e0c1389bf07116cc038c7ff8f0ee23784588852047eb32be93\n  synced: \"2026-06-09\"\n---\n\n> **SDK-first**: always use the official SDK — see [gladia-sdk-integration](../gladia-sdk-integration/SKILL.md) for policy, setup, and fallback criteria.\n\n## References\n\nConsult these sibling skills as needed:\n\n- ../gladia-sdk-integration/SKILL.md -- SDK setup, client initialization, error handling, and SDK vs raw API decision guide\n- ../gladia-sdk-integration/references/sdk-versions.md -- Current SDK versions (auto-synced by CI)\n- ../gladia-troubleshooting/SKILL.md -- Common errors, gotchas, and verification checklist\n- ../gladia-live-transcription/SKILL.md -- Live streaming transcription\n- ../gladia-pre-recorded-transcription/SKILL.md -- Pre-recorded file transcription\n\n---\nname: Gladia\ndescription: Use when building speech transcription features, processing audio/video files, implementing real-time transcription, extracting insights from audio (speaker identification, translation, sentiment), or integrating voice capabilities into applications. Agents should reach for this skill when users request transcription, audio analysis, or voice-driven features.\nmetadata:\n    mintlify-proj: gladia\n    version: \"1.0\"\n---\n\n# Gladia Speech-to-Text API\n\n## Product summary\n\nGladia is a speech-to-text API that transcribes audio and video files in two modes: **pre-recorded** (asynchronous, file-based) and **live** (real-time, WebSocket-based). Beyond transcription, it provides audio intelligence features like speaker diarization, translation, sentiment analysis, PII redaction, and custom vocabulary matching. Agents use Gladia to build transcription workflows, extract structured data from audio, and power voice-driven applications.\n\n**Key files and commands:**\n- SDKs: JavaScript (`@gladiaio/sdk`) and Python (`gladiaio-sdk`)\n- Authentication: Pass `x-gladia-key` header with your API key\n- Pre-recorded endpoint: `POST /v2/pre-recorded` (create job), `GET /v2/pre-recorded/:id` (poll results)\n- Live endpoint: `POST /v2/live` (init session), WebSocket connection for streaming audio\n- Primary docs: https://docs.gladia.io\n\n## When to use\n\nReach for this skill when:\n- A user wants to transcribe audio or video files (meetings, podcasts, calls, interviews)\n- Building real-time transcription (voice agents, live captions, meeting recorders)\n- Extracting speaker information (who said what, speaker count, speaker identification)\n- Translating transcripts to other languages or generating subtitles\n- Detecting sensitive information (PII redaction for GDPR/HIPAA compliance)\n- Improving transcription accuracy with domain-specific vocabulary\n- Analyzing sentiment or emotions in speech\n- Integrating with third-party platforms (Twilio, Vapi, LiveKit, Pipecat, etc.)\n\n## Quick reference\n\n### Authentication\n```bash\n# All requests require the x-gladia-key header\ncurl --header 'x-gladia-key: YOUR_API_KEY' https://api.gladia.io/v2/...\n```\n\n### Pre-recorded workflow (file-based)\n1. **Upload** audio: `POST /v2/upload` → get `audio_url`\n2. **Create job**: `POST /v2/pre-recorded` with `audio_url` and options\n3. **Poll result**: `GET /v2/pre-recorded/:id` until `status: \"done\"`\n\nOr use SDK's `transcribe()` method for end-to-end in one call.\n\n### Live workflow (real-time)\n1. **Init session**: `POST /v2/live` with audio config (encoding, sample_rate, bit_depth, channels)\n2. **Connect WebSocket**: Use returned `url` to open WebSocket connection\n3. **Send audio**: Stream audio chunks as binary or base64-encoded JSON\n4. **Read messages**: Receive transcript, translation, sentiment, etc. via WebSocket\n5. **Stop**: Send `stop_recording` message; WebSocket closes when post-processing done\n\n### Audio formats supported\n| Type | Examples |\n|------|----------|\n| Audio | MP3, WAV, FLAC, AAC, OGG, Opus, M4A |\n| Video | MP4, MOV, AVI, WebM, Matroska |\n| Online | TikTok, Instagram, Facebook, Vimeo, LinkedIn, YouTube (via URL) |\n\n### File limits\n- **Pre-recorded**: Max 135 minutes (2h15m) or 1000 MB; enterprise plans support 4h15m\n- **Live**: Max 3 hours per session\n- **Recommendation**: Split files >60 minutes for better quality\n\n### Common audio intelligence features\n\n| Feature | Pre-recorded | Live | Use case |\n|---------|--------------|------|----------|\n| **Diarization** | ✓ | ✓ | Identify speakers, separate voices |\n| **Translation** | ✓ | ✓ | Translate to 100+ languages |\n| **Subtitles** | ✓ | - | Generate SRT/VTT files |\n| **Custom vocabulary** | ✓ | ✓ | Fix domain-specific terms |\n| **Custom spelling** | ✓ | ✓ | Normalize misspelled words |\n| **Sentiment analysis** | ✓ | ✓ | Detect sentiment & emotions |\n| **PII redaction** | ✓ | - | Mask sensitive data (GDPR/HIPAA) |\n| **Named entity recognition** | ✓ | ✓ | Extract people, places, dates |\n| **Summarization** | ✓ | - | Auto-generate summaries |\n| **Chapterization** | ✓ | - | Split into chapters/segments |\n\n## Decision guidance\n\n### When to use pre-recorded vs. live\n\n| Scenario | Use pre-recorded | Use live |\n|----------|------------------|----------|\n| User uploads a file to transcribe | ✓ | - |\n| Real-time transcription (voice agent, meeting) | - | ✓ |\n| Post-processing (subtitles, translation, summarization) | ✓ | - |\n| Low-latency response needed | - | ✓ |\n| Batch processing multiple files | ✓ | - |\n\n### When to use custom vocabulary vs. custom spelling\n\n| Situation | Use custom vocabulary | Use custom spelling |\n|-----------|----------------------|---------------------|\n| Model outputs garbled/phonetically wrong text | ✓ | - |\n| Model outputs recognizable but misspelled word | - | ✓ |\n| Domain-specific terms (brand names, jargon) | ✓ | - |\n| Normalizing variant spellings | - | ✓ |\n\n### When to use diarization vs. multi-channel audio\n\n| Scenario | Use diarization | Use multi-channel |\n|----------|-----------------|-------------------|\n| Single audio stream, multiple speakers | ✓ | - |\n| Separate audio tracks per speaker | - | ✓ |\n| Unknown number of speakers | ✓ | - |\n| Known speaker count and channels | - | ✓ |\n\n## Workflow\n\n### Pre-recorded transcription (typical task)\n\n1. **Understand requirements**: Confirm audio format, language, desired features (diarization, translation, subtitles, PII redaction).\n\n2. **Check file constraints**: Verify file is <1000 MB and <135 minutes (or split if needed).\n\n3. **Upload audio** (if local file):\n   ```javascript\n   const uploadResponse = await gladiaClient.preRecorded().uploadFile(\"path/to/audio.mp3\");\n   const audioUrl = uploadResponse.audio_url;\n   ```\n\n4. **Create transcription job** with options:\n   ```javascript\n   const job = await gladiaClient.preRecorded().createUntyped({\n     audio_url: audioUrl,\n     language_config: { languages: [\"en\"], code_switching: false },\n     diarization: true,\n     diarization_config: { min_speakers: 1, max_speakers: 5 },\n     custom_vocabulary: true,\n     custom_vocabulary_config: { vocabulary: [\"Gladia\", \"Solaria\"] },\n     translation: true,\n     translation_config: { target_languages: [\"fr\"], model: \"base\" },\n     sentiment_analysis: true,\n     pii_redaction: true,\n     pii_redaction_config: { entity_types: [\"GDPR\"] }\n   });\n   ```\n\n5. **Poll for results** (or use webhooks/callbacks):\n   ```javascript\n   let result = await gladiaClient.preRecorded().get(job.id);\n   while (result.status !== \"done\") {\n     await new Promise(r => setTimeout(r, 2000));\n     result = await gladiaClient.preRecorded().get(job.id);\n   }\n   ```\n\n6. **Extract and validate results**: Check `transcription.utterances`, `translation`, `sentiment_analysis`, `diarization` fields.\n\n7. **Verify output**: Confirm speaker attribution, translation accuracy, PII masking, and custom vocabulary replacements.\n\n### Live transcription (typical task)\n\n1. **Understand audio source**: Confirm encoding (wav/pcm, sample_rate, bit_depth, channels).\n\n2. **Initialize session**:\n   ```javascript\n   const liveSession = gladiaClient.liveV2().startSession({\n     model: \"solaria-1\",\n     encoding: \"wav/pcm\",\n     sample_rate: 16000,\n     bit_depth: 16,\n     channels: 1,\n     language_config: { languages: [\"en\"], code_switching: false },\n     messages_config: { receive_partial_transcripts: true }\n   });\n   ```\n\n3. **Connect WebSocket and set up handlers**:\n   ```javascript\n   liveSession.on(\"message\", (message) => {\n     if (message.type === \"transcript\" && message.data.is_final) {\n       console.log(message.data.utterance.text);\n     }\n   });\n   ```\n\n4. **Stream audio chunks** as they arrive:\n   ```javascript\n   liveSession.sendAudio(audioChunk);\n   ```\n\n5. **Stop recording** when done:\n   ```javascript\n   liveSession.stopRecording();\n   ```\n\n6. **Retrieve final results** (optional):\n   ```javascript\n   const result = await fetch(`https://api.gladia.io/v2/live/${sessionId}`, {\n     headers: { \"x-gladia-key\": apiKey }\n   });\n   ```\n\n## Common gotchas\n\n- **Empty language list with code switching**: Do not set `languages: []` and `code_switching: true` together. The detector will evaluate every utterance against 100+ languages, causing misdetections. Always provide a constrained list (3-5 languages max).\n\n- **Forgetting audio metadata**: For live transcription, `encoding`, `sample_rate`, `bit_depth`, and `channels` must match your actual audio stream. Mismatches cause garbled output.\n\n- **Custom vocabulary intensity too high**: Start at `intensity: 0.4` and raise only if terms are missed. High intensity causes false positives (unrelated words get replaced). Add `pronunciations` variants before raising intensity.\n\n- **Polling without backoff**: Don't hammer the API with rapid polls. Use 2-3 second intervals or webhooks/callbacks instead.\n\n- **Exceeding file limits silently**: Pre-recorded files >135 minutes or >1000 MB will fail. Split large files into ~60-minute chunks before uploading.\n\n- **Not setting language when known**: If you know the language, set `languages: [\"en\"]` explicitly. Omitting it forces detection, adding latency and risk of misdetection.\n\n- **Diarization without hints**: If you know the speaker count, set `number_of_speakers` or `min_speakers`/`max_speakers`. Hints improve accuracy.\n\n- **PII redaction only for pre-recorded**: PII redaction is not available for live transcription. Plan accordingly for compliance workflows.\n\n- **Webhook/callback URL not reachable**: If using webhooks, ensure your callback URL is publicly accessible and returns 2xx status. Gladia will retry failed deliveries.\n\n- **Multi-channel audio billing**: Transcribing multi-channel audio is billed by total duration × number of channels. A 1-hour 3-channel stream costs 3 hours of transcription.\n\n## Verification checklist\n\nBefore submitting transcription work:\n\n- [ ] Audio file is valid format (MP3, WAV, MP4, etc.) and <1000 MB\n- [ ] File duration is <135 minutes (or split if longer)\n- [ ] API key is valid and has `x-gladia-key` header set\n- [ ] Language is set explicitly if known; avoid empty `languages` with `code_switching: true`\n- [ ] Custom vocabulary entries are tested; intensity is 0.4-0.6 unless tuned\n- [ ] Diarization hints (min/max speakers) are provided if speaker count is known\n- [ ] Webhook/callback URL (if used) is publicly accessible and returns 2xx\n- [ ] Results include expected fields: `transcription.utterances`, `translation`, `sentiment_analysis`, etc.\n- [ ] Speaker attribution is correct (diarization `speaker` field matches expected speakers)\n- [ ] PII redaction is applied (if required) and sensitive data is masked\n- [ ] Translation accuracy is spot-checked for domain-specific terms\n- [ ] Subtitles (if generated) have correct timing and formatting\n\n## Resources\n\n- **Comprehensive page listing**: https://docs.gladia.io/llms.txt\n- **Getting started**: https://docs.gladia.io/chapters/introduction/getting-started\n- **Pre-recorded quickstart**: https://docs.gladia.io/chapters/pre-recorded-stt/quickstart\n- **Live transcription quickstart**: https://docs.gladia.io/chapters/live-stt/quickstart\n- **Audio intelligence features**: https://docs.gladia.io/chapters/audio-intelligence/\n- **Recommended parameters by use case**: https://docs.gladia.io/chapters/pre-recorded-stt/recommended-parameters\n- **API reference**: https://docs.gladia.io/api-reference/\n- **SDK documentation**: https://docs.gladia.io/chapters/integrations/sdk\n\n---\n\n> For additional documentation and navigation, see: https://docs.gladia.io/llms.txt\n---\n\n> This file is auto-synced from https://docs.gladia.io/.well-known/agent-skills/gladia/skill.md\n> Do not edit manually — changes will be overwritten by CI.\n> For additional documentation and navigation, see: https://docs.gladia.io/llms.txt\n\nFile v1.0.2:_meta.json\n\n{\n  \"ownerId\": \"kn7fs6cmj9hqg7232sbkacf31d882wp2\",\n  \"slug\": \"gladia-documentation-auto\",\n  \"version\": \"1.0.2\",\n  \"publishedAt\": 1781005333972\n}\n\nFile v1.0.2:skill-card.md\n\n## Description: <br>\nComprehensive Gladia speech-to-text reference auto-synced from docs.gladia.io for agents that need broad guidance on Gladia capabilities, endpoints, decision criteria, and workflows. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[gladiaio](https://clawhub.ai/user/gladiaio) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and engineers use this skill to choose and implement Gladia speech-to-text workflows for pre-recorded files, live transcription, and audio intelligence features such as diarization, translation, sentiment analysis, and PII redaction. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Audio and video recordings sent for transcription may contain sensitive information. <br>\nMitigation: Confirm the recordings are approved for Gladia processing and avoid uploading highly sensitive content unless explicitly authorized. <br>\nRisk: Gladia API keys are sensitive credentials. <br>\nMitigation: Store and handle the API key as a secret, and avoid exposing it in logs, prompts, code examples, or committed files. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/gladiaio/gladia-documentation-auto) <br>\n- [Gladia agent skill source](https://docs.gladia.io/.well-known/agent-skills/gladia/skill.md) <br>\n- [Gladia documentation](https://docs.gladia.io) <br>\n- [Gladia documentation index](https://docs.gladia.io/llms.txt) <br>\n- [Pre-recorded transcription quickstart](https://docs.gladia.io/chapters/pre-recorded-stt/quickstart) <br>\n- [Live transcription quickstart](https://docs.gladia.io/chapters/live-stt/quickstart) <br>\n- [Gladia API reference](https://docs.gladia.io/api-reference/) <br>\n- [Gladia SDK documentation](https://docs.gladia.io/chapters/integrations/sdk) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, code, shell commands, configuration, guidance] <br>\n**Output Format:** [Markdown with code examples, endpoint references, workflow guidance, and configuration notes] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [May include Gladia API requests, SDK usage examples, transcription workflow checks, and validation guidance] <br>\n\n## Skill Version(s): <br>\n1.0.2 (source: server release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.0.1: 3 files, 7159 bytes\n\nFiles: skill-card.md (3298b), SKILL.md (13288b), _meta.json (144b)\n\nFile v1.0.1:SKILL.md\n\n---\nname: gladia-documentation-auto\ndescription: Comprehensive Gladia speech-to-text reference auto-synced from docs.gladia.io. Use as a general-purpose fallback when other specialized skills don't match, or when the user needs a broad overview of Gladia capabilities, endpoints, decision guidance, or workflows. Always prefer the official SDK; fall back to raw REST/WebSocket only when SDK cannot satisfy the requirement.\nlicense: MIT\nmetadata:\n  source: https://docs.gladia.io/.well-known/agent-skills/gladia/skill.md\n  digest: sha256:f87953eda33c8e6132d8a78343532fddc576ebeab7bf83bb3b6fb3aca2b5c96b\n  synced: \"2026-06-04\"\n---\n\n> **SDK-first**: always use the official SDK — see [gladia-sdk-integration](../gladia-sdk-integration/SKILL.md) for policy, setup, and fallback criteria.\n\n## References\n\nConsult these sibling skills as needed:\n\n- ../gladia-sdk-integration/SKILL.md -- SDK setup, client initialization, error handling, and SDK vs raw API decision guide\n- ../gladia-sdk-integration/references/sdk-versions.md -- Current SDK versions (auto-synced by CI)\n- ../gladia-troubleshooting/SKILL.md -- Common errors, gotchas, and verification checklist\n- ../gladia-live-transcription/SKILL.md -- Live streaming transcription\n- ../gladia-pre-recorded-transcription/SKILL.md -- Pre-recorded file transcription\n\n---\nname: Gladia\ndescription: Use when building speech-to-text transcription features, processing audio or video files, implementing real-time transcription, extracting insights from audio (translation, summarization, speaker identification), or integrating audio intelligence into applications.\nmetadata:\n    mintlify-proj: gladia\n    version: \"1.0\"\n---\n\n# Gladia Skill\n\n## Product summary\n\nGladia is a speech-to-text API that transcribes audio and video files in two modes: **pre-recorded** (asynchronous, batch) and **live** (real-time, WebSocket-based). The API returns structured transcripts with word-level timing, confidence scores, and optional audio intelligence features (translation, diarization, summarization, entity recognition, sentiment analysis, PII redaction, subtitles). Use the JavaScript/TypeScript SDK (`@gladiaio/sdk`) or Python SDK (`gladiaio-sdk`) for simplified integration, or call REST/WebSocket endpoints directly. Authenticate with `x-gladia-key` header. Primary docs: https://docs.gladia.io\n\n## When to use\n\n- **Pre-recorded transcription**: Transcribe uploaded audio/video files (MP3, WAV, MP4, YouTube links, etc.) asynchronously. Typical latency: seconds to minutes depending on file length.\n- **Live transcription**: Stream audio in real-time via WebSocket for immediate transcripts (e.g., call centers, live events, voice assistants).\n- **Audio intelligence**: Extract metadata from transcripts — translate to multiple languages, identify speakers, detect sentiment, redact PII, generate summaries, create subtitles, recognize named entities.\n- **Custom vocabulary**: Improve accuracy for domain-specific terms, brand names, proper nouns by providing phonetic hints.\n- **Multi-speaker scenarios**: Use diarization to attribute speech to individual speakers, or send multi-channel audio to preserve speaker identity.\n\n## Quick reference\n\n### Authentication\n```bash\n# All requests require x-gladia-key header\ncurl -H \"x-gladia-key: YOUR_API_KEY\" https://api.gladia.io/v2/pre-recorded\n```\n\n### Pre-recorded workflow (SDK)\n```javascript\nimport { GladiaClient } from \"@gladiaio/sdk\";\nconst client = new GladiaClient({ apiKey: \"YOUR_KEY\" });\nconst result = await client.preRecorded().transcribe(\"audio_url_or_local_path\");\n```\n\n### Live workflow (SDK)\n```javascript\nconst session = client.liveV2().startSession({\n  encoding: \"wav/pcm\",\n  sample_rate: 16000,\n  bit_depth: 16,\n  channels: 1,\n  language_config: { languages: [\"en\"] }\n});\nsession.on(\"message\", (msg) => console.log(msg));\nsession.sendAudio(audioChunk);\nsession.stopRecording();\n```\n\n### Audio formats\n| Type | Examples |\n|------|----------|\n| Audio | MP3, WAV, FLAC, AAC, OGG, Opus |\n| Video | MP4, MOV, AVI, WebM, Matroska |\n| Online | YouTube, TikTok, Instagram, Facebook, Vimeo, LinkedIn |\n\n### Limits\n| Limit | Value |\n|-------|-------|\n| Pre-recorded max duration | 135 minutes (free/paid); 4h15 (enterprise) |\n| Pre-recorded max file size | 1000 MB |\n| Live session max duration | 3 hours |\n| Free tier monthly usage | 10 hours |\n| Concurrent pre-recorded jobs (free) | 3 |\n| Concurrent pre-recorded jobs (paid) | 25 |\n| Concurrent live sessions (free) | 1 |\n| Concurrent live sessions (paid) | 30 |\n\n### Audio intelligence features\n| Feature | Pre-recorded | Live | Purpose |\n|---------|--------------|------|---------|\n| Diarization | ✓ | ✗ | Identify speakers |\n| Translation | ✓ | ✓ | Multi-language output |\n| Summarization | ✓ | ✗ | Generate summaries/bullet points |\n| Sentiment analysis | ✓ | ✓ | Detect emotions and tone |\n| Named entity recognition | ✓ | ✓ | Extract people, orgs, dates |\n| PII redaction | ✓ | ✗ | Anonymize sensitive data |\n| Subtitles | ✓ | ✗ | Generate SRT/VTT files |\n| Custom vocabulary | ✓ | ✓ | Improve domain-specific terms |\n| Custom spelling | ✓ | ✓ | Normalize misspellings |\n| Chapterization | ✓ | ✗ | Segment long audio into chapters |\n| Audio-to-LLM | ✓ | ✗ | Run custom prompts on transcript |\n\n## Decision guidance\n\n### When to use pre-recorded vs. live\n\n| Scenario | Pre-recorded | Live |\n|----------|--------------|------|\n| Batch processing uploaded files | ✓ | ✗ |\n| Real-time streaming (calls, events) | ✗ | ✓ |\n| Need diarization | ✓ | ✗ |\n| Need immediate partial results | ✗ | ✓ (with `receive_partial_transcripts: true`) |\n| Need summarization | ✓ | ✗ |\n| Multi-hour content | ✓ (up to 135 min) | ✓ (up to 3 hours per session) |\n\n### When to use SDK vs. raw API\n\n| Approach | Best for |\n|----------|----------|\n| SDK | Rapid development, automatic error handling, built-in polling/retry logic |\n| Raw API | Custom workflows, specific language/framework, fine-grained control |\n\n### When to use diarization vs. multi-channel audio\n\n| Approach | Use when |\n|----------|----------|\n| Diarization | Single audio file with multiple speakers; you want the API to separate them |\n| Multi-channel | Multiple audio sources (e.g., separate participant feeds); you can merge them into one multi-channel stream |\n\n### When to use custom vocabulary vs. custom spelling\n\n| Feature | Use when |\n|---------|----------|\n| Custom vocabulary | Word is mispronounced/garbled; you provide phonetic hints (e.g., \"Nietzsche\" → [\"Niche\", \"Neechee\"]) |\n| Custom spelling | Word is recognized but misspelled (e.g., \"Salesforce\" → \"Sales Force\"); literal text matching |\n\n## Workflow\n\n### Pre-recorded transcription (typical task)\n\n1. **Prepare audio**: Ensure file is under 1000 MB and 135 minutes. Supported formats: MP3, WAV, MP4, YouTube URL, etc.\n2. **Choose delivery method**: Use SDK for simplicity, or raw API for control.\n3. **Configure transcription**:\n   - Set `language_config.languages` explicitly if known (avoids detection overhead).\n   - Enable `diarization: true` if multiple speakers.\n   - Add `custom_vocabulary` for domain terms.\n   - Enable audio intelligence features (translation, summarization, etc.) as needed.\n4. **Submit job**: Call `transcribe()` (SDK) or `POST /v2/pre-recorded` (API).\n5. **Retrieve results**: Poll `GET /v2/pre-recorded/:id` or configure webhooks/callbacks.\n6. **Parse response**: Extract `transcription.utterances[]` for text and timing, plus any audio intelligence results.\n\n### Live transcription (typical task)\n\n1. **Initialize session**: Call `POST /v2/live` with audio config (encoding, sample_rate, bit_depth, channels).\n2. **Connect WebSocket**: Use returned URL to open WebSocket connection.\n3. **Configure messages**: Set `messages_config` to specify which message types to receive (transcripts, partial transcripts, post-processing events).\n4. **Stream audio**: Send audio chunks via `sendAudio()` (SDK) or binary/base64 JSON (raw API).\n5. **Handle messages**: Listen for `transcript` messages; check `is_final` to distinguish partials from finals.\n6. **Stop recording**: Call `stopRecording()` to trigger post-processing (diarization, translation, etc.).\n7. **Retrieve final result**: Poll `GET /v2/live/:id` or wait for callback with complete result.\n\n### Adding custom vocabulary\n\n1. **Identify problem terms**: Transcribe without custom vocabulary; note mis-transcribed words.\n2. **Categorize**: Garbled/phonetically wrong → custom vocabulary; recognizable but misspelled → custom spelling.\n3. **Build vocabulary list**:\n   ```json\n   {\n     \"custom_vocabulary\": true,\n     \"custom_vocabulary_config\": {\n       \"vocabulary\": [\n         \"Gladia\",\n         { \"value\": \"Salesforce\", \"pronunciations\": [\"sell force\"], \"intensity\": 0.5 }\n       ],\n       \"default_intensity\": 0.4\n     }\n   }\n   ```\n4. **Test**: Transcribe again; confirm targets appear and check for false positives.\n5. **Refine**: Lower intensity, add pronunciations, or move stubborn terms to custom spelling.\n\n## Common gotchas\n\n- **Language detection overhead**: Always set `language_config.languages` explicitly if you know the language. Auto-detection adds latency and can fail if audio starts with silence or music.\n- **Code switching without language list**: Never enable `code_switching: true` with an empty `languages` array — the model will evaluate against 100+ languages, causing frequent misdetections. Always provide a constrained list (3–5 languages).\n- **Diarization hints are not hard constraints**: `number_of_speakers`, `min_speakers`, `max_speakers` are hints, not guarantees. The model may detect a different count.\n- **Custom vocabulary intensity tuning**: Start at `default_intensity: 0.4` and adjust per-entry only. Raising intensity globally increases false positives. Add `pronunciations` variants before raising intensity.\n- **Live session 3-hour limit**: A single WebSocket session cannot exceed 3 hours. For longer events, close the session and start a new one before hitting the limit.\n- **Pre-recorded 135-minute limit**: Files longer than 135 minutes will fail. Split into ~60-minute chunks using ffmpeg or similar tools.\n- **Audio format conversion overhead**: Large video files (e.g., AVI, MOV) take ~1 minute to convert to WAV/PCM. Plan for this latency.\n- **Polling without webhooks**: If you poll `GET /v2/pre-recorded/:id` in a tight loop, you'll hit rate limits. Use webhooks or callbacks instead, or poll with exponential backoff.\n- **Multi-channel billing**: Transcribing multi-channel audio is billed as `duration × number_of_channels`. A 10-minute 3-channel stream costs 30 minutes of usage.\n- **Partial transcripts in live mode**: Partial transcripts are low-latency but less accurate. Always check `is_final: true` before using a transcript for critical decisions.\n- **Missing audio_url on upload**: After uploading a file, the response includes `audio_url` — use this URL in the transcription request, not the local file path.\n- **WebSocket reconnection**: If the WebSocket disconnects, reconnect to the same URL (returned from init) to resume the session without losing context.\n\n## Verification checklist\n\nBefore submitting transcription work:\n\n- [ ] API key is valid and passed in `x-gladia-key` header.\n- [ ] Audio file is under 1000 MB and 135 minutes (pre-recorded) or 3 hours (live).\n- [ ] Audio format is supported (MP3, WAV, MP4, etc.).\n- [ ] Language is set explicitly in `language_config.languages` if known.\n- [ ] If using code switching, `languages` list is constrained to 3–5 expected languages.\n- [ ] Diarization is enabled if multiple speakers need attribution.\n- [ ] Custom vocabulary entries have realistic `intensity` (0.4–0.6) and `pronunciations`.\n- [ ] Webhooks or callbacks are configured if polling is not feasible.\n- [ ] Live sessions are closed before 3 hours; pre-recorded jobs are split if over 135 minutes.\n- [ ] Response includes expected fields: `transcription.utterances[]`, `metadata`, and any requested audio intelligence results.\n- [ ] Confidence scores and timing (`start`, `end`) are present for quality validation.\n- [ ] Multi-channel audio is correctly interleaved if merging multiple sources.\n\n## Resources\n\n- **Comprehensive page listing**: https://docs.gladia.io/llms.txt\n- **Getting started guide**: https://docs.gladia.io/chapters/introduction/getting-started\n- **Pre-recorded quickstart**: https://docs.gladia.io/chapters/pre-recorded-stt/quickstart\n- **Live transcription quickstart**: https://docs.gladia.io/chapters/live-stt/quickstart\n- **API reference**: https://docs.gladia.io/api-reference\n- **Recommended parameters by use case**: https://docs.gladia.io/chapters/pre-recorded-stt/recommended-parameters\n- **Audio intelligence features**: https://docs.gladia.io/chapters/audio-intelligence\n- **Supported formats and limits**: https://docs.gladia.io/chapters/limits-and-specifications/supported-formats\n\n---\n\n> For additional documentation and navigation, see: https://docs.gladia.io/llms.txt\n---\n\n> This file is auto-synced from https://docs.gladia.io/.well-known/agent-skills/gladia/skill.md\n> Do not edit manually — changes will be overwritten by CI.\n> For additional documentation and navigation, see: https://docs.gladia.io/llms.txt\n\nFile v1.0.1:_meta.json\n\n{\n  \"ownerId\": \"kn7fs6cmj9hqg7232sbkacf31d882wp2\",\n  \"slug\": \"gladia-documentation-auto\",\n  \"version\": \"1.0.1\",\n  \"publishedAt\": 1780942626069\n}\n\nFile v1.0.1:skill-card.md\n\n## Description: <br>\nComprehensive Gladia speech-to-text reference auto-synced from docs.gladia.io for broad guidance on Gladia capabilities, endpoints, decisions, and workflows. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[gladiaio](https://clawhub.ai/user/gladiaio) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and engineers use this skill to choose and implement Gladia speech-to-text workflows for pre-recorded files, live WebSocket transcription, and audio intelligence features. It helps agents produce integration guidance, API examples, SDK-first recommendations, and verification checklists for Gladia deployments. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Using the documented workflows can send audio, video, transcripts, and metadata to Gladia. <br>\nMitigation: Process only media the user is authorized to upload or stream, and review data handling requirements before using Gladia APIs. <br>\nRisk: The skill requires a Gladia API key for real integrations. <br>\nMitigation: Keep API keys out of source code, prompts, and logs; pass them through approved secret management or environment configuration. <br>\nRisk: Generated transcription guidance can be inaccurate if service limits or audio constraints are ignored. <br>\nMitigation: Check file size, duration, format, language settings, and live session limits before submitting transcription jobs. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/gladiaio/gladia-documentation-auto) <br>\n- [Gladia agent skill source](https://docs.gladia.io/.well-known/agent-skills/gladia/skill.md) <br>\n- [Gladia documentation](https://docs.gladia.io) <br>\n- [Gladia documentation index](https://docs.gladia.io/llms.txt) <br>\n- [Getting started guide](https://docs.gladia.io/chapters/introduction/getting-started) <br>\n- [Pre-recorded transcription quickstart](https://docs.gladia.io/chapters/pre-recorded-stt/quickstart) <br>\n- [Live transcription quickstart](https://docs.gladia.io/chapters/live-stt/quickstart) <br>\n- [Gladia API reference](https://docs.gladia.io/api-reference) <br>\n- [Recommended parameters by use case](https://docs.gladia.io/chapters/pre-recorded-stt/recommended-parameters) <br>\n- [Audio intelligence features](https://docs.gladia.io/chapters/audio-intelligence) <br>\n- [Supported formats and limits](https://docs.gladia.io/chapters/limits-and-specifications/supported-formats) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, code, shell commands, configuration, guidance] <br>\n**Output Format:** [Markdown guidance with code, shell command, JSON, and configuration examples] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Documentation-only output; generated guidance may reference Gladia SDKs, REST endpoints, WebSocket workflows, and verification checklists.] <br>\n\n## Skill Version(s): <br>\n1.0.1 (source: server release metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.0.0: 3 files, 6953 bytes\n\nFiles: skill-card.md (2437b), SKILL.md (13284b), _meta.json (144b)\n\nFile v1.0.0:SKILL.md\n\nname: gladia-documentation-auto\ndescription: Comprehensive Gladia speech-to-text reference auto-synced from docs.gladia.io. Use as a general-purpose fallback when other specialized skills don't match, or when the user needs a broad overview of Gladia capabilities, endpoints, decision guidance, or workflows. Always prefer the official SDK; fall back to raw REST/WebSocket only when SDK cannot satisfy the requirement.\nlicense: MIT\nmetadata:\n  source: https://docs.gladia.io/.well-known/agent-skills/gladia/skill.md\n  digest: sha256:f87953eda33c8e6132d8a78343532fddc576ebeab7bf83bb3b6fb3aca2b5c96b\n  synced: \"2026-06-04\"\n---\n\n> **SDK-first**: always use the official SDK — see [gladia-sdk-integration](../gladia-sdk-integration/SKILL.md) for policy, setup, and fallback criteria.\n\n## References\n\nConsult these sibling skills as needed:\n\n- ../gladia-sdk-integration/SKILL.md -- SDK setup, client initialization, error handling, and SDK vs raw API decision guide\n- ../gladia-sdk-integration/references/sdk-versions.md -- Current SDK versions (auto-synced by CI)\n- ../gladia-troubleshooting/SKILL.md -- Common errors, gotchas, and verification checklist\n- ../gladia-live-transcription/SKILL.md -- Live streaming transcription\n- ../gladia-pre-recorded-transcription/SKILL.md -- Pre-recorded file transcription\n\n---\nname: Gladia\ndescription: Use when building speech-to-text transcription features, processing audio or video files, implementing real-time transcription, extracting insights from audio (translation, summarization, speaker identification), or integrating audio intelligence into applications.\nmetadata:\n    mintlify-proj: gladia\n    version: \"1.0\"\n---\n\n# Gladia Skill\n\n## Product summary\n\nGladia is a speech-to-text API that transcribes audio and video files in two modes: **pre-recorded** (asynchronous, batch) and **live** (real-time, WebSocket-based). The API returns structured transcripts with word-level timing, confidence scores, and optional audio intelligence features (translation, diarization, summarization, entity recognition, sentiment analysis, PII redaction, subtitles). Use the JavaScript/TypeScript SDK (`@gladiaio/sdk`) or Python SDK (`gladiaio-sdk`) for simplified integration, or call REST/WebSocket endpoints directly. Authenticate with `x-gladia-key` header. Primary docs: https://docs.gladia.io\n\n## When to use\n\n- **Pre-recorded transcription**: Transcribe uploaded audio/video files (MP3, WAV, MP4, YouTube links, etc.) asynchronously. Typical latency: seconds to minutes depending on file length.\n- **Live transcription**: Stream audio in real-time via WebSocket for immediate transcripts (e.g., call centers, live events, voice assistants).\n- **Audio intelligence**: Extract metadata from transcripts — translate to multiple languages, identify speakers, detect sentiment, redact PII, generate summaries, create subtitles, recognize named entities.\n- **Custom vocabulary**: Improve accuracy for domain-specific terms, brand names, proper nouns by providing phonetic hints.\n- **Multi-speaker scenarios**: Use diarization to attribute speech to individual speakers, or send multi-channel audio to preserve speaker identity.\n\n## Quick reference\n\n### Authentication\n```bash\n# All requests require x-gladia-key header\ncurl -H \"x-gladia-key: YOUR_API_KEY\" https://api.gladia.io/v2/pre-recorded\n```\n\n### Pre-recorded workflow (SDK)\n```javascript\nimport { GladiaClient } from \"@gladiaio/sdk\";\nconst client = new GladiaClient({ apiKey: \"YOUR_KEY\" });\nconst result = await client.preRecorded().transcribe(\"audio_url_or_local_path\");\n```\n\n### Live workflow (SDK)\n```javascript\nconst session = client.liveV2().startSession({\n  encoding: \"wav/pcm\",\n  sample_rate: 16000,\n  bit_depth: 16,\n  channels: 1,\n  language_config: { languages: [\"en\"] }\n});\nsession.on(\"message\", (msg) => console.log(msg));\nsession.sendAudio(audioChunk);\nsession.stopRecording();\n```\n\n### Audio formats\n| Type | Examples |\n|------|----------|\n| Audio | MP3, WAV, FLAC, AAC, OGG, Opus |\n| Video | MP4, MOV, AVI, WebM, Matroska |\n| Online | YouTube, TikTok, Instagram, Facebook, Vimeo, LinkedIn |\n\n### Limits\n| Limit | Value |\n|-------|-------|\n| Pre-recorded max duration | 135 minutes (free/paid); 4h15 (enterprise) |\n| Pre-recorded max file size | 1000 MB |\n| Live session max duration | 3 hours |\n| Free tier monthly usage | 10 hours |\n| Concurrent pre-recorded jobs (free) | 3 |\n| Concurrent pre-recorded jobs (paid) | 25 |\n| Concurrent live sessions (free) | 1 |\n| Concurrent live sessions (paid) | 30 |\n\n### Audio intelligence features\n| Feature | Pre-recorded | Live | Purpose |\n|---------|--------------|------|---------|\n| Diarization | ✓ | ✗ | Identify speakers |\n| Translation | ✓ | ✓ | Multi-language output |\n| Summarization | ✓ | ✗ | Generate summaries/bullet points |\n| Sentiment analysis | ✓ | ✓ | Detect emotions and tone |\n| Named entity recognition | ✓ | ✓ | Extract people, orgs, dates |\n| PII redaction | ✓ | ✗ | Anonymize sensitive data |\n| Subtitles | ✓ | ✗ | Generate SRT/VTT files |\n| Custom vocabulary | ✓ | ✓ | Improve domain-specific terms |\n| Custom spelling | ✓ | ✓ | Normalize misspellings |\n| Chapterization | ✓ | ✗ | Segment long audio into chapters |\n| Audio-to-LLM | ✓ | ✗ | Run custom prompts on transcript |\n\n## Decision guidance\n\n### When to use pre-recorded vs. live\n\n| Scenario | Pre-recorded | Live |\n|----------|--------------|------|\n| Batch processing uploaded files | ✓ | ✗ |\n| Real-time streaming (calls, events) | ✗ | ✓ |\n| Need diarization | ✓ | ✗ |\n| Need immediate partial results | ✗ | ✓ (with `receive_partial_transcripts: true`) |\n| Need summarization | ✓ | ✗ |\n| Multi-hour content | ✓ (up to 135 min) | ✓ (up to 3 hours per session) |\n\n### When to use SDK vs. raw API\n\n| Approach | Best for |\n|----------|----------|\n| SDK | Rapid development, automatic error handling, built-in polling/retry logic |\n| Raw API | Custom workflows, specific language/framework, fine-grained control |\n\n### When to use diarization vs. multi-channel audio\n\n| Approach | Use when |\n|----------|----------|\n| Diarization | Single audio file with multiple speakers; you want the API to separate them |\n| Multi-channel | Multiple audio sources (e.g., separate participant feeds); you can merge them into one multi-channel stream |\n\n### When to use custom vocabulary vs. custom spelling\n\n| Feature | Use when |\n|---------|----------|\n| Custom vocabulary | Word is mispronounced/garbled; you provide phonetic hints (e.g., \"Nietzsche\" → [\"Niche\", \"Neechee\"]) |\n| Custom spelling | Word is recognized but misspelled (e.g., \"Salesforce\" → \"Sales Force\"); literal text matching |\n\n## Workflow\n\n### Pre-recorded transcription (typical task)\n\n1. **Prepare audio**: Ensure file is under 1000 MB and 135 minutes. Supported formats: MP3, WAV, MP4, YouTube URL, etc.\n2. **Choose delivery method**: Use SDK for simplicity, or raw API for control.\n3. **Configure transcription**:\n   - Set `language_config.languages` explicitly if known (avoids detection overhead).\n   - Enable `diarization: true` if multiple speakers.\n   - Add `custom_vocabulary` for domain terms.\n   - Enable audio intelligence features (translation, summarization, etc.) as needed.\n4. **Submit job**: Call `transcribe()` (SDK) or `POST /v2/pre-recorded` (API).\n5. **Retrieve results**: Poll `GET /v2/pre-recorded/:id` or configure webhooks/callbacks.\n6. **Parse response**: Extract `transcription.utterances[]` for text and timing, plus any audio intelligence results.\n\n### Live transcription (typical task)\n\n1. **Initialize session**: Call `POST /v2/live` with audio config (encoding, sample_rate, bit_depth, channels).\n2. **Connect WebSocket**: Use returned URL to open WebSocket connection.\n3. **Configure messages**: Set `messages_config` to specify which message types to receive (transcripts, partial transcripts, post-processing events).\n4. **Stream audio**: Send audio chunks via `sendAudio()` (SDK) or binary/base64 JSON (raw API).\n5. **Handle messages**: Listen for `transcript` messages; check `is_final` to distinguish partials from finals.\n6. **Stop recording**: Call `stopRecording()` to trigger post-processing (diarization, translation, etc.).\n7. **Retrieve final result**: Poll `GET /v2/live/:id` or wait for callback with complete result.\n\n### Adding custom vocabulary\n\n1. **Identify problem terms**: Transcribe without custom vocabulary; note mis-transcribed words.\n2. **Categorize**: Garbled/phonetically wrong → custom vocabulary; recognizable but misspelled → custom spelling.\n3. **Build vocabulary list**:\n   ```json\n   {\n     \"custom_vocabulary\": true,\n     \"custom_vocabulary_config\": {\n       \"vocabulary\": [\n         \"Gladia\",\n         { \"value\": \"Salesforce\", \"pronunciations\": [\"sell force\"], \"intensity\": 0.5 }\n       ],\n       \"default_intensity\": 0.4\n     }\n   }\n   ```\n4. **Test**: Transcribe again; confirm targets appear and check for false positives.\n5. **Refine**: Lower intensity, add pronunciations, or move stubborn terms to custom spelling.\n\n## Common gotchas\n\n- **Language detection overhead**: Always set `language_config.languages` explicitly if you know the language. Auto-detection adds latency and can fail if audio starts with silence or music.\n- **Code switching without language list**: Never enable `code_switching: true` with an empty `languages` array — the model will evaluate against 100+ languages, causing frequent misdetections. Always provide a constrained list (3–5 languages).\n- **Diarization hints are not hard constraints**: `number_of_speakers`, `min_speakers`, `max_speakers` are hints, not guarantees. The model may detect a different count.\n- **Custom vocabulary intensity tuning**: Start at `default_intensity: 0.4` and adjust per-entry only. Raising intensity globally increases false positives. Add `pronunciations` variants before raising intensity.\n- **Live session 3-hour limit**: A single WebSocket session cannot exceed 3 hours. For longer events, close the session and start a new one before hitting the limit.\n- **Pre-recorded 135-minute limit**: Files longer than 135 minutes will fail. Split into ~60-minute chunks using ffmpeg or similar tools.\n- **Audio format conversion overhead**: Large video files (e.g., AVI, MOV) take ~1 minute to convert to WAV/PCM. Plan for this latency.\n- **Polling without webhooks**: If you poll `GET /v2/pre-recorded/:id` in a tight loop, you'll hit rate limits. Use webhooks or callbacks instead, or poll with exponential backoff.\n- **Multi-channel billing**: Transcribing multi-channel audio is billed as `duration × number_of_channels`. A 10-minute 3-channel stream costs 30 minutes of usage.\n- **Partial transcripts in live mode**: Partial transcripts are low-latency but less accurate. Always check `is_final: true` before using a transcript for critical decisions.\n- **Missing audio_url on upload**: After uploading a file, the response includes `audio_url` — use this URL in the transcription request, not the local file path.\n- **WebSocket reconnection**: If the WebSocket disconnects, reconnect to the same URL (returned from init) to resume the session without losing context.\n\n## Verification checklist\n\nBefore submitting transcription work:\n\n- [ ] API key is valid and passed in `x-gladia-key` header.\n- [ ] Audio file is under 1000 MB and 135 minutes (pre-recorded) or 3 hours (live).\n- [ ] Audio format is supported (MP3, WAV, MP4, etc.).\n- [ ] Language is set explicitly in `language_config.languages` if known.\n- [ ] If using code switching, `languages` list is constrained to 3–5 expected languages.\n- [ ] Diarization is enabled if multiple speakers need attribution.\n- [ ] Custom vocabulary entries have realistic `intensity` (0.4–0.6) and `pronunciations`.\n- [ ] Webhooks or callbacks are configured if polling is not feasible.\n- [ ] Live sessions are closed before 3 hours; pre-recorded jobs are split if over 135 minutes.\n- [ ] Response includes expected fields: `transcription.utterances[]`, `metadata`, and any requested audio intelligence results.\n- [ ] Confidence scores and timing (`start`, `end`) are present for quality validation.\n- [ ] Multi-channel audio is correctly interleaved if merging multiple sources.\n\n## Resources\n\n- **Comprehensive page listing**: https://docs.gladia.io/llms.txt\n- **Getting started guide**: https://docs.gladia.io/chapters/introduction/getting-started\n- **Pre-recorded quickstart**: https://docs.gladia.io/chapters/pre-recorded-stt/quickstart\n- **Live transcription quickstart**: https://docs.gladia.io/chapters/live-stt/quickstart\n- **API reference**: https://docs.gladia.io/api-reference\n- **Recommended parameters by use case**: https://docs.gladia.io/chapters/pre-recorded-stt/recommended-parameters\n- **Audio intelligence features**: https://docs.gladia.io/chapters/audio-intelligence\n- **Supported formats and limits**: https://docs.gladia.io/chapters/limits-and-specifications/supported-formats\n\n---\n\n> For additional documentation and navigation, see: https://docs.gladia.io/llms.txt\n---\n\n> This file is auto-synced from https://docs.gladia.io/.well-known/agent-skills/gladia/skill.md\n> Do not edit manually — changes will be overwritten by CI.\n> For additional documentation and navigation, see: https://docs.gladia.io/llms.txt\n\nFile v1.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn7fs6cmj9hqg7232sbkacf31d882wp2\",\n  \"slug\": \"gladia-documentation-auto\",\n  \"version\": \"1.0.0\",\n  \"publishedAt\": 1780941683795\n}\n\nFile v1.0.0:skill-card.md\n\n## Description: <br>\nComprehensive Gladia speech-to-text reference auto-synced from docs.gladia.io for choosing workflows, SDK usage, API endpoints, limits, and transcription best practices. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[gladiaio](https://clawhub.ai/user/gladiaio) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers building with Gladia use this skill to understand speech-to-text workflows, select pre-recorded or live transcription paths, configure SDK/API requests, and check limits and integration gotchas. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill discusses workflows that use Gladia API keys and may process sensitive audio, video, transcripts, and callbacks. <br>\nMitigation: Confirm the user is authorized to send content to Gladia, protect API keys, and review Gladia retention, privacy, and compliance terms before processing private content. <br>\nRisk: Incorrect transcription configuration can produce misleading transcripts or expose more data than intended. <br>\nMitigation: Review generated configuration, set language and privacy options deliberately, and validate results before using transcripts for critical decisions. <br>\n\n\n## Reference(s): <br>\n- [ClawHub release page](https://clawhub.ai/gladiaio/gladia-documentation-auto) <br>\n- [Gladia documentation](https://docs.gladia.io) <br>\n- [Getting started guide](https://docs.gladia.io/chapters/introduction/getting-started) <br>\n- [API reference](https://docs.gladia.io/api-reference) <br>\n- [Supported formats and limits](https://docs.gladia.io/chapters/limits-and-specifications/supported-formats) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [guidance, code, shell commands, configuration] <br>\n**Output Format:** [Markdown with inline shell, JavaScript, and JSON examples] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [May include API-key handling guidance and API request examples; no files are generated by default.] <br>\n\n## Skill Version(s): <br>\n1.0.0 (source: server release metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>","readmeExcerpt":"Skill: gladia-documentation-auto Owner: gladiaio Summary: Comprehensive Gladia speech-to-text reference auto-synced from docs.gladia.io. Use as a general-purpose fallback when other specialized skills don't match, or when the user needs a broad overview of Gladia capabilities, endpoints, decision guidance, or workflows. Always prefer the official SDK; fall back to raw REST/WebSocket only when SDK cannot satisfy the r","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"curl -H \"x-gladia-key: your_key\" https://api.gladia.io/v2/pre-recorded"},{"language":"bash","snippet":"# Set API key in environment\nexport GLADIA_API_KEY=your_key\n\n# Or pass per request\ncurl -H \"x-gladia-key: your_key\" https://api.gladia.io/v2/pre-recorded"},{"language":"javascript","snippet":"const gladia = new GladiaClient({ apiKey: \"YOUR_KEY\" });\nconst result = await gladia.preRecorded().transcribe(\"audio.mp3\");"},{"language":"python","snippet":"gladia = GladiaClient(api_key=\"YOUR_KEY\").prerecorded()\nresult = gladia.transcribe(\"audio.mp3\")"},{"language":"javascript","snippet":"const session = gladia.liveV2().startSession({\n  encoding: \"wav/pcm\",\n  sample_rate: 16000,\n  bit_depth: 16,\n  channels: 1,\n});\nsession.on(\"message\", (msg) => {\n  if (msg.type === \"transcript\" && msg.data.is_final) {\n    console.log(msg.data.utterance.text);\n  }\n});\nsession.sendAudio(audioChunk);\nsession.stopRecording();"},{"language":"bash","snippet":"gladia auth set your_key\ngladia transcribe meeting.wav                    # text output\ngladia transcribe podcast.mp3 -o json            # JSON output\ngladia transcribe call.wav --diarize -o srt      # subtitles with speakers\ngladia transcribe mixed.mp3 --code-switching     # mixed languages\ngladia transcribe audio.mp3 --model solaria-3 --language en"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: gladia-documentation-auto\ndescription: Comprehensive Gladia speech-to-text reference auto-synced from docs.gladia.io. Use as a general-purpose fallback when other specialized skills don't match, or when the user needs a broad overview of Gladia capabilities, endpoints, decision guidance, or workflows. Always prefer the official SDK; fall back to raw REST/WebSocket only when SDK cannot satisfy the requirement.\nlicense: MIT\nmetadata:\n  source: https://docs.gladia.io/.well-known/agent-skills/gladia/skill.md\n  digest: sha256:fce0f1bdb678fca35d434a7f9f589187478d079847f0b9fcd6b846c032d1365c\n  synced: \"2026-10-02\"\n---\n\n> **SDK-first**: always use the official SDK — see [gladia-sdk-integration](../gladia-sdk-integration/SKILL.md) for policy, setup, and fallback criteria.\n\n## References\n\nConsult these sibling skills as needed:\n\n- ../gladia-sdk-integration/SKILL.md -- SDK setup, client initialization, error handling, and SDK vs raw API decision guide\n- ../gladia-sdk-integration/references/sdk-versions.md -- Current SDK versions (auto-synced by CI)\n- ../gladia-troubleshooting/SKILL.md -- Common errors, gotchas, and verification checklist\n- ../gladia-live-transcription/SKILL.md -- Live streaming transcription\n- ../gladia-pre-recorded-transcription/SKILL.md -- Pre-recorded file transcription\n\n---\nname: gladia\ndescription: Use when transcribing audio or video to text, building real-time voice applications, extracting insights from speech (diarization, translation, sentiment), or integrating speech-to-text into voice agents, meeting recorders, or multilingual applications. Supports both pre-recorded (async) and live (streaming) transcription with 100+ languages.\nmetadata:\n    mintlify-proj: gladia\n    version: \"1.0\"\n---\n\n# Gladia Skill\n\n## Product summary\n\nGladia is a speech-to-text (STT) API for transcribing audio and video to text. It supports two modes: **pre-recorded** (async file upload) and **live** (real-time WebSocket streaming). The API includes audio intelligence features (diarization, translation, sentiment analysis, PII redaction, summarization) and two models: **Solaria-3** (highest accuracy on European audio, pre-recorded only, 5 languages) and **Solaria-1** (default, 100+ languages, live + async, code switching).\n\n**Key files and endpoints:**\n- API key: Get from https://app.gladia.io/apikeys\n- Pre-recorded: `POST /v2/pre-recorded` (create job), `GET /v2/pre-recorded/:id` (poll result)\n- Live: `POST /v2/live` (init session), WebSocket connection for streaming\n- Authentication: Header `x-gladia-key: YOUR_API_KEY`\n- SDKs: `@gladiaio/sdk` (JavaScript/TypeScript), `gladiaio-sdk` (Python)\n- CLI: `gladia transcribe <file>` for terminal use\n\n**Primary docs:** https://docs.gladia.io\n\n## When to use\n\nReach for Gladia when:\n- Transcribing pre-recorded audio/video files (MP3, WAV, M4A, etc.) asynchronously\n- Building real-time voice applications (live captions, voice agents, meeting recorders)\n- Extracting structured data from speech (who spoke whe"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7fs6cmj9hqg7232sbkacf31d882wp2\",\n  \"slug\": \"gladia-documentation-auto\",\n  \"version\": \"1.0.6\",\n  \"publishedAt\": 1790953726054\n}"},{"path":"skill-card.md","content":"## Description:\n\nProvides Gladia speech-to-text integration guidance for pre-recorded and live transcription, including SDK examples, API workflows, and feature selection.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[gladiaio](https://clawhub.ai/user/gladiaio)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers use this reference to choose Gladia transcription modes and features, integrate the SDK or API, and handle audio transcription results.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Transcription can send sensitive audio or video to Gladia and retain audio and transcripts.\n\nMitigation: Obtain consent before uploading; configure zero data retention when privacy or compliance requires it.\n\nRisk: Exposing the Gladia API key in client code can allow unauthorized use.\n\nMitigation: Keep the API key on the backend or in a protected environment variable; never hardcode it in client code.\n\n## Reference(s):\n\n- [Gladia skill documentation](https://docs.gladia.io/.well-known/agent-skills/gladia/skill.md)\n- [Gladia documentation](https://docs.gladia.io/llms.txt)\n- [Pre-recorded transcription quickstart](https://docs.gladia.io/chapters/pre-recorded-stt/quickstart)\n- [Live transcription quickstart](https://docs.gladia.io/chapters/live-stt/quickstart)\n- [Gladia API reference](https://docs.gladia.io/api-reference/)\n\n## Skill Output:\n\n**Output Type(s):** [Guidance, Code, Shell commands, Configuration instructions]\n\n**Output Format:** [Markdown with code examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [SDK-first guidance for pre-recorded and live transcription.]\n\n## Skill Version(s):\n\n1.0.6 (source: ClawHub release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":1538,"uniquenessScore":40,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-10T15:44:38.092Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-10T15:44:38.092Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T21:43:13.555Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}