{"id":"2af5f375-476b-46a7-b806-d1e698855822","entityType":"agent","slug":"clawhub-psyb0t-aigate","name":"aigate","canonicalUrl":"https://www.xpersona.co/agent/clawhub-psyb0t-aigate","canonicalPath":"/agent/clawhub-psyb0t-aigate","generatedAt":"2026-10-10T06:43:08.340Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T15:17:18.986Z","emptyReason":null},"description":"Self-hosted AI platform — one `docker-compose up`, one OpenAI-compatible endpoint at http://localhost:4000. Bundles inference (Groq/Cerebras/OpenRouter/HuggingFace/Mistral/Cohere/Ollama/vLLM/llama.cpp/claudebox/pibox-zai/Anthropic/OpenAI), MCP tool use, a stealth browser cluster, image generation (FLUX/DALL-E/SD), speech synthesis (Kokoro/Qwen3-TTS/Chatterbox/OpenAI TTS), transcription (Whisper/Parakeet), S3-compatible object storage, agentic code execution (Claude Code + pi-coding-agent + sandboxed piston), web search (SearXNG), an email gateway (mailbox), a Telegram client (Telethon), time-series forecasting + tabular ML (predictalot), audio/video production (audiolla/flickies), an async job queue (proxq), and a web UI (LibreChat) — all reachable through one bearer token and automatic per-model fallback routing. Use when the user wants a one-command self-hosted OpenAI-compatible stack that aggregates many providers/tools behind a single endpoint instead of wiring each service up individually.","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 2.4K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s17fq93tmpky791n7516jcn08n83sfn2:aigate","sourceUrl":"https://clawhub.ai/psyb0t/aigate","homepage":"https://clawhub.ai/psyb0t/skills/aigate","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/psyb0t/aigate","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/psyb0t/skills/aigate","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":68,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"aigate technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T15:17:18.986Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T15:17:18.986Z","emptyReason":null},"stars":null,"forks":null,"downloads":2411,"packageName":null,"latestVersion":"10.0.0","tractionLabel":"2.4K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T15:17:18.985Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T15:17:18.986Z","lastCrawledAt":"2026-10-09T15:17:18.985Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T15:17:18.985Z","lastVerifiedAt":null,"highlights":[{"version":"10.0.0","createdAt":"2026-10-08T16:04:37.433Z","changelog":"- Removed the skill-card.md file. - No functional or user-facing changes; this update only deletes sample documentation.","fileCount":4,"zipByteSize":10978},{"version":"9.0.1","createdAt":"2026-10-07T22:58:44.654Z","changelog":"- Removed the skill-card.md file. - No other functional or code changes.","fileCount":4,"zipByteSize":10930},{"version":"9.0.0","createdAt":"2026-10-04T13:35:11.571Z","changelog":"aigate 9.0.0 - Updated documentation in `references/setup.md`. - Removed the `skill-card.md` file, streamlining documentation structure.","fileCount":4,"zipByteSize":11024},{"version":"8.0.0","createdAt":"2026-10-03T15:46:58.914Z","changelog":"aigate 8.0.0 - Removed the file: skill-card.md. - No feature, API, or functionality changes noted.","fileCount":4,"zipByteSize":10628},{"version":"7.0.0","createdAt":"2026-10-03T11:43:52.594Z","changelog":"aigate 7.0.0 - Updated documentation in SKILL.md and references/setup.md for clarity and completeness. - Removed legacy file skill-card.md. - No functional changes to the core platform; this release focuses on documentation and housekeeping.","fileCount":4,"zipByteSize":10655},{"version":"6.3.0","createdAt":"2026-10-02T20:21:21.959Z","changelog":"aigate 6.3.0 - Removed the skill-card.md file. - Updated SKILL.md and references/setup.md with content or documentation changes. - No user-facing features or behavioral changes; primarily documentation and housekeeping updates.","fileCount":4,"zipByteSize":10545},{"version":"6.2.0","createdAt":"2026-09-28T08:16:41.513Z","changelog":"aigate 6.2.0 - Documentation updated in SKILL.md and references/setup.md. - Removed obsolete skill-card.md file. - No functional or feature changes; this release focuses on code/documentation cleanup.","fileCount":4,"zipByteSize":10371},{"version":"6.1.0","createdAt":"2026-09-28T07:33:56.589Z","changelog":"## aigate 6.1.0 - Removed the file `skill-card.md` from the repository. - No other changes to the skill logic or metadata.","fileCount":4,"zipByteSize":10141}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17fq93tmpky791n7516jcn08n83sfn2:aigate","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s17fq93tmpky791n7516jcn08n83sfn2:aigate` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/psyb0t/aigate before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-psyb0t-aigate/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-psyb0t-aigate/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-psyb0t-aigate/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-psyb0t-aigate/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-psyb0t-aigate/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-psyb0t-aigate/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T06:43:08.337Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-psyb0t-aigate/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-psyb0t-aigate/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-psyb0t-aigate/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-psyb0t-aigate/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T15:17:18.986Z","emptyReason":null},"readme":"Skill: aigate\n\nOwner: psyb0t\n\nSummary: Self-hosted AI platform — one `docker-compose up`, one OpenAI-compatible endpoint at http://localhost:4000. Bundles inference (Groq/Cerebras/OpenRouter/HuggingFace/Mistral/Cohere/Ollama/vLLM/llama.cpp/claudebox/pibox-zai/Anthropic/OpenAI), MCP tool use, a stealth browser cluster, image generation (FLUX/DALL-E/SD), speech synthesis (Kokoro/Qwen3-TTS/Chatterbox/OpenAI TTS), transcription (Whisper/Parakeet), S3-compatible object storage, agentic code execution (Claude Code + pi-coding-agent + sandboxed piston), web search (SearXNG), an email gateway (mailbox), a Telegram client (Telethon), time-series forecasting + tabular ML (predictalot), audio/video production (audiolla/flickies), an async job queue (proxq), and a web UI (LibreChat) — all reachable through one bearer token and automatic per-model fallback routing. Use when the user wants a one-command self-hosted OpenAI-compatible stack that aggregates many providers/tools behind a single endpoint instead of wiring each service up individually.\n\nTags: latest:10.0.0\n\nVersion history:\n\nv10.0.0 | 2026-10-08T16:04:37.433Z | auto\n\n- Removed the skill-card.md file.\n- No functional or user-facing changes; this update only deletes sample documentation.\n\nv9.0.1 | 2026-10-07T22:58:44.654Z | auto\n\n- Removed the skill-card.md file.\n- No other functional or code changes.\n\nv9.0.0 | 2026-10-04T13:35:11.571Z | auto\n\naigate 9.0.0\n\n- Updated documentation in `references/setup.md`.\n- Removed the `skill-card.md` file, streamlining documentation structure.\n\nv8.0.0 | 2026-10-03T15:46:58.914Z | auto\n\naigate 8.0.0\n\n- Removed the file: skill-card.md.\n- No feature, API, or functionality changes noted.\n\nv7.0.0 | 2026-10-03T11:43:52.594Z | auto\n\naigate 7.0.0\n\n- Updated documentation in SKILL.md and references/setup.md for clarity and completeness.\n- Removed legacy file skill-card.md.\n- No functional changes to the core platform; this release focuses on documentation and housekeeping.\n\nv6.3.0 | 2026-10-02T20:21:21.959Z | auto\n\naigate 6.3.0\n\n- Removed the skill-card.md file.\n- Updated SKILL.md and references/setup.md with content or documentation changes.\n- No user-facing features or behavioral changes; primarily documentation and housekeeping updates.\n\nv6.2.0 | 2026-09-28T08:16:41.513Z | auto\n\naigate 6.2.0\n\n- Documentation updated in SKILL.md and references/setup.md.\n- Removed obsolete skill-card.md file.\n- No functional or feature changes; this release focuses on code/documentation cleanup.\n\nv6.1.0 | 2026-09-28T07:33:56.589Z | auto\n\n## aigate 6.1.0\n\n- Removed the file `skill-card.md` from the repository.\n- No other changes to the skill logic or metadata.\n\nv6.0.0 | 2026-09-26T16:24:38.973Z | auto\n\naigate 6.0.0\n\n- Updated setup instructions: `make limits` now checks enabled services and writes CPU caps to `.env.limits`.\n- Clarified instructions regarding `.env.limits` and machine-specific RAM/CPU sizing.\n- Removed `skill-card.md` file.\n- Improved documentation in SKILL.md for clarity and current usage.\n\nv5.7.0 | 2026-09-26T01:11:06.452Z | auto\n\naigate 5.7.0 changelog\n\n- Updated setup instructions in references/setup.md for improved clarity or accuracy.\n- Removed the skill-card.md file to streamline documentation.\n- No changes to the core functionality or permissions.\n\nv5.6.0 | 2026-09-24T13:39:39.588Z | auto\n\n- Removed the skill card documentation file (skill-card.md).\n- No changes to core functionality or user features.\n- Documentation now streamlined; all essential usage information remains in SKILL.md.\n\nv5.5.0 | 2026-09-24T12:49:03.234Z | auto\n\naigate 5.5.0\n\n- Added DECIDEALOT_AUTH_TOKEN (and related) to documented per-service token overrides in the SKILL.md security section\n- Updated references/setup.md and SKILL.md with clarifications and minor corrections\n- Removed deprecated skill-card.md file for streamlined documentation\n\nv5.4.1 | 2026-09-23T22:37:23.028Z | auto\n\naigate 5.4.1\n\n- Removed the skill-card.md file to streamline documentation and reduce redundancy.\n- No functional or behavioral changes; this is a documentation clean-up release.\n\nv5.4.0 | 2026-09-23T21:32:13.717Z | auto\n\n## aigate 5.4.0 changelog\n\n- Removed the sample file skill-card.md to streamline documentation.\n- No functional or user-facing changes; only the documentation file list was modified.\n\nv5.3.0 | 2026-09-22T14:51:15.103Z | auto\n\n- Removed the skill-card.md file.\n- No changes to core functionality or documentation content.\n- Internal documentation or packaging cleanup only.\n\nv5.2.0 | 2026-09-13T01:51:41.111Z | auto\n\n- Removed the file skill-card.md.\n- No functional changes to the skill itself; documentation cleanup only.\n\nv5.1.1 | 2026-09-13T00:53:43.769Z | auto\n\n- Removed the file skill-card.md.\n- No changes to feature set or core documentation; functionality is unchanged.\n- Minor housekeeping/removal of redundant or unused documentation files.\n\nv5.1.0 | 2026-09-09T15:21:55.385Z | auto\n\naigate 5.1.0\n\n- Added support for a third agentic coding agent (pibox) alongside claudebox and pibox-zai.\n- Updated documentation to reflect pibox's shell and file access capabilities.\n- Removed the skill-card.md file.\n- Minor clarifications in security and capability descriptions.\n\nv5.0.0 | 2026-09-09T09:46:05.031Z | auto\n\naigate 5.0.0\n\n- Updated setup flow: `.env` is now created from `.env.example` only; `docker-compose.yml` is tracked and overridden safely via user-defined `docker-compose.override.yml` (which is gitignored).\n- Improved documentation in SKILL.md to clarify configuration, file overrides, and best practices for customizing deployments.\n- References to quick start and configuration updated for clarity and maintainability.\n- No changes to core permissions or security model.\n\nv4.0.0 | 2026-09-09T09:31:34.937Z | auto\n\naigate 4.0.0\n\n- Updated documentation for improved clarity and quick start: added `make bootstrap` instructions to streamline environment and compose file creation.\n- Clarified service tiers: \"subscription\" now replaces \"flat-rate\" to describe agent backends more accurately.\n- Refined recommended usage for production setup, emphasizing using trusted hosts and secure exposure methods.\n- Updated references/setup.md and removed the deprecated skill-card.md documentation file to reduce redundancy.\n\nv3.24.0 | 2026-09-06T07:42:50.400Z | auto\n\n- Removed the sample file skill-card.md.\n- No user-facing or functional changes; this version only removes documentation.\n\nv3.23.0 | 2026-09-03T12:27:46.363Z | auto\n\naigate 3.23.0\n\n- Removed the skill-card.md file from the project.\n- No functional or behavior changes; internal documentation file cleanup only.\n\nv3.22.0 | 2026-08-27T23:58:32.645Z | auto\n\naigate 3.22.0\n\n- Documentation update: SKILL.md updated; no functional/service changes included.\n- Removed the file: skill-card.md.\n- No changes to permissions or endpoints.\n\nv3.21.0 | 2026-08-21T18:06:48.373Z | auto\n\n- Removed the file skill-card.md from the repository.\n- No changes to features or functionality; this update involves only documentation cleanup.\n\nv3.20.1 | 2026-08-14T01:45:21.133Z | auto\n\n## aigate 3.20.1\n\n- Removed the `skill-card.md` file from the repository.\n- No user-facing functionality changes; all core capabilities and documentation remain as before.\n\nv3.20.0 | 2026-08-13T17:20:31.291Z | auto\n\nVersion 3.20.0\n\n- Removed the skill-card.md file. \n- No user-facing changes to core features or functionality.\n- The overall documentation, capabilities, and security guidance remain unchanged.\n\nv3.19.3 | 2026-08-08T20:52:07.157Z | auto\n\naigate 3.19.3\n\n- Updated setup instructions and information in `references/setup.md`.\n- Removed the `skill-card.md` file for streamlined documentation.\n- No functional changes to the core skill or endpoint.\n\nv3.19.2 | 2026-08-08T15:02:03.146Z | auto\n\n- Removed the file: skill-card.md.\n- No other functional or user-facing changes in this version.\n\nv3.19.1 | 2026-08-08T10:02:32.099Z | auto\n\n- Removed the file skill-card.md.\n- No changes to SKILL.md content.\n- No new features or behavioral changes introduced in this release.\n\nv3.19.0 | 2026-08-07T22:37:52.830Z | auto\n\naigate 3.19.0\n\n- Updated documentation (SKILL.md) to reflect the addition of Chatterbox to supported speech synthesis providers.\n- Removed the deprecated skill-card.md file.\n- No functional or interface changes; maintenance and minor doc accuracy improvements only.\n\nv3.18.0 | 2026-08-06T02:57:24.532Z | auto\n\naigate 3.18.0\n\n- Cleaned up documentation by removing the redundant skill-card.md.\n- Updated SKILL.md for improved clarity, keeping all platform details and usage instructions current.\n- No functional or interface changes; this update is documentation/metadata only.\n\nv3.17.2 | 2026-08-02T19:03:10.121Z | auto\n\n- Removed the skill-card.md file.\n- No changes to core functionality or features.\n\nv3.17.1 | 2026-08-01T20:01:09.781Z | auto\n\n### aigate 3.17.1\n\n- Removed the file `skill-card.md` from the project.\n- No changes to primary functionality or user-facing features.\n\nv3.17.0 | 2026-07-31T19:50:27.792Z | auto\n\n## aigate 3.17.0 changelog\n\n- Removed the file: `skill-card.md`\n- No user-facing features or documentation changed in this version.\n- No functional or security-affecting changes detected.\n\nv3.16.2 | 2026-07-31T13:56:52.267Z | auto\n\nNo changes detected in this version; internal version bump only.\n\nv3.16.1 | 2026-07-31T13:30:42.676Z | auto\n\n- Removed the file: skill-card.md\n- No changes to functionality or features.\n- This update contains only documentation cleanup.\n\nv3.16.0 | 2026-07-30T12:06:30.670Z | auto\n\n- Removed the skill-card.md file.\n- No changes to core features or documentation, other than removing this file.\n\nv3.15.8 | 2026-07-27T23:23:21.955Z | auto\n\n- Removed the skill-card.md file.\n- No functional or user-facing changes; this update is limited to documentation cleanup.\n\nv3.15.7 | 2026-07-27T22:49:51.542Z | auto\n\n- Removed the file skill-card.md.\n- No functional or behavioral changes to the skill itself.\n- This update impacts only documentation (removal of an auxiliary file) and does not affect end users.\n\nv3.15.6 | 2026-07-27T15:05:43.477Z | auto\n\n- Removed the file skill-card.md.\n- No functional or feature changes to the skill; documentation file only was removed.\n\nv3.15.5 | 2026-07-27T13:09:47.841Z | auto\n\n## aigate 3.15.5 Changelog\n\n- Removed the file `skill-card.md`.\n- No user-facing feature or configuration changes in this release.\n\nv3.15.4 | 2026-07-27T11:58:26.376Z | auto\n\n## aigate 3.15.4 Changelog\n\n- Removed redundant `skill-card.md` documentation file for a leaner repo.\n- No changes to app functionality or features.\n\nv3.15.3 | 2026-07-26T03:42:08.719Z | auto\n\n- Removed the file: skill-card.md.\n- No feature or behavior changes; this is a documentation/file cleanup release only.\n\nv3.15.2 | 2026-07-26T01:23:49.027Z | auto\n\naigate 3.15.2\n\n- Clarified that AIGATE_TOKEN is an all-or-nothing capability grant with no per-tool scoping by default; every service token defaults to it unless specifically overridden.\n- Updated security section in documentation with strong new warnings about bearer token blast radius and agent security expectations.\n- Removed the skill-card.md file.\n\nv3.15.1 | 2026-07-26T00:09:15.119Z | auto\n\n- Core job queue (proxq, at /q/) is now always on; removed separate flag for enabling it.\n- SKILL.md updated to clarify always-on components and reflect the new default behavior.\n- skill-card.md file removed.\n\nv3.15.0 | 2026-07-25T23:41:28.586Z | auto\n\n- Major update: Expanded platform support and feature set, exposing multiple AI services via a single OpenAI-compatible endpoint.\n- Bundles inference, tool use, browser automation, image generation, speech synthesis, transcription, agentic code execution, S3-compatible storage, web search, email/Telegram clients, time-series forecasting, async job queue, and a web UI—all behind one bearer token.\n- Introduces automatic per-model fallback routing across providers (cloud, flat-rate, pay-per-token, local).\n- All services are opt-in by environment variables; secure-by-default with strict bearer-token access.\n- Updated setup instructions, security guidance, and comprehensive overview of bundled services and routing.\n\nArchive index:\n\nArchive v10.0.0: 4 files, 10978 bytes\n\nFiles: references/setup.md (8435b), skill-card.md (1985b), SKILL.md (11921b), _meta.json (126b)\n\nFile v10.0.0:SKILL.md\n\n---\nname: aigate\ndescription: Self-hosted AI platform — one `docker-compose up`, one OpenAI-compatible endpoint at http://localhost:4000. Bundles inference (Groq/Cerebras/OpenRouter/HuggingFace/Mistral/Cohere/Ollama/vLLM/llama.cpp/claudebox/pibox-zai/Anthropic/OpenAI), MCP tool use, a stealth browser cluster, image generation (FLUX/DALL-E/SD), speech synthesis (Kokoro/Qwen3-TTS/Chatterbox/OpenAI TTS), transcription (Whisper/Parakeet), S3-compatible object storage, agentic code execution (Claude Code + pi-coding-agent + sandboxed piston), web search (SearXNG), an email gateway (mailbox), a Telegram client (Telethon), time-series forecasting + tabular ML (predictalot), audio/video production (audiolla/flickies), an async job queue (proxq), and a web UI (LibreChat) — all reachable through one bearer token and automatic per-model fallback routing. Use when the user wants a one-command self-hosted OpenAI-compatible stack that aggregates many providers/tools behind a single endpoint instead of wiring each service up individually.\nhomepage: https://github.com/psyb0t/aigate\nuser-invocable: true\npermissions:\n  network:\n    - outbound HTTP to the aigate endpoint (default http://localhost:4000, or wherever it's deployed)\n    - aigate itself reaches out to model providers, MCP tool backends, and the open internet on the user's behalf (web search, browser automation, email, Telegram)\n  shell:\n    - docker / docker compose (bring the stack up/down, read logs)\n    - curl (call the OpenAI-compatible endpoint and direct service routes)\nmetadata:\n  openclaw:\n    emoji: \"🚪\"\n    primaryEnv: AIGATE_TOKEN\n    requires:\n      bins:\n        - docker\n        - curl\n---\n\n# aigate — self-hosted AI platform behind one endpoint\n\nA self-hosted AI platform. One `docker-compose up` stands up inference, tool use, browser automation, image generation, speech synthesis, transcription, object storage, agentic code execution, web search, an email gateway, a Telegram client, time-series forecasting, an async job queue, and a web UI — all behind a single OpenAI-compatible endpoint at `http://localhost:4000`. Point any existing OpenAI-client library or `curl` at it and it works. Everything else is opt-in via `.env` flags; the always-on core is nginx, LiteLLM, PostgreSQL, Redis, and the `proxq` async job queue (at `/q/`, no flag needed).\n\n## Security & safety\n\n**This is a very high-capability, very high-blast-radius stack. Treat the endpoint and its token like root on the host.** A single `AIGATE_TOKEN` bearer can, depending on what's enabled:\n\n- Hold API keys/credentials for many cloud model providers (Groq, Cerebras, OpenRouter, HuggingFace, Mistral, Cohere, Anthropic, OpenAI) plus subscription agent backends (Claude Code OAuth/API key, z.ai GLM Coding Plan).\n- Execute arbitrary code — three full agentic coding agents (claudebox, pibox-zai, pibox) with shell + file access, plus sandboxed multi-language execution (piston).\n- Drive a real browser (stealth Camoufox cluster) that can log into sites, fill forms, and act as the user across the open web.\n- Send email and Telegram messages on the user's behalf (mailbox, Telethon) — mailbox additionally holds plaintext IMAP/SMTP credentials in its YAML config.\n- Read/write S3-compatible object storage with a public-read bucket.\n\n**No per-tool scoping by default.** `AIGATE_TOKEN` is a single all-or-nothing capability grant — every per-service token (`CLAUDEBOX_API_TOKEN`, `PIBOX_ZAI_API_TOKEN`, `PREDICTALOT_AUTH_TOKEN`, `DECIDEALOT_AUTH_TOKEN`, `AUDIOLLA_AUTH_TOKEN`, `FLICKIES_AUTH_TOKEN`, `STEALTHY_AUTO_BROWSE_AUTH_TOKEN`, `HYBRIDS3_MASTER_KEY`, `MCP_TOOLS_AUTH_TOKEN`, `TELETHON_AUTH_KEY`, etc.) defaults to it unless the operator explicitly overrides each one separately. Handing an agent the token is not \"give it chat access\" — it's granting code execution, browser automation, and messaging in one shot, with no way to grant a narrower subset unless the operator has pre-split the per-service tokens. An agent must only be given `AIGATE_TOKEN` when it is fully trusted and only for the specific action the user explicitly requested — never pass it to an agent \"just in case it needs something.\"\n\nTreat aigate as a **trusted host only**. Concretely:\n- Never expose port `4000` directly to the public internet. Use Cloudflare Tunnel (`CLOUDFLARED=1`) or Tailscale (`TAILSCALE=1`) — both keep no ports open on the host — or put a real authenticating gateway/reverse-proxy in front of it.\n- Every request needs `Authorization: Bearer $AIGATE_TOKEN` (or a per-service override token) — there is no unauthenticated path once a service is enabled. Don't hardcode the token in scripts committed to a repo; source it from `.env`/environment.\n- Internal services (Postgres, Redis, LiteLLM, and most optional services) bind to no host ports at all — only nginx is exposed. Don't add host port mappings for internal services unless you specifically need direct access and understand you're widening the blast radius.\n- `piston` runs `privileged: true` (required for nsjail's own isolation, not a bypass of it) and lives on an internal-only network with no outbound internet — don't change that without understanding why.\n- Guard `.env` and any mailbox/Telethon config files — they hold plaintext secrets and are gitignored for a reason.\n\n## When to use\n\n- The user wants a single self-hosted endpoint that speaks the OpenAI API and routes across many providers with automatic fallback (free-tier cloud → subscription → pay-per-token → local).\n- The user wants bundled AI tooling (browser automation, image/speech/transcription, code execution, storage, search, email, Telegram, forecasting) reachable via MCP tools or REST without standing up each service by hand.\n- The user wants to run models fully locally (CPU or NVIDIA GPU) with no external calls, or mix local + cloud with automatic fallback between them.\n- The user needs a chat web UI (LibreChat) pre-wired to every enabled model and tool.\n\n## When NOT to use\n\n- The user only needs one specific provider's API directly — aigate is overhead if the goal is just \"call OpenAI\" with no routing/tooling/fallback need.\n- Untrusted/multi-tenant exposure without a real auth gateway in front — aigate's bearer-token model is not a substitute for per-user authz.\n- The user needs Windows-native or non-Docker deployment — this stack is Docker Compose only.\n\n## Quick start\n\n```bash\ngit clone https://github.com/psyb0t/aigate\ncd aigate\nmake bootstrap # creates .env from .env.example (any target does this)\n# edit .env: set AIGATE_TOKEN, flip the flags for the providers/services you want to 1\n# .env is gitignored. docker-compose.yml is tracked, so put compose changes in\n# docker-compose.override.yml, which is gitignored and merges last\nmake limits    # checks enabled services fit this machine, writes CPU caps to .env.limits\nmake run-bg    # start the stack in the background\n```\n\nGateway is at `http://localhost:4000`. Call it like any OpenAI-compatible endpoint:\n\n```bash\ncurl http://localhost:4000/chat/completions \\\n  -H \"Authorization: Bearer $AIGATE_TOKEN\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"model\": \"local-ollama-cpu-llama3.2-3b\", \"messages\": [{\"role\": \"user\", \"content\": \"hello\"}]}'\n```\n\n`model` selects the provider/route; LiteLLM handles fallback automatically if the requested one rate-limits or fails. See `references/setup.md` for the full env/routing story.\n\n## What's bundled and how to reach it\n\nEverything below sits behind the same `http://localhost:4000` endpoint and the same `AIGATE_TOKEN` bearer — aigate's job is exposing them, not reimplementing them. Enable each with its `.env` flag; disabled services are excluded from routing/fallback entirely.\n\n- **Inference + routing** — `/chat/completions`, `/embeddings`, `/images/generations`, `/audio/*` (OpenAI-compatible, via LiteLLM). Model name picks the provider: free-tier cloud (Groq, Cerebras, OpenRouter, HuggingFace, Mistral, Cohere), subscription agents (claudebox = Claude Code, pibox-zai = pi-coding-agent on a GLM Coding Plan), pay-per-token (Anthropic, OpenAI), or fully local CPU/CUDA (Ollama, vLLM, llama.cpp, talkies, sd.cpp). `pibox` runs the same pi agent on whichever of these models you point it at, so it costs whatever that model costs. Fallback chains retry the next provider automatically on 429/5xx.\n- **MCP tool use** — any function-calling model can autonomously invoke `generate_image`, `generate_tts`, `search_web`, `execute_code`, and per-service MCP tools (browser, storage, mailbox, Telethon, predictalot, decidealot, audiolla, flickies, claudebox/pibox-zai/pibox agent tools). Auto-enabled with the underlying service.\n- **Browser automation** — `stealthy-auto-browse`, 5-replica stealth Camoufox cluster behind HAProxy. REST + MCP (`BROWSER=1`).\n- **Agentic code execution** — claudebox (Claude Code) and pibox-zai (pi-coding-agent/z.ai) for full shell+file agentic tasks; piston at `/piston/` for sandboxed nsjail-isolated one-shot code execution (`CLAUDEBOX=1`, `PIBOX_ZAI=1`, `PISTON=1`).\n- **Object storage** — `hybrids3` at `/storage/`, S3-compatible, plain HTTP + boto3, public-read uploads, presigned URLs (`HYBRIDS3=1`).\n- **Image generation** — cloud (FLUX, DALL-E, SD) and local CPU/CUDA (`SDCPP=1` / `SDCPP_CUDA=1`) via `/images/generations` or MCP.\n- **Speech synthesis + transcription** — `talkies` unifies both under `/audio/speech` and `/audio/transcriptions` (Kokoro, Qwen3-TTS, Chatterbox Turbo, Whisper, Parakeet, Canary, Sherpa-ONNX, Vosk, wav2vec2/ZIPA phoneme ASR — `TALKIES=1` / `TALKIES_CUDA=1`); cloud TTS/ASR routes through the same endpoints.\n- **Web search** — SearXNG at `/searxng/`, plus MCP `search_web` (`SEARXNG=1`).\n- **Email gateway** — `mailbox` at `/mailbox/`, stateless IMAP+SMTP across N accounts from one YAML config, REST + MCP (`MAILBOX=1`, needs `MAILBOX_CONFIG` + `MAILBOX_AUTH_TOKEN`).\n- **Telegram client** — `telethon` at `/telethon/`, REST + MCP (`TELETHON=1`, needs API ID/hash + string session).\n- **Time-series forecasting + tabular ML** — `predictalot` at `/predictalot/` (CPU) and `/predictalot-cuda/` (GPU), REST + MCP (`PREDICTALOT=1` / `PREDICTALOT_CUDA=1`).\n- **Typed decisions**: `decidealot` at `/decidealot/` (CPU) and `/decidealot-cuda/` (GPU), REST + MCP (`DECIDEALOT=1` / `DECIDEALOT_CUDA=1`). Laya, Von, and CLM are enabled by default. `make run-bg` also starts CLM's CUDA Qwen3-8B encoder. CLM calls AIGate internally at `http://litellm:4000/v1/embeddings`, selecting `local-llamacpp-cuda-qwen3-8b`. Set `DECIDEALOT_CLM_ENABLED=false` to remove that GPU dependency. Set `DECIDEALOT_TYPESAFE_API_KEY` in the gitignored `.env` to enable hosted Jev, then select a name from `list_models`. Answers have `choice` / `score` / `noul` types and probabilities. `system_one_batch` accepts independent requests. The caller owns the threshold and the action that follows.\n- **Audio production** — `audiolla` at `/audiolla/` / `/audiolla-cuda/` — stem separation, mastering, MIDI, text-to-audio, REST + MCP (`AUDIOLLA=1` / `AUDIOLLA_CUDA=1`).\n- **Video toolkit** — `flickies` at `/flickies/` / `/flickies-cuda/` — lipsync, face restore, ffmpeg ops, REST + MCP (`FLICKIES=1` / `FLICKIES_CUDA=1`).\n- **Async job queue** — `proxq` at `/q/` — queue any OpenAI-path request, poll `/q/__jobs/{id}`, avoids client-side timeouts on long inference.\n\n## Web UI\n\nLibreChat at `/librechat/` (`LIBRECHAT=1`) — pre-configured with every enabled model and MCP tool, conversation history, file uploads, WebSocket streaming. Email/password auth; the first registered user becomes admin (then set `LIBRECHAT_ALLOW_REGISTRATION=false`). Admin UI for LiteLLM itself is at `/ui/`, optionally behind nginx basic auth (`LITELLM_UI_BASIC_AUTH`).\n\n## Setup details\n\nFor docker-compose bring-up, required env/keys, ports, and model/routing config, see `references/setup.md`.\n\nFile v10.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn79dhvmpjng4rp2jjk8k0v5xx80ccbk\",\n  \"slug\": \"aigate\",\n  \"version\": \"10.0.0\",\n  \"publishedAt\": 1791475477433\n}\n\nFile v10.0.0:references/setup.md\n\n# aigate setup\n\nAccurate to the aigate `README.md` / `docker-compose.yml` / `.env.example` at time of writing. Re-check those files if this drifts. `.env` and `docker-compose.override.yml` are the operator's local files and are gitignored.\n\n## Bring-up\n\n```bash\ngit clone https://github.com/psyb0t/aigate\ncd aigate\nmake bootstrap   # creates .env from .env.example (any target does this)\n# edit .env — see \"Required env\" below, then flip service flags to 1\n# .env is gitignored. docker-compose.yml is tracked, so put compose changes in\n# docker-compose.override.yml, which is gitignored and merges last\nmake limits      # checks enabled services fit this machine, writes CPU caps to .env.limits\nmake run-bg      # start detached\n# or: make run   # start in foreground with logs\n```\n\n`make restart-audiolla` recreates only enabled Audiolla CPU/CUDA variants from their locally available pinned images. It does not rebuild or pull images and leaves other services running.\n\n`make run` / `make run-bg` regenerate `litellm/config.yaml` from fragments (only enabled providers + filtered fallback chains) and pre-flight-validate any file-path env vars (e.g. `MAILBOX_CONFIG`, `CLOUDFLARED_CONFIG`) actually exist before starting containers.\n\nOther Makefile targets: `make down`, `make restart`, `make logs`, `make build-config` (regenerate litellm config only), `make test` (stack must already be running).\n\n## Ports\n\nSingle exposed port: **`4000`** (nginx), hardcoded — not env-configurable. Everything else (Postgres, Redis, LiteLLM, and most optional services) binds to no host port at all; they're reached only through nginx's path-based routing on `:4000`. Do not add host port mappings for internal services unless you specifically need direct access.\n\n- Gateway (OpenAI-compatible): `http://localhost:4000`\n- LiteLLM admin UI: `http://localhost:4000/ui/`\n- LibreChat (if `LIBRECHAT=1`): `http://localhost:4000/librechat/`\n- SearXNG (if `SEARXNG=1`): `http://localhost:4000/searxng/`\n- Async queue (proxq): `http://localhost:4000/q/`\n- Direct-routed services (not via LiteLLM): `/predictalot/`, `/predictalot-cuda/`, `/decidealot/`, `/decidealot-cuda/`, `/audiolla/`, `/audiolla-cuda/`, `/flickies/`, `/flickies-cuda/`, `/mailbox/`, `/telethon/`, `/piston/`, `/storage/` (hybrids3), `/claudebox/`, `/pibox-zai/`, `/pibox/`, `/stealthy-auto-browse/`\n\n## Required env / keys\n\nCore, always needed regardless of which optional services you enable:\n\n| Variable | Purpose |\n| --- | --- |\n| `AIGATE_TOKEN` | Master bearer token. Every per-service token below defaults to this value when left unset — one token authenticates against LiteLLM, claudebox, pibox-zai, predictalot, decidealot, mcp_tools, stealthy-auto-browse, hybrids3, telethon, audiolla, flickies, talkies, talkies-cuda. Override a specific `*_AUTH_TOKEN` / `*_API_TOKEN` var to scope that service separately. |\n| `LITELLM_MASTER_KEY` | Optional override; defaults to `AIGATE_TOKEN` when unset. |\n| `POSTGRES_DB` / `POSTGRES_USER` / `POSTGRES_PASSWORD` / `DATABASE_URL` | LiteLLM's key/usage/budget store. |\n| `REDIS_PASSWORD` | Password of the `proxq` Redis ACL user (job queue in DB 1). The `default` user is disabled. No whitespace. |\n| `LITELLM_REDIS_PASSWORD` | Password of the `litellm` Redis ACL user, which only holds the resource manager's hardware locks. Falls back to `REDIS_PASSWORD`. |\n| `LITELLM_UI_BASIC_AUTH` | `user:pass` for nginx basic auth in front of `/ui/`. Leave empty to disable (LiteLLM's own login still applies). |\n| `LITELLM_USERNAME` / `LITELLM_PASSWORD` | LiteLLM's own admin UI login. |\n\nAll requests carry `Authorization: Bearer $AIGATE_TOKEN` (or the relevant per-service override):\n\n```bash\ncurl http://localhost:4000/chat/completions \\\n  -H \"Authorization: Bearer $AIGATE_TOKEN\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"model\": \"local-ollama-cpu-llama3.2-3b\", \"messages\": [{\"role\": \"user\", \"content\": \"hello\"}]}'\n```\n\nPer-service keys/tokens only matter once you flip that service's flag to `1`. Every variable is documented inline in `.env.example` — read the comment above each block before enabling. Notable ones:\n\n- Cloud providers: `GROQ=1`, `CEREBRAS=1`, `OPENROUTER=1`, `HUGGINGFACE=1`, `MISTRAL=1`, `COHERE=1`, `ANTHROPIC=1`, `OPENAI=1` each need their own API key var alongside the flag (e.g. `OPENAI_API_KEY`).\n- `CLAUDEBOX=1` needs Claude OAuth token or Anthropic API key; token defaults to `AIGATE_TOKEN` via `CLAUDEBOX_API_TOKEN`.\n- `PIBOX_ZAI=1` needs a z.ai key; token defaults via `PIBOX_ZAI_API_TOKEN`.\n- `PIBOX=1` needs no outside account; it runs pi on this stack's own models. Set `PIBOX_MODELS` to enabled models that can call tools, and keep agent models (`claudebox-*`, `pibox-*`) out of that list to avoid recursion.\n- `DECIDEALOT=1` starts CPU typed decisions. `DECIDEALOT_CUDA=1` starts its GPU sibling. Laya, Von, and CLM are on by default. `make run-bg` also enables the CUDA Qwen encoder profile. CLM calls `http://litellm:4000/v1/embeddings` on the internal network with model `local-llamacpp-cuda-qwen3-8b`. Set `DECIDEALOT_CLM_ENABLED=false` to run without that GPU dependency. Set `DECIDEALOT_TYPESAFE_API_KEY` in the gitignored `.env` to enable hosted Jev. The `list_models` tool returns TypeSafe's current selectors, and `system_one_batch` accepts independent requests.\n- `MAILBOX=1` needs `MAILBOX_CONFIG` pointing at an existing host YAML file (copy `mailbox/config.example.yaml`, fill IMAP/SMTP creds, put a token under `auth.tokens:`) and `MAILBOX_AUTH_TOKEN` mirroring that token.\n- `TELETHON=1` needs `TELETHON_API_ID`, `TELETHON_API_HASH`, `TELETHON_SESSION` (generate the string session once via the telethon-plus `login` command — see `docs/services/telethon.md` upstream).\n- `CLOUDFLARED=1` / `TAILSCALE=1` are the two supported ways to expose the gateway beyond localhost without opening host ports — prefer these over publishing `4000` directly.\n- CUDA variants of any service (`*_CUDA=1`) require `nvidia-container-toolkit` on the host.\n\n## Model routing\n\nLiteLLM regenerates its config on every `make run`/`make run-bg`, including only enabled providers. Fallback chains (`litellm/config/fallbacks.json`) are priority-ordered and filtered to what's actually enabled:\n\n1. Free cloud (Groq, Cerebras, OpenRouter, HuggingFace, Mistral, Cohere) — rate-limited/capped, not unlimited.\n2. Subscription (claudebox, pibox-zai) — no per-token billing, but the allowance is metered.\n3. Pay-per-token (Anthropic, OpenAI) — real money per token.\n4. Local (Ollama, talkies, sd.cpp, vLLM, llama.cpp) — no external limits, bounded only by local hardware.\n\nModel names encode the route, e.g. `groq-gpt-oss-120b`, `local-ollama-cpu-llama3.2-3b`, `local-sdcpp-cuda-sd-turbo`. On a rate-limit or failure, LiteLLM automatically retries the next model in that model's fallback chain; the response's `model` field reports who actually served it. Async/long-running calls can go through `/q/` (proxq) instead of the sync path — submit, get a job ID back immediately, poll `/q/__jobs/{id}`.\n\nLiteLLM's resource manager serializes local jobs per hardware class and asks competing services to unload. Aigate's Decidealot launcher shares that Redis lock for Laya and Von, including direct REST, MCP, and batch items. Before CUDA inference it unloads Aigate's local llama.cpp encoder. CLM takes no outer lock because its nested LiteLLM embeddings call owns admission. Remote embeddings remain the remote service's responsibility. This launcher does not evict every other resident GPU service. Direct audiolla, flickies, and predictalot HTTP requests still bypass LiteLLM admission. `POST /v1/unload/{cuda,cpu}` requests manual cleanup across that hardware class. See [resource management](../../../../docs/resource-management.md).\n\n## Data / persistence\n\nAll persistent state lives under `.data/` (bind mounts), overridable via `DATA_DIR` or per-service `DATA_DIR_*`. Contents are gitignored; the directory tree itself is tracked via `.gitkeep`.\n\nAudiolla v2.0.0 deletes staged uploads and outputs in `${DATA_DIR_AUDIOLLA}/files` after 24 hours by modification time, including files retained before upgrading. Download results before expiry. `AUDIOLLA_FILES_TTL` changes this duration; `0` disables cleanup. Model caches are excluded. Active processing and downloads can postpone cleanup across CPU and CUDA. See [Audiolla storage and upgrade settings](../../../../docs/services/audiolla.md).\n\nFile v10.0.0:skill-card.md\n\n## Description:\n\nGuides developers in deploying and using a self-hosted, OpenAI-compatible gateway for model routing and optional AI tools.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[psyb0t](https://clawhub.ai/user/psyb0t)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and engineers use this skill to set up a Docker-based AI gateway and connect clients to model providers and optional tools through one endpoint.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: One bearer token may grant access to code execution, browser automation, messaging, and storage across enabled services.\n\nMitigation: Enable only necessary services, split per-service tokens, and grant agent access only for explicitly authorized tasks.\n\nRisk: Direct internet exposure of the gateway could expose high-privilege capabilities.\n\nMitigation: Install only on a trusted host and avoid exposing port 4000 directly to the internet; use a protected access path.\n\nRisk: Gateway, mailbox, and Telegram configuration may contain sensitive credentials.\n\nMitigation: Keep AIGATE_TOKEN and mailbox/Telethon configuration private and out of committed files.\n\n## Reference(s):\n\n- [aigate ClawHub release](https://clawhub.ai/psyb0t/skills/aigate)\n- [aigate setup guide](references/setup.md)\n\n## Skill Output:\n\n**Output Type(s):** [Guidance, Shell commands, Configuration instructions]\n\n**Output Format:** [Markdown with shell and configuration examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Deployment and API usage depend on enabled services and operator-provided credentials.]\n\n## Skill Version(s):\n\n10.0.0 (source: ClawHub release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v9.0.1: 4 files, 10930 bytes\n\nFiles: references/setup.md (8435b), skill-card.md (1911b), SKILL.md (11921b), _meta.json (125b)\n\nFile v9.0.1:SKILL.md\n\n---\nname: aigate\ndescription: Self-hosted AI platform — one `docker-compose up`, one OpenAI-compatible endpoint at http://localhost:4000. Bundles inference (Groq/Cerebras/OpenRouter/HuggingFace/Mistral/Cohere/Ollama/vLLM/llama.cpp/claudebox/pibox-zai/Anthropic/OpenAI), MCP tool use, a stealth browser cluster, image generation (FLUX/DALL-E/SD), speech synthesis (Kokoro/Qwen3-TTS/Chatterbox/OpenAI TTS), transcription (Whisper/Parakeet), S3-compatible object storage, agentic code execution (Claude Code + pi-coding-agent + sandboxed piston), web search (SearXNG), an email gateway (mailbox), a Telegram client (Telethon), time-series forecasting + tabular ML (predictalot), audio/video production (audiolla/flickies), an async job queue (proxq), and a web UI (LibreChat) — all reachable through one bearer token and automatic per-model fallback routing. Use when the user wants a one-command self-hosted OpenAI-compatible stack that aggregates many providers/tools behind a single endpoint instead of wiring each service up individually.\nhomepage: https://github.com/psyb0t/aigate\nuser-invocable: true\npermissions:\n  network:\n    - outbound HTTP to the aigate endpoint (default http://localhost:4000, or wherever it's deployed)\n    - aigate itself reaches out to model providers, MCP tool backends, and the open internet on the user's behalf (web search, browser automation, email, Telegram)\n  shell:\n    - docker / docker compose (bring the stack up/down, read logs)\n    - curl (call the OpenAI-compatible endpoint and direct service routes)\nmetadata:\n  openclaw:\n    emoji: \"🚪\"\n    primaryEnv: AIGATE_TOKEN\n    requires:\n      bins:\n        - docker\n        - curl\n---\n\n# aigate — self-hosted AI platform behind one endpoint\n\nA self-hosted AI platform. One `docker-compose up` stands up inference, tool use, browser automation, image generation, speech synthesis, transcription, object storage, agentic code execution, web search, an email gateway, a Telegram client, time-series forecasting, an async job queue, and a web UI — all behind a single OpenAI-compatible endpoint at `http://localhost:4000`. Point any existing OpenAI-client library or `curl` at it and it works. Everything else is opt-in via `.env` flags; the always-on core is nginx, LiteLLM, PostgreSQL, Redis, and the `proxq` async job queue (at `/q/`, no flag needed).\n\n## Security & safety\n\n**This is a very high-capability, very high-blast-radius stack. Treat the endpoint and its token like root on the host.** A single `AIGATE_TOKEN` bearer can, depending on what's enabled:\n\n- Hold API keys/credentials for many cloud model providers (Groq, Cerebras, OpenRouter, HuggingFace, Mistral, Cohere, Anthropic, OpenAI) plus subscription agent backends (Claude Code OAuth/API key, z.ai GLM Coding Plan).\n- Execute arbitrary code — three full agentic coding agents (claudebox, pibox-zai, pibox) with shell + file access, plus sandboxed multi-language execution (piston).\n- Drive a real browser (stealth Camoufox cluster) that can log into sites, fill forms, and act as the user across the open web.\n- Send email and Telegram messages on the user's behalf (mailbox, Telethon) — mailbox additionally holds plaintext IMAP/SMTP credentials in its YAML config.\n- Read/write S3-compatible object storage with a public-read bucket.\n\n**No per-tool scoping by default.** `AIGATE_TOKEN` is a single all-or-nothing capability grant — every per-service token (`CLAUDEBOX_API_TOKEN`, `PIBOX_ZAI_API_TOKEN`, `PREDICTALOT_AUTH_TOKEN`, `DECIDEALOT_AUTH_TOKEN`, `AUDIOLLA_AUTH_TOKEN`, `FLICKIES_AUTH_TOKEN`, `STEALTHY_AUTO_BROWSE_AUTH_TOKEN`, `HYBRIDS3_MASTER_KEY`, `MCP_TOOLS_AUTH_TOKEN`, `TELETHON_AUTH_KEY`, etc.) defaults to it unless the operator explicitly overrides each one separately. Handing an agent the token is not \"give it chat access\" — it's granting code execution, browser automation, and messaging in one shot, with no way to grant a narrower subset unless the operator has pre-split the per-service tokens. An agent must only be given `AIGATE_TOKEN` when it is fully trusted and only for the specific action the user explicitly requested — never pass it to an agent \"just in case it needs something.\"\n\nTreat aigate as a **trusted host only**. Concretely:\n- Never expose port `4000` directly to the public internet. Use Cloudflare Tunnel (`CLOUDFLARED=1`) or Tailscale (`TAILSCALE=1`) — both keep no ports open on the host — or put a real authenticating gateway/reverse-proxy in front of it.\n- Every request needs `Authorization: Bearer $AIGATE_TOKEN` (or a per-service override token) — there is no unauthenticated path once a service is enabled. Don't hardcode the token in scripts committed to a repo; source it from `.env`/environment.\n- Internal services (Postgres, Redis, LiteLLM, and most optional services) bind to no host ports at all — only nginx is exposed. Don't add host port mappings for internal services unless you specifically need direct access and understand you're widening the blast radius.\n- `piston` runs `privileged: true` (required for nsjail's own isolation, not a bypass of it) and lives on an internal-only network with no outbound internet — don't change that without understanding why.\n- Guard `.env` and any mailbox/Telethon config files — they hold plaintext secrets and are gitignored for a reason.\n\n## When to use\n\n- The user wants a single self-hosted endpoint that speaks the OpenAI API and routes across many providers with automatic fallback (free-tier cloud → subscription → pay-per-token → local).\n- The user wants bundled AI tooling (browser automation, image/speech/transcription, code execution, storage, search, email, Telegram, forecasting) reachable via MCP tools or REST without standing up each service by hand.\n- The user wants to run models fully locally (CPU or NVIDIA GPU) with no external calls, or mix local + cloud with automatic fallback between them.\n- The user needs a chat web UI (LibreChat) pre-wired to every enabled model and tool.\n\n## When NOT to use\n\n- The user only needs one specific provider's API directly — aigate is overhead if the goal is just \"call OpenAI\" with no routing/tooling/fallback need.\n- Untrusted/multi-tenant exposure without a real auth gateway in front — aigate's bearer-token model is not a substitute for per-user authz.\n- The user needs Windows-native or non-Docker deployment — this stack is Docker Compose only.\n\n## Quick start\n\n```bash\ngit clone https://github.com/psyb0t/aigate\ncd aigate\nmake bootstrap # creates .env from .env.example (any target does this)\n# edit .env: set AIGATE_TOKEN, flip the flags for the providers/services you want to 1\n# .env is gitignored. docker-compose.yml is tracked, so put compose changes in\n# docker-compose.override.yml, which is gitignored and merges last\nmake limits    # checks enabled services fit this machine, writes CPU caps to .env.limits\nmake run-bg    # start the stack in the background\n```\n\nGateway is at `http://localhost:4000`. Call it like any OpenAI-compatible endpoint:\n\n```bash\ncurl http://localhost:4000/chat/completions \\\n  -H \"Authorization: Bearer $AIGATE_TOKEN\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"model\": \"local-ollama-cpu-llama3.2-3b\", \"messages\": [{\"role\": \"user\", \"content\": \"hello\"}]}'\n```\n\n`model` selects the provider/route; LiteLLM handles fallback automatically if the requested one rate-limits or fails. See `references/setup.md` for the full env/routing story.\n\n## What's bundled and how to reach it\n\nEverything below sits behind the same `http://localhost:4000` endpoint and the same `AIGATE_TOKEN` bearer — aigate's job is exposing them, not reimplementing them. Enable each with its `.env` flag; disabled services are excluded from routing/fallback entirely.\n\n- **Inference + routing** — `/chat/completions`, `/embeddings`, `/images/generations`, `/audio/*` (OpenAI-compatible, via LiteLLM). Model name picks the provider: free-tier cloud (Groq, Cerebras, OpenRouter, HuggingFace, Mistral, Cohere), subscription agents (claudebox = Claude Code, pibox-zai = pi-coding-agent on a GLM Coding Plan), pay-per-token (Anthropic, OpenAI), or fully local CPU/CUDA (Ollama, vLLM, llama.cpp, talkies, sd.cpp). `pibox` runs the same pi agent on whichever of these models you point it at, so it costs whatever that model costs. Fallback chains retry the next provider automatically on 429/5xx.\n- **MCP tool use** — any function-calling model can autonomously invoke `generate_image`, `generate_tts`, `search_web`, `execute_code`, and per-service MCP tools (browser, storage, mailbox, Telethon, predictalot, decidealot, audiolla, flickies, claudebox/pibox-zai/pibox agent tools). Auto-enabled with the underlying service.\n- **Browser automation** — `stealthy-auto-browse`, 5-replica stealth Camoufox cluster behind HAProxy. REST + MCP (`BROWSER=1`).\n- **Agentic code execution** — claudebox (Claude Code) and pibox-zai (pi-coding-agent/z.ai) for full shell+file agentic tasks; piston at `/piston/` for sandboxed nsjail-isolated one-shot code execution (`CLAUDEBOX=1`, `PIBOX_ZAI=1`, `PISTON=1`).\n- **Object storage** — `hybrids3` at `/storage/`, S3-compatible, plain HTTP + boto3, public-read uploads, presigned URLs (`HYBRIDS3=1`).\n- **Image generation** — cloud (FLUX, DALL-E, SD) and local CPU/CUDA (`SDCPP=1` / `SDCPP_CUDA=1`) via `/images/generations` or MCP.\n- **Speech synthesis + transcription** — `talkies` unifies both under `/audio/speech` and `/audio/transcriptions` (Kokoro, Qwen3-TTS, Chatterbox Turbo, Whisper, Parakeet, Canary, Sherpa-ONNX, Vosk, wav2vec2/ZIPA phoneme ASR — `TALKIES=1` / `TALKIES_CUDA=1`); cloud TTS/ASR routes through the same endpoints.\n- **Web search** — SearXNG at `/searxng/`, plus MCP `search_web` (`SEARXNG=1`).\n- **Email gateway** — `mailbox` at `/mailbox/`, stateless IMAP+SMTP across N accounts from one YAML config, REST + MCP (`MAILBOX=1`, needs `MAILBOX_CONFIG` + `MAILBOX_AUTH_TOKEN`).\n- **Telegram client** — `telethon` at `/telethon/`, REST + MCP (`TELETHON=1`, needs API ID/hash + string session).\n- **Time-series forecasting + tabular ML** — `predictalot` at `/predictalot/` (CPU) and `/predictalot-cuda/` (GPU), REST + MCP (`PREDICTALOT=1` / `PREDICTALOT_CUDA=1`).\n- **Typed decisions**: `decidealot` at `/decidealot/` (CPU) and `/decidealot-cuda/` (GPU), REST + MCP (`DECIDEALOT=1` / `DECIDEALOT_CUDA=1`). Laya, Von, and CLM are enabled by default. `make run-bg` also starts CLM's CUDA Qwen3-8B encoder. CLM calls AIGate internally at `http://litellm:4000/v1/embeddings`, selecting `local-llamacpp-cuda-qwen3-8b`. Set `DECIDEALOT_CLM_ENABLED=false` to remove that GPU dependency. Set `DECIDEALOT_TYPESAFE_API_KEY` in the gitignored `.env` to enable hosted Jev, then select a name from `list_models`. Answers have `choice` / `score` / `noul` types and probabilities. `system_one_batch` accepts independent requests. The caller owns the threshold and the action that follows.\n- **Audio production** — `audiolla` at `/audiolla/` / `/audiolla-cuda/` — stem separation, mastering, MIDI, text-to-audio, REST + MCP (`AUDIOLLA=1` / `AUDIOLLA_CUDA=1`).\n- **Video toolkit** — `flickies` at `/flickies/` / `/flickies-cuda/` — lipsync, face restore, ffmpeg ops, REST + MCP (`FLICKIES=1` / `FLICKIES_CUDA=1`).\n- **Async job queue** — `proxq` at `/q/` — queue any OpenAI-path request, poll `/q/__jobs/{id}`, avoids client-side timeouts on long inference.\n\n## Web UI\n\nLibreChat at `/librechat/` (`LIBRECHAT=1`) — pre-configured with every enabled model and MCP tool, conversation history, file uploads, WebSocket streaming. Email/password auth; the first registered user becomes admin (then set `LIBRECHAT_ALLOW_REGISTRATION=false`). Admin UI for LiteLLM itself is at `/ui/`, optionally behind nginx basic auth (`LITELLM_UI_BASIC_AUTH`).\n\n## Setup details\n\nFor docker-compose bring-up, required env/keys, ports, and model/routing config, see `references/setup.md`.\n\nFile v9.0.1:_meta.json\n\n{\n  \"ownerId\": \"kn79dhvmpjng4rp2jjk8k0v5xx80ccbk\",\n  \"slug\": \"aigate\",\n  \"version\": \"9.0.1\",\n  \"publishedAt\": 1791413924654\n}\n\nFile v9.0.1:references/setup.md\n\n# aigate setup\n\nAccurate to the aigate `README.md` / `docker-compose.yml` / `.env.example` at time of writing. Re-check those files if this drifts. `.env` and `docker-compose.override.yml` are the operator's local files and are gitignored.\n\n## Bring-up\n\n```bash\ngit clone https://github.com/psyb0t/aigate\ncd aigate\nmake bootstrap   # creates .env from .env.example (any target does this)\n# edit .env — see \"Required env\" below, then flip service flags to 1\n# .env is gitignored. docker-compose.yml is tracked, so put compose changes in\n# docker-compose.override.yml, which is gitignored and merges last\nmake limits      # checks enabled services fit this machine, writes CPU caps to .env.limits\nmake run-bg      # start detached\n# or: make run   # start in foreground with logs\n```\n\n`make restart-audiolla` recreates only enabled Audiolla CPU/CUDA variants from their locally available pinned images. It does not rebuild or pull images and leaves other services running.\n\n`make run` / `make run-bg` regenerate `litellm/config.yaml` from fragments (only enabled providers + filtered fallback chains) and pre-flight-validate any file-path env vars (e.g. `MAILBOX_CONFIG`, `CLOUDFLARED_CONFIG`) actually exist before starting containers.\n\nOther Makefile targets: `make down`, `make restart`, `make logs`, `make build-config` (regenerate litellm config only), `make test` (stack must already be running).\n\n## Ports\n\nSingle exposed port: **`4000`** (nginx), hardcoded — not env-configurable. Everything else (Postgres, Redis, LiteLLM, and most optional services) binds to no host port at all; they're reached only through nginx's path-based routing on `:4000`. Do not add host port mappings for internal services unless you specifically need direct access.\n\n- Gateway (OpenAI-compatible): `http://localhost:4000`\n- LiteLLM admin UI: `http://localhost:4000/ui/`\n- LibreChat (if `LIBRECHAT=1`): `http://localhost:4000/librechat/`\n- SearXNG (if `SEARXNG=1`): `http://localhost:4000/searxng/`\n- Async queue (proxq): `http://localhost:4000/q/`\n- Direct-routed services (not via LiteLLM): `/predictalot/`, `/predictalot-cuda/`, `/decidealot/`, `/decidealot-cuda/`, `/audiolla/`, `/audiolla-cuda/`, `/flickies/`, `/flickies-cuda/`, `/mailbox/`, `/telethon/`, `/piston/`, `/storage/` (hybrids3), `/claudebox/`, `/pibox-zai/`, `/pibox/`, `/stealthy-auto-browse/`\n\n## Required env / keys\n\nCore, always needed regardless of which optional services you enable:\n\n| Variable | Purpose |\n| --- | --- |\n| `AIGATE_TOKEN` | Master bearer token. Every per-service token below defaults to this value when left unset — one token authenticates against LiteLLM, claudebox, pibox-zai, predictalot, decidealot, mcp_tools, stealthy-auto-browse, hybrids3, telethon, audiolla, flickies, talkies, talkies-cuda. Override a specific `*_AUTH_TOKEN` / `*_API_TOKEN` var to scope that service separately. |\n| `LITELLM_MASTER_KEY` | Optional override; defaults to `AIGATE_TOKEN` when unset. |\n| `POSTGRES_DB` / `POSTGRES_USER` / `POSTGRES_PASSWORD` / `DATABASE_URL` | LiteLLM's key/usage/budget store. |\n| `REDIS_PASSWORD` | Password of the `proxq` Redis ACL user (job queue in DB 1). The `default` user is disabled. No whitespace. |\n| `LITELLM_REDIS_PASSWORD` | Password of the `litellm` Redis ACL user, which only holds the resource manager's hardware locks. Falls back to `REDIS_PASSWORD`. |\n| `LITELLM_UI_BASIC_AUTH` | `user:pass` for nginx basic auth in front of `/ui/`. Leave empty to disable (LiteLLM's own login still applies). |\n| `LITELLM_USERNAME` / `LITELLM_PASSWORD` | LiteLLM's own admin UI login. |\n\nAll requests carry `Authorization: Bearer $AIGATE_TOKEN` (or the relevant per-service override):\n\n```bash\ncurl http://localhost:4000/chat/completions \\\n  -H \"Authorization: Bearer $AIGATE_TOKEN\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"model\": \"local-ollama-cpu-llama3.2-3b\", \"messages\": [{\"role\": \"user\", \"content\": \"hello\"}]}'\n```\n\nPer-service keys/tokens only matter once you flip that service's flag to `1`. Every variable is documented inline in `.env.example` — read the comment above each block before enabling. Notable ones:\n\n- Cloud providers: `GROQ=1`, `CEREBRAS=1`, `OPENROUTER=1`, `HUGGINGFACE=1`, `MISTRAL=1`, `COHERE=1`, `ANTHROPIC=1`, `OPENAI=1` each need their own API key var alongside the flag (e.g. `OPENAI_API_KEY`).\n- `CLAUDEBOX=1` needs Claude OAuth token or Anthropic API key; token defaults to `AIGATE_TOKEN` via `CLAUDEBOX_API_TOKEN`.\n- `PIBOX_ZAI=1` needs a z.ai key; token defaults via `PIBOX_ZAI_API_TOKEN`.\n- `PIBOX=1` needs no outside account; it runs pi on this stack's own models. Set `PIBOX_MODELS` to enabled models that can call tools, and keep agent models (`claudebox-*`, `pibox-*`) out of that list to avoid recursion.\n- `DECIDEALOT=1` starts CPU typed decisions. `DECIDEALOT_CUDA=1` starts its GPU sibling. Laya, Von, and CLM are on by default. `make run-bg` also enables the CUDA Qwen encoder profile. CLM calls `http://litellm:4000/v1/embeddings` on the internal network with model `local-llamacpp-cuda-qwen3-8b`. Set `DECIDEALOT_CLM_ENABLED=false` to run without that GPU dependency. Set `DECIDEALOT_TYPESAFE_API_KEY` in the gitignored `.env` to enable hosted Jev. The `list_models` tool returns TypeSafe's current selectors, and `system_one_batch` accepts independent requests.\n- `MAILBOX=1` needs `MAILBOX_CONFIG` pointing at an existing host YAML file (copy `mailbox/config.example.yaml`, fill IMAP/SMTP creds, put a token under `auth.tokens:`) and `MAILBOX_AUTH_TOKEN` mirroring that token.\n- `TELETHON=1` needs `TELETHON_API_ID`, `TELETHON_API_HASH`, `TELETHON_SESSION` (generate the string session once via the telethon-plus `login` command — see `docs/services/telethon.md` upstream).\n- `CLOUDFLARED=1` / `TAILSCALE=1` are the two supported ways to expose the gateway beyond localhost without opening host ports — prefer these over publishing `4000` directly.\n- CUDA variants of any service (`*_CUDA=1`) require `nvidia-container-toolkit` on the host.\n\n## Model routing\n\nLiteLLM regenerates its config on every `make run`/`make run-bg`, including only enabled providers. Fallback chains (`litellm/config/fallbacks.json`) are priority-ordered and filtered to what's actually enabled:\n\n1. Free cloud (Groq, Cerebras, OpenRouter, HuggingFace, Mistral, Cohere) — rate-limited/capped, not unlimited.\n2. Subscription (claudebox, pibox-zai) — no per-token billing, but the allowance is metered.\n3. Pay-per-token (Anthropic, OpenAI) — real money per token.\n4. Local (Ollama, talkies, sd.cpp, vLLM, llama.cpp) — no external limits, bounded only by local hardware.\n\nModel names encode the route, e.g. `groq-gpt-oss-120b`, `local-ollama-cpu-llama3.2-3b`, `local-sdcpp-cuda-sd-turbo`. On a rate-limit or failure, LiteLLM automatically retries the next model in that model's fallback chain; the response's `model` field reports who actually served it. Async/long-running calls can go through `/q/` (proxq) instead of the sync path — submit, get a job ID back immediately, poll `/q/__jobs/{id}`.\n\nLiteLLM's resource manager serializes local jobs per hardware class and asks competing services to unload. Aigate's Decidealot launcher shares that Redis lock for Laya and Von, including direct REST, MCP, and batch items. Before CUDA inference it unloads Aigate's local llama.cpp encoder. CLM takes no outer lock because its nested LiteLLM embeddings call owns admission. Remote embeddings remain the remote service's responsibility. This launcher does not evict every other resident GPU service. Direct audiolla, flickies, and predictalot HTTP requests still bypass LiteLLM admission. `POST /v1/unload/{cuda,cpu}` requests manual cleanup across that hardware class. See [resource management](../../../../docs/resource-management.md).\n\n## Data / persistence\n\nAll persistent state lives under `.data/` (bind mounts), overridable via `DATA_DIR` or per-service `DATA_DIR_*`. Contents are gitignored; the directory tree itself is tracked via `.gitkeep`.\n\nAudiolla v2.0.0 deletes staged uploads and outputs in `${DATA_DIR_AUDIOLLA}/files` after 24 hours by modification time, including files retained before upgrading. Download results before expiry. `AUDIOLLA_FILES_TTL` changes this duration; `0` disables cleanup. Model caches are excluded. Active processing and downloads can postpone cleanup across CPU and CUDA. See [Audiolla storage and upgrade settings](../../../../docs/services/audiolla.md).\n\nFile v9.0.1:skill-card.md\n\n## Description:\n\nGuides developers in setting up and using a self-hosted, OpenAI-compatible gateway for model routing and optional AI tools.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[psyb0t](https://clawhub.ai/user/psyb0t)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and engineers use this skill to deploy and operate a self-hosted AI gateway, route requests across local and cloud models, and selectively enable tools such as search, browser automation, and code execution.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: A single gateway token can grant broad access to enabled host, browser, messaging, and storage services.\n\nMitigation: Set strong, separate per-service tokens before granting agent access; give the master token only to fully trusted agents for explicitly requested actions.\n\nRisk: Exposing the gateway publicly can make its powerful services accessible to unauthorized users.\n\nMitigation: Keep the gateway local or behind an authenticated tunnel, and enable only the services needed.\n\n## Reference(s):\n\n- [aigate on ClawHub](https://clawhub.ai/psyb0t/skills/aigate)\n- [aigate setup guide](references/setup.md)\n- [Project homepage listed by the skill](https://github.com/psyb0t/aigate)\n\n## Skill Output:\n\n**Output Type(s):** [Guidance, Shell commands, Configuration instructions]\n\n**Output Format:** [Markdown with shell examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Deployment and API guidance depends on the services the operator enables.]\n\n## Skill Version(s):\n\n9.0.1 (source: ClawHub release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v9.0.0: 4 files, 11024 bytes\n\nFiles: references/setup.md (8435b), skill-card.md (2150b), SKILL.md (11921b), _meta.json (125b)\n\nFile v9.0.0:SKILL.md\n\n---\nname: aigate\ndescription: Self-hosted AI platform — one `docker-compose up`, one OpenAI-compatible endpoint at http://localhost:4000. Bundles inference (Groq/Cerebras/OpenRouter/HuggingFace/Mistral/Cohere/Ollama/vLLM/llama.cpp/claudebox/pibox-zai/Anthropic/OpenAI), MCP tool use, a stealth browser cluster, image generation (FLUX/DALL-E/SD), speech synthesis (Kokoro/Qwen3-TTS/Chatterbox/OpenAI TTS), transcription (Whisper/Parakeet), S3-compatible object storage, agentic code execution (Claude Code + pi-coding-agent + sandboxed piston), web search (SearXNG), an email gateway (mailbox), a Telegram client (Telethon), time-series forecasting + tabular ML (predictalot), audio/video production (audiolla/flickies), an async job queue (proxq), and a web UI (LibreChat) — all reachable through one bearer token and automatic per-model fallback routing. Use when the user wants a one-command self-hosted OpenAI-compatible stack that aggregates many providers/tools behind a single endpoint instead of wiring each service up individually.\nhomepage: https://github.com/psyb0t/aigate\nuser-invocable: true\npermissions:\n  network:\n    - outbound HTTP to the aigate endpoint (default http://localhost:4000, or wherever it's deployed)\n    - aigate itself reaches out to model providers, MCP tool backends, and the open internet on the user's behalf (web search, browser automation, email, Telegram)\n  shell:\n    - docker / docker compose (bring the stack up/down, read logs)\n    - curl (call the OpenAI-compatible endpoint and direct service routes)\nmetadata:\n  openclaw:\n    emoji: \"🚪\"\n    primaryEnv: AIGATE_TOKEN\n    requires:\n      bins:\n        - docker\n        - curl\n---\n\n# aigate — self-hosted AI platform behind one endpoint\n\nA self-hosted AI platform. One `docker-compose up` stands up inference, tool use, browser automation, image generation, speech synthesis, transcription, object storage, agentic code execution, web search, an email gateway, a Telegram client, time-series forecasting, an async job queue, and a web UI — all behind a single OpenAI-compatible endpoint at `http://localhost:4000`. Point any existing OpenAI-client library or `curl` at it and it works. Everything else is opt-in via `.env` flags; the always-on core is nginx, LiteLLM, PostgreSQL, Redis, and the `proxq` async job queue (at `/q/`, no flag needed).\n\n## Security & safety\n\n**This is a very high-capability, very high-blast-radius stack. Treat the endpoint and its token like root on the host.** A single `AIGATE_TOKEN` bearer can, depending on what's enabled:\n\n- Hold API keys/credentials for many cloud model providers (Groq, Cerebras, OpenRouter, HuggingFace, Mistral, Cohere, Anthropic, OpenAI) plus subscription agent backends (Claude Code OAuth/API key, z.ai GLM Coding Plan).\n- Execute arbitrary code — three full agentic coding agents (claudebox, pibox-zai, pibox) with shell + file access, plus sandboxed multi-language execution (piston).\n- Drive a real browser (stealth Camoufox cluster) that can log into sites, fill forms, and act as the user across the open web.\n- Send email and Telegram messages on the user's behalf (mailbox, Telethon) — mailbox additionally holds plaintext IMAP/SMTP credentials in its YAML config.\n- Read/write S3-compatible object storage with a public-read bucket.\n\n**No per-tool scoping by default.** `AIGATE_TOKEN` is a single all-or-nothing capability grant — every per-service token (`CLAUDEBOX_API_TOKEN`, `PIBOX_ZAI_API_TOKEN`, `PREDICTALOT_AUTH_TOKEN`, `DECIDEALOT_AUTH_TOKEN`, `AUDIOLLA_AUTH_TOKEN`, `FLICKIES_AUTH_TOKEN`, `STEALTHY_AUTO_BROWSE_AUTH_TOKEN`, `HYBRIDS3_MASTER_KEY`, `MCP_TOOLS_AUTH_TOKEN`, `TELETHON_AUTH_KEY`, etc.) defaults to it unless the operator explicitly overrides each one separately. Handing an agent the token is not \"give it chat access\" — it's granting code execution, browser automation, and messaging in one shot, with no way to grant a narrower subset unless the operator has pre-split the per-service tokens. An agent must only be given `AIGATE_TOKEN` when it is fully trusted and only for the specific action the user explicitly requested — never pass it to an agent \"just in case it needs something.\"\n\nTreat aigate as a **trusted host only**. Concretely:\n- Never expose port `4000` directly to the public internet. Use Cloudflare Tunnel (`CLOUDFLARED=1`) or Tailscale (`TAILSCALE=1`) — both keep no ports open on the host — or put a real authenticating gateway/reverse-proxy in front of it.\n- Every request needs `Authorization: Bearer $AIGATE_TOKEN` (or a per-service override token) — there is no unauthenticated path once a service is enabled. Don't hardcode the token in scripts committed to a repo; source it from `.env`/environment.\n- Internal services (Postgres, Redis, LiteLLM, and most optional services) bind to no host ports at all — only nginx is exposed. Don't add host port mappings for internal services unless you specifically need direct access and understand you're widening the blast radius.\n- `piston` runs `privileged: true` (required for nsjail's own isolation, not a bypass of it) and lives on an internal-only network with no outbound internet — don't change that without understanding why.\n- Guard `.env` and any mailbox/Telethon config files — they hold plaintext secrets and are gitignored for a reason.\n\n## When to use\n\n- The user wants a single self-hosted endpoint that speaks the OpenAI API and routes across many providers with automatic fallback (free-tier cloud → subscription → pay-per-token → local).\n- The user wants bundled AI tooling (browser automation, image/speech/transcription, code execution, storage, search, email, Telegram, forecasting) reachable via MCP tools or REST without standing up each service by hand.\n- The user wants to run models fully locally (CPU or NVIDIA GPU) with no external calls, or mix local + cloud with automatic fallback between them.\n- The user needs a chat web UI (LibreChat) pre-wired to every enabled model and tool.\n\n## When NOT to use\n\n- The user only needs one specific provider's API directly — aigate is overhead if the goal is just \"call OpenAI\" with no routing/tooling/fallback need.\n- Untrusted/multi-tenant exposure without a real auth gateway in front — aigate's bearer-token model is not a substitute for per-user authz.\n- The user needs Windows-native or non-Docker deployment — this stack is Docker Compose only.\n\n## Quick start\n\n```bash\ngit clone https://github.com/psyb0t/aigate\ncd aigate\nmake bootstrap # creates .env from .env.example (any target does this)\n# edit .env: set AIGATE_TOKEN, flip the flags for the providers/services you want to 1\n# .env is gitignored. docker-compose.yml is tracked, so put compose changes in\n# docker-compose.override.yml, which is gitignored and merges last\nmake limits    # checks enabled services fit this machine, writes CPU caps to .env.limits\nmake run-bg    # start the stack in the background\n```\n\nGateway is at `http://localhost:4000`. Call it like any OpenAI-compatible endpoint:\n\n```bash\ncurl http://localhost:4000/chat/completions \\\n  -H \"Authorization: Bearer $AIGATE_TOKEN\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"model\": \"local-ollama-cpu-llama3.2-3b\", \"messages\": [{\"role\": \"user\", \"content\": \"hello\"}]}'\n```\n\n`model` selects the provider/route; LiteLLM handles fallback automatically if the requested one rate-limits or fails. See `references/setup.md` for the full env/routing story.\n\n## What's bundled and how to reach it\n\nEverything below sits behind the same `http://localhost:4000` endpoint and the same `AIGATE_TOKEN` bearer — aigate's job is exposing them, not reimplementing them. Enable each with its `.env` flag; disabled services are excluded from routing/fallback entirely.\n\n- **Inference + routing** — `/chat/completions`, `/embeddings`, `/images/generations`, `/audio/*` (OpenAI-compatible, via LiteLLM). Model name picks the provider: free-tier cloud (Groq, Cerebras, OpenRouter, HuggingFace, Mistral, Cohere), subscription agents (claudebox = Claude Code, pibox-zai = pi-coding-agent on a GLM Coding Plan), pay-per-token (Anthropic, OpenAI), or fully local CPU/CUDA (Ollama, vLLM, llama.cpp, talkies, sd.cpp). `pibox` runs the same pi agent on whichever of these models you point it at, so it costs whatever that model costs. Fallback chains retry the next provider automatically on 429/5xx.\n- **MCP tool use** — any function-calling model can autonomously invoke `generate_image`, `generate_tts`, `search_web`, `execute_code`, and per-service MCP tools (browser, storage, mailbox, Telethon, predictalot, decidealot, audiolla, flickies, claudebox/pibox-zai/pibox agent tools). Auto-enabled with the underlying service.\n- **Browser automation** — `stealthy-auto-browse`, 5-replica stealth Camoufox cluster behind HAProxy. REST + MCP (`BROWSER=1`).\n- **Agentic code execution** — claudebox (Claude Code) and pibox-zai (pi-coding-agent/z.ai) for full shell+file agentic tasks; piston at `/piston/` for sandboxed nsjail-isolated one-shot code execution (`CLAUDEBOX=1`, `PIBOX_ZAI=1`, `PISTON=1`).\n- **Object storage** — `hybrids3` at `/storage/`, S3-compatible, plain HTTP + boto3, public-read uploads, presigned URLs (`HYBRIDS3=1`).\n- **Image generation** — cloud (FLUX, DALL-E, SD) and local CPU/CUDA (`SDCPP=1` / `SDCPP_CUDA=1`) via `/images/generations` or MCP.\n- **Speech synthesis + transcription** — `talkies` unifies both under `/audio/speech` and `/audio/transcriptions` (Kokoro, Qwen3-TTS, Chatterbox Turbo, Whisper, Parakeet, Canary, Sherpa-ONNX, Vosk, wav2vec2/ZIPA phoneme ASR — `TALKIES=1` / `TALKIES_CUDA=1`); cloud TTS/ASR routes through the same endpoints.\n- **Web search** — SearXNG at `/searxng/`, plus MCP `search_web` (`SEARXNG=1`).\n- **Email gateway** — `mailbox` at `/mailbox/`, stateless IMAP+SMTP across N accounts from one YAML config, REST + MCP (`MAILBOX=1`, needs `MAILBOX_CONFIG` + `MAILBOX_AUTH_TOKEN`).\n- **Telegram client** — `telethon` at `/telethon/`, REST + MCP (`TELETHON=1`, needs API ID/hash + string session).\n- **Time-series forecasting + tabular ML** — `predictalot` at `/predictalot/` (CPU) and `/predictalot-cuda/` (GPU), REST + MCP (`PREDICTALOT=1` / `PREDICTALOT_CUDA=1`).\n- **Typed decisions**: `decidealot` at `/decidealot/` (CPU) and `/decidealot-cuda/` (GPU), REST + MCP (`DECIDEALOT=1` / `DECIDEALOT_CUDA=1`). Laya, Von, and CLM are enabled by default. `make run-bg` also starts CLM's CUDA Qwen3-8B encoder. CLM calls AIGate internally at `http://litellm:4000/v1/embeddings`, selecting `local-llamacpp-cuda-qwen3-8b`. Set `DECIDEALOT_CLM_ENABLED=false` to remove that GPU dependency. Set `DECIDEALOT_TYPESAFE_API_KEY` in the gitignored `.env` to enable hosted Jev, then select a name from `list_models`. Answers have `choice` / `score` / `noul` types and probabilities. `system_one_batch` accepts independent requests. The caller owns the threshold and the action that follows.\n- **Audio production** — `audiolla` at `/audiolla/` / `/audiolla-cuda/` — stem separation, mastering, MIDI, text-to-audio, REST + MCP (`AUDIOLLA=1` / `AUDIOLLA_CUDA=1`).\n- **Video toolkit** — `flickies` at `/flickies/` / `/flickies-cuda/` — lipsync, face restore, ffmpeg ops, REST + MCP (`FLICKIES=1` / `FLICKIES_CUDA=1`).\n- **Async job queue** — `proxq` at `/q/` — queue any OpenAI-path request, poll `/q/__jobs/{id}`, avoids client-side timeouts on long inference.\n\n## Web UI\n\nLibreChat at `/librechat/` (`LIBRECHAT=1`) — pre-configured with every enabled model and MCP tool, conversation history, file uploads, WebSocket streaming. Email/password auth; the first registered user becomes admin (then set `LIBRECHAT_ALLOW_REGISTRATION=false`). Admin UI for LiteLLM itself is at `/ui/`, optionally behind nginx basic auth (`LITELLM_UI_BASIC_AUTH`).\n\n## Setup details\n\nFor docker-compose bring-up, required env/keys, ports, and model/routing config, see `references/setup.md`.\n\nFile v9.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn79dhvmpjng4rp2jjk8k0v5xx80ccbk\",\n  \"slug\": \"aigate\",\n  \"version\": \"9.0.0\",\n  \"publishedAt\": 1791120911571\n}\n\nFile v9.0.0:references/setup.md\n\n# aigate setup\n\nAccurate to the aigate `README.md` / `docker-compose.yml` / `.env.example` at time of writing. Re-check those files if this drifts. `.env` and `docker-compose.override.yml` are the operator's local files and are gitignored.\n\n## Bring-up\n\n```bash\ngit clone https://github.com/psyb0t/aigate\ncd aigate\nmake bootstrap   # creates .env from .env.example (any target does this)\n# edit .env — see \"Required env\" below, then flip service flags to 1\n# .env is gitignored. docker-compose.yml is tracked, so put compose changes in\n# docker-compose.override.yml, which is gitignored and merges last\nmake limits      # checks enabled services fit this machine, writes CPU caps to .env.limits\nmake run-bg      # start detached\n# or: make run   # start in foreground with logs\n```\n\n`make restart-audiolla` recreates only enabled Audiolla CPU/CUDA variants from their locally available pinned images. It does not rebuild or pull images and leaves other services running.\n\n`make run` / `make run-bg` regenerate `litellm/config.yaml` from fragments (only enabled providers + filtered fallback chains) and pre-flight-validate any file-path env vars (e.g. `MAILBOX_CONFIG`, `CLOUDFLARED_CONFIG`) actually exist before starting containers.\n\nOther Makefile targets: `make down`, `make restart`, `make logs`, `make build-config` (regenerate litellm config only), `make test` (stack must already be running).\n\n## Ports\n\nSingle exposed port: **`4000`** (nginx), hardcoded — not env-configurable. Everything else (Postgres, Redis, LiteLLM, and most optional services) binds to no host port at all; they're reached only through nginx's path-based routing on `:4000`. Do not add host port mappings for internal services unless you specifically need direct access.\n\n- Gateway (OpenAI-compatible): `http://localhost:4000`\n- LiteLLM admin UI: `http://localhost:4000/ui/`\n- LibreChat (if `LIBRECHAT=1`): `http://localhost:4000/librechat/`\n- SearXNG (if `SEARXNG=1`): `http://localhost:4000/searxng/`\n- Async queue (proxq): `http://localhost:4000/q/`\n- Direct-routed services (not via LiteLLM): `/predictalot/`, `/predictalot-cuda/`, `/decidealot/`, `/decidealot-cuda/`, `/audiolla/`, `/audiolla-cuda/`, `/flickies/`, `/flickies-cuda/`, `/mailbox/`, `/telethon/`, `/piston/`, `/storage/` (hybrids3), `/claudebox/`, `/pibox-zai/`, `/pibox/`, `/stealthy-auto-browse/`\n\n## Required env / keys\n\nCore, always needed regardless of which optional services you enable:\n\n| Variable | Purpose |\n| --- | --- |\n| `AIGATE_TOKEN` | Master bearer token. Every per-service token below defaults to this value when left unset — one token authenticates against LiteLLM, claudebox, pibox-zai, predictalot, decidealot, mcp_tools, stealthy-auto-browse, hybrids3, telethon, audiolla, flickies, talkies, talkies-cuda. Override a specific `*_AUTH_TOKEN` / `*_API_TOKEN` var to scope that service separately. |\n| `LITELLM_MASTER_KEY` | Optional override; defaults to `AIGATE_TOKEN` when unset. |\n| `POSTGRES_DB` / `POSTGRES_USER` / `POSTGRES_PASSWORD` / `DATABASE_URL` | LiteLLM's key/usage/budget store. |\n| `REDIS_PASSWORD` | Password of the `proxq` Redis ACL user (job queue in DB 1). The `default` user is disabled. No whitespace. |\n| `LITELLM_REDIS_PASSWORD` | Password of the `litellm` Redis ACL user, which only holds the resource manager's hardware locks. Falls back to `REDIS_PASSWORD`. |\n| `LITELLM_UI_BASIC_AUTH` | `user:pass` for nginx basic auth in front of `/ui/`. Leave empty to disable (LiteLLM's own login still applies). |\n| `LITELLM_USERNAME` / `LITELLM_PASSWORD` | LiteLLM's own admin UI login. |\n\nAll requests carry `Authorization: Bearer $AIGATE_TOKEN` (or the relevant per-service override):\n\n```bash\ncurl http://localhost:4000/chat/completions \\\n  -H \"Authorization: Bearer $AIGATE_TOKEN\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"model\": \"local-ollama-cpu-llama3.2-3b\", \"messages\": [{\"role\": \"user\", \"content\": \"hello\"}]}'\n```\n\nPer-service keys/tokens only matter once you flip that service's flag to `1`. Every variable is documented inline in `.env.example` — read the comment above each block before enabling. Notable ones:\n\n- Cloud providers: `GROQ=1`, `CEREBRAS=1`, `OPENROUTER=1`, `HUGGINGFACE=1`, `MISTRAL=1`, `COHERE=1`, `ANTHROPIC=1`, `OPENAI=1` each need their own API key var alongside the flag (e.g. `OPENAI_API_KEY`).\n- `CLAUDEBOX=1` needs Claude OAuth token or Anthropic API key; token defaults to `AIGATE_TOKEN` via `CLAUDEBOX_API_TOKEN`.\n- `PIBOX_ZAI=1` needs a z.ai key; token defaults via `PIBOX_ZAI_API_TOKEN`.\n- `PIBOX=1` needs no outside account; it runs pi on this stack's own models. Set `PIBOX_MODELS` to enabled models that can call tools, and keep agent models (`claudebox-*`, `pibox-*`) out of that list to avoid recursion.\n- `DECIDEALOT=1` starts CPU typed decisions. `DECIDEALOT_CUDA=1` starts its GPU sibling. Laya, Von, and CLM are on by default. `make run-bg` also enables the CUDA Qwen encoder profile. CLM calls `http://litellm:4000/v1/embeddings` on the internal network with model `local-llamacpp-cuda-qwen3-8b`. Set `DECIDEALOT_CLM_ENABLED=false` to run without that GPU dependency. Set `DECIDEALOT_TYPESAFE_API_KEY` in the gitignored `.env` to enable hosted Jev. The `list_models` tool returns TypeSafe's current selectors, and `system_one_batch` accepts independent requests.\n- `MAILBOX=1` needs `MAILBOX_CONFIG` pointing at an existing host YAML file (copy `mailbox/config.example.yaml`, fill IMAP/SMTP creds, put a token under `auth.tokens:`) and `MAILBOX_AUTH_TOKEN` mirroring that token.\n- `TELETHON=1` needs `TELETHON_API_ID`, `TELETHON_API_HASH`, `TELETHON_SESSION` (generate the string session once via the telethon-plus `login` command — see `docs/services/telethon.md` upstream).\n- `CLOUDFLARED=1` / `TAILSCALE=1` are the two supported ways to expose the gateway beyond localhost without opening host ports — prefer these over publishing `4000` directly.\n- CUDA variants of any service (`*_CUDA=1`) require `nvidia-container-toolkit` on the host.\n\n## Model routing\n\nLiteLLM regenerates its config on every `make run`/`make run-bg`, including only enabled providers. Fallback chains (`litellm/config/fallbacks.json`) are priority-ordered and filtered to what's actually enabled:\n\n1. Free cloud (Groq, Cerebras, OpenRouter, HuggingFace, Mistral, Cohere) — rate-limited/capped, not unlimited.\n2. Subscription (claudebox, pibox-zai) — no per-token billing, but the allowance is metered.\n3. Pay-per-token (Anthropic, OpenAI) — real money per token.\n4. Local (Ollama, talkies, sd.cpp, vLLM, llama.cpp) — no external limits, bounded only by local hardware.\n\nModel names encode the route, e.g. `groq-gpt-oss-120b`, `local-ollama-cpu-llama3.2-3b`, `local-sdcpp-cuda-sd-turbo`. On a rate-limit or failure, LiteLLM automatically retries the next model in that model's fallback chain; the response's `model` field reports who actually served it. Async/long-running calls can go through `/q/` (proxq) instead of the sync path — submit, get a job ID back immediately, poll `/q/__jobs/{id}`.\n\nLiteLLM's resource manager serializes local jobs per hardware class and asks competing services to unload. Aigate's Decidealot launcher shares that Redis lock for Laya and Von, including direct REST, MCP, and batch items. Before CUDA inference it unloads Aigate's local llama.cpp encoder. CLM takes no outer lock because its nested LiteLLM embeddings call owns admission. Remote embeddings remain the remote service's responsibility. This launcher does not evict every other resident GPU service. Direct audiolla, flickies, and predictalot HTTP requests still bypass LiteLLM admission. `POST /v1/unload/{cuda,cpu}` requests manual cleanup across that hardware class. See [resource management](../../../../docs/resource-management.md).\n\n## Data / persistence\n\nAll persistent state lives under `.data/` (bind mounts), overridable via `DATA_DIR` or per-service `DATA_DIR_*`. Contents are gitignored; the directory tree itself is tracked via `.gitkeep`.\n\nAudiolla v2.0.0 deletes staged uploads and outputs in `${DATA_DIR_AUDIOLLA}/files` after 24 hours by modification time, including files retained before upgrading. Download results before expiry. `AUDIOLLA_FILES_TTL` changes this duration; `0` disables cleanup. Model caches are excluded. Active processing and downloads can postpone cleanup across CPU and CUDA. See [Audiolla storage and upgrade settings](../../../../docs/services/audiolla.md).\n\nFile v9.0.0:skill-card.md\n\n## Description:\n\nHelps developers deploy and use a self-hosted, OpenAI-compatible gateway that combines model routing with optional AI tools and services.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[psyb0t](https://clawhub.ai/user/psyb0t)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and operators use this skill to set up a self-hosted AI gateway and access model inference, optional tools, and related services through a single endpoint.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: A single gateway token can grant enabled tools access to code execution, browsing, messaging, storage, and provider credentials.\n\nMitigation: Treat AIGATE_TOKEN as a root-level secret, use separate per-service tokens, and grant agents access only to the services and actions they need.\n\nRisk: Public exposure of the gateway could make privileged tools accessible remotely.\n\nMitigation: Install on a trusted machine and do not expose port 4000 directly to the public internet; use an authenticated access layer.\n\nRisk: Optional browser, email, Telegram, storage, and coding services can act using the operator's credentials.\n\nMitigation: Enable only necessary services and review their permissions and credentials before authorizing agent use.\n\n## Reference(s):\n\n- [aigate on ClawHub](https://clawhub.ai/psyb0t/skills/aigate)\n- [aigate setup guide](references/setup.md)\n- [Project homepage (listed in skill metadata)](https://github.com/psyb0t/aigate)\n\n## Skill Output:\n\n**Output Type(s):** [Guidance, Shell commands, Configuration instructions]\n\n**Output Format:** [Markdown with inline shell examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Guidance for a Docker Compose deployment and authenticated gateway requests.]\n\n## Skill Version(s):\n\n9.0.0 (source: ClawHub release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v8.0.0: 4 files, 10628 bytes\n\nFiles: references/setup.md (7799b), skill-card.md (1839b), SKILL.md (11921b), _meta.json (125b)\n\nFile v8.0.0:SKILL.md\n\n---\nname: aigate\ndescription: Self-hosted AI platform — one `docker-compose up`, one OpenAI-compatible endpoint at http://localhost:4000. Bundles inference (Groq/Cerebras/OpenRouter/HuggingFace/Mistral/Cohere/Ollama/vLLM/llama.cpp/claudebox/pibox-zai/Anthropic/OpenAI), MCP tool use, a stealth browser cluster, image generation (FLUX/DALL-E/SD), speech synthesis (Kokoro/Qwen3-TTS/Chatterbox/OpenAI TTS), transcription (Whisper/Parakeet), S3-compatible object storage, agentic code execution (Claude Code + pi-coding-agent + sandboxed piston), web search (SearXNG), an email gateway (mailbox), a Telegram client (Telethon), time-series forecasting + tabular ML (predictalot), audio/video production (audiolla/flickies), an async job queue (proxq), and a web UI (LibreChat) — all reachable through one bearer token and automatic per-model fallback routing. Use when the user wants a one-command self-hosted OpenAI-compatible stack that aggregates many providers/tools behind a single endpoint instead of wiring each service up individually.\nhomepage: https://github.com/psyb0t/aigate\nuser-invocable: true\npermissions:\n  network:\n    - outbound HTTP to the aigate endpoint (default http://localhost:4000, or wherever it's deployed)\n    - aigate itself reaches out to model providers, MCP tool backends, and the open internet on the user's behalf (web search, browser automation, email, Telegram)\n  shell:\n    - docker / docker compose (bring the stack up/down, read logs)\n    - curl (call the OpenAI-compatible endpoint and direct service routes)\nmetadata:\n  openclaw:\n    emoji: \"🚪\"\n    primaryEnv: AIGATE_TOKEN\n    requires:\n      bins:\n        - docker\n        - curl\n---\n\n# aigate — self-hosted AI platform behind one endpoint\n\nA self-hosted AI platform. One `docker-compose up` stands up inference, tool use, browser automation, image generation, speech synthesis, transcription, object storage, agentic code execution, web search, an email gateway, a Telegram client, time-series forecasting, an async job queue, and a web UI — all behind a single OpenAI-compatible endpoint at `http://localhost:4000`. Point any existing OpenAI-client library or `curl` at it and it works. Everything else is opt-in via `.env` flags; the always-on core is nginx, LiteLLM, PostgreSQL, Redis, and the `proxq` async job queue (at `/q/`, no flag needed).\n\n## Security & safety\n\n**This is a very high-capability, very high-blast-radius stack. Treat the endpoint and its token like root on the host.** A single `AIGATE_TOKEN` bearer can, depending on what's enabled:\n\n- Hold API keys/credentials for many cloud model providers (Groq, Cerebras, OpenRouter, HuggingFace, Mistral, Cohere, Anthropic, OpenAI) plus subscription agent backends (Claude Code OAuth/API key, z.ai GLM Coding Plan).\n- Execute arbitrary code — three full agentic coding agents (claudebox, pibox-zai, pibox) with shell + file access, plus sandboxed multi-language execution (piston).\n- Drive a real browser (stealth Camoufox cluster) that can log into sites, fill forms, and act as the user across the open web.\n- Send email and Telegram messages on the user's behalf (mailbox, Telethon) — mailbox additionally holds plaintext IMAP/SMTP credentials in its YAML config.\n- Read/write S3-compatible object storage with a public-read bucket.\n\n**No per-tool scoping by default.** `AIGATE_TOKEN` is a single all-or-nothing capability grant — every per-service token (`CLAUDEBOX_API_TOKEN`, `PIBOX_ZAI_API_TOKEN`, `PREDICTALOT_AUTH_TOKEN`, `DECIDEALOT_AUTH_TOKEN`, `AUDIOLLA_AUTH_TOKEN`, `FLICKIES_AUTH_TOKEN`, `STEALTHY_AUTO_BROWSE_AUTH_TOKEN`, `HYBRIDS3_MASTER_KEY`, `MCP_TOOLS_AUTH_TOKEN`, `TELETHON_AUTH_KEY`, etc.) defaults to it unless the operator explicitly overrides each one separately. Handing an agent the token is not \"give it chat access\" — it's granting code execution, browser automation, and messaging in one shot, with no way to grant a narrower subset unless the operator has pre-split the per-service tokens. An agent must only be given `AIGATE_TOKEN` when it is fully trusted and only for the specific action the user explicitly requested — never pass it to an agent \"just in case it needs something.\"\n\nTreat aigate as a **trusted host only**. Concretely:\n- Never expose port `4000` directly to the public internet. Use Cloudflare Tunnel (`CLOUDFLARED=1`) or Tailscale (`TAILSCALE=1`) — both keep no ports open on the host — or put a real authenticating gateway/reverse-proxy in front of it.\n- Every request needs `Authorization: Bearer $AIGATE_TOKEN` (or a per-service override token) — there is no unauthenticated path once a service is enabled. Don't hardcode the token in scripts committed to a repo; source it from `.env`/environment.\n- Internal services (Postgres, Redis, LiteLLM, and most optional services) bind to no host ports at all — only nginx is exposed. Don't add host port mappings for internal services unless you specifically need direct access and understand you're widening the blast radius.\n- `piston` runs `privileged: true` (required for nsjail's own isolation, not a bypass of it) and lives on an internal-only network with no outbound internet — don't change that without understanding why.\n- Guard `.env` and any mailbox/Telethon config files — they hold plaintext secrets and are gitignored for a reason.\n\n## When to use\n\n- The user wants a single self-hosted endpoint that speaks the OpenAI API and routes across many providers with automatic fallback (free-tier cloud → subscription → pay-per-token → local).\n- The user wants bundled AI tooling (browser automation, image/speech/transcription, code execution, storage, search, email, Telegram, forecasting) reachable via MCP tools or REST without standing up each service by hand.\n- The user wants to run models fully locally (CPU or NVIDIA GPU) with no external calls, or mix local + cloud with automatic fallback between them.\n- The user needs a chat web UI (LibreChat) pre-wired to every enabled model and tool.\n\n## When NOT to use\n\n- The user only needs one specific provider's API directly — aigate is overhead if the goal is just \"call OpenAI\" with no routing/tooling/fallback need.\n- Untrusted/multi-tenant exposure without a real auth gateway in front — aigate's bearer-token model is not a substitute for per-user authz.\n- The user needs Windows-native or non-Docker deployment — this stack is Docker Compose only.\n\n## Quick start\n\n```bash\ngit clone https://github.com/psyb0t/aigate\ncd aigate\nmake bootstrap # creates .env from .env.example (any target does this)\n# edit .env: set AIGATE_TOKEN, flip the flags for the providers/services you want to 1\n# .env is gitignored. docker-compose.yml is tracked, so put compose changes in\n# docker-compose.override.yml, which is gitignored and merges last\nmake limits    # checks enabled services fit this machine, writes CPU caps to .env.limits\nmake run-bg    # start the stack in the background\n```\n\nGateway is at `http://localhost:4000`. Call it like any OpenAI-compatible endpoint:\n\n```bash\ncurl http://localhost:4000/chat/completions \\\n  -H \"Authorization: Bearer $AIGATE_TOKEN\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"model\": \"local-ollama-cpu-llama3.2-3b\", \"messages\": [{\"role\": \"user\", \"content\": \"hello\"}]}'\n```\n\n`model` selects the provider/route; LiteLLM handles fallback automatically if the requested one rate-limits or fails. See `references/setup.md` for the full env/routing story.\n\n## What's bundled and how to reach it\n\nEverything below sits behind the same `http://localhost:4000` endpoint and the same `AIGATE_TOKEN` bearer — aigate's job is exposing them, not reimplementing them. Enable each with its `.env` flag; disabled services are excluded from routing/fallback entirely.\n\n- **Inference + routing** — `/chat/completions`, `/embeddings`, `/images/generations`, `/audio/*` (OpenAI-compatible, via LiteLLM). Model name picks the provider: free-tier cloud (Groq, Cerebras, OpenRouter, HuggingFace, Mistral, Cohere), subscription agents (claudebox = Claude Code, pibox-zai = pi-coding-agent on a GLM Coding Plan), pay-per-token (Anthropic, OpenAI), or fully local CPU/CUDA (Ollama, vLLM, llama.cpp, talkies, sd.cpp). `pibox` runs the same pi agent on whichever of these models you point it at, so it costs whatever that model costs. Fallback chains retry the next provider automatically on 429/5xx.\n- **MCP tool use** — any function-calling model can autonomously invoke `generate_image`, `generate_tts`, `search_web`, `execute_code`, and per-service MCP tools (browser, storage, mailbox, Telethon, predictalot, decidealot, audiolla, flickies, claudebox/pibox-zai/pibox agent tools). Auto-enabled with the underlying service.\n- **Browser automation** — `stealthy-auto-browse`, 5-replica stealth Camoufox cluster behind HAProxy. REST + MCP (`BROWSER=1`).\n- **Agentic code execution** — claudebox (Claude Code) and pibox-zai (pi-coding-agent/z.ai) for full shell+file agentic tasks; piston at `/piston/` for sandboxed nsjail-isolated one-shot code execution (`CLAUDEBOX=1`, `PIBOX_ZAI=1`, `PISTON=1`).\n- **Object storage** — `hybrids3` at `/storage/`, S3-compatible, plain HTTP + boto3, public-read uploads, presigned URLs (`HYBRIDS3=1`).\n- **Image generation** — cloud (FLUX, DALL-E, SD) and local CPU/CUDA (`SDCPP=1` / `SDCPP_CUDA=1`) via `/images/generations` or MCP.\n- **Speech synthesis + transcription** — `talkies` unifies both under `/audio/speech` and `/audio/transcriptions` (Kokoro, Qwen3-TTS, Chatterbox Turbo, Whisper, Parakeet, Canary, Sherpa-ONNX, Vosk, wav2vec2/ZIPA phoneme ASR — `TALKIES=1` / `TALKIES_CUDA=1`); cloud TTS/ASR routes through the same endpoints.\n- **Web search** — SearXNG at `/searxng/`, plus MCP `search_web` (`SEARXNG=1`).\n- **Email gateway** — `mailbox` at `/mailbox/`, stateless IMAP+SMTP across N accounts from one YAML config, REST + MCP (`MAILBOX=1`, needs `MAILBOX_CONFIG` + `MAILBOX_AUTH_TOKEN`).\n- **Telegram client** — `telethon` at `/telethon/`, REST + MCP (`TELETHON=1`, needs API ID/hash + string session).\n- **Time-series forecasting + tabular ML** — `predictalot` at `/predictalot/` (CPU) and `/predictalot-cuda/` (GPU), REST + MCP (`PREDICTALOT=1` / `PREDICTALOT_CUDA=1`).\n- **Typed decisions**: `decidealot` at `/decidealot/` (CPU) and `/decidealot-cuda/` (GPU), REST + MCP (`DECIDEALOT=1` / `DECIDEALOT_CUDA=1`). Laya, Von, and CLM are enabled by default. `make run-bg` also starts CLM's CUDA Qwen3-8B encoder. CLM calls AIGate internally at `http://litellm:4000/v1/embeddings`, selecting `local-llamacpp-cuda-qwen3-8b`. Set `DECIDEALOT_CLM_ENABLED=false` to remove that GPU dependency. Set `DECIDEALOT_TYPESAFE_API_KEY` in the gitignored `.env` to enable hosted Jev, then select a name from `list_models`. Answers have `choice` / `score` / `noul` types and probabilities. `system_one_batch` accepts independent requests. The caller owns the threshold and the action that follows.\n- **Audio production** — `audiolla` at `/audiolla/` / `/audiolla-cuda/` — stem separation, mastering, MIDI, text-to-audio, REST + MCP (`AUDIOLLA=1` / `AUDIOLLA_CUDA=1`).\n- **Video toolkit** — `flickies` at `/flickies/` / `/flickies-cuda/` — lipsync, face restore, ffmpeg ops, REST + MCP (`FLICKIES=1` / `FLICKIES_CUDA=1`).\n- **Async job queue** — `proxq` at `/q/` — queue any OpenAI-path request, poll `/q/__jobs/{id}`, avoids client-side timeouts on long inference.\n\n## Web UI\n\nLibreChat at `/librechat/` (`LIBRECHAT=1`) — pre-configured with every enabled model and MCP tool, conversation history, file uploads, WebSocket streaming. Email/password auth; the first registered user becomes admin (then set `LIBRECHAT_ALLOW_REGISTRATION=false`). Admin UI for LiteLLM itself is at `/ui/`, optionally behind nginx basic auth (`LITELLM_UI_BASIC_AUTH`).\n\n## Setup details\n\nFor docker-compose bring-up, required env/keys, ports, and model/routing config, see `references/setup.md`.\n\nFile v8.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn79dhvmpjng4rp2jjk8k0v5xx80ccbk\",\n  \"slug\": \"aigate\",\n  \"version\": \"8.0.0\",\n  \"publishedAt\": 1791042418914\n}\n\nFile v8.0.0:references/setup.md\n\n# aigate setup\n\nAccurate to the aigate `README.md` / `docker-compose.yml` / `.env.example` at time of writing. Re-check those files if this drifts. `.env` and `docker-compose.override.yml` are the operator's local files and are gitignored.\n\n## Bring-up\n\n```bash\ngit clone https://github.com/psyb0t/aigate\ncd aigate\nmake bootstrap   # creates .env from .env.example (any target does this)\n# edit .env — see \"Required env\" below, then flip service flags to 1\n# .env is gitignored. docker-compose.yml is tracked, so put compose changes in\n# docker-compose.override.yml, which is gitignored and merges last\nmake limits      # checks enabled services fit this machine, writes CPU caps to .env.limits\nmake run-bg      # start detached\n# or: make run   # start in foreground with logs\n```\n\n`make run` / `make run-bg` regenerate `litellm/config.yaml` from fragments (only enabled providers + filtered fallback chains) and pre-flight-validate any file-path env vars (e.g. `MAILBOX_CONFIG`, `CLOUDFLARED_CONFIG`) actually exist before starting containers.\n\nOther Makefile targets: `make down`, `make restart`, `make logs`, `make build-config` (regenerate litellm config only), `make test` (stack must already be running).\n\n## Ports\n\nSingle exposed port: **`4000`** (nginx), hardcoded — not env-configurable. Everything else (Postgres, Redis, LiteLLM, and most optional services) binds to no host port at all; they're reached only through nginx's path-based routing on `:4000`. Do not add host port mappings for internal services unless you specifically need direct access.\n\n- Gateway (OpenAI-compatible): `http://localhost:4000`\n- LiteLLM admin UI: `http://localhost:4000/ui/`\n- LibreChat (if `LIBRECHAT=1`): `http://localhost:4000/librechat/`\n- SearXNG (if `SEARXNG=1`): `http://localhost:4000/searxng/`\n- Async queue (proxq): `http://localhost:4000/q/`\n- Direct-routed services (not via LiteLLM): `/predictalot/`, `/predictalot-cuda/`, `/decidealot/`, `/decidealot-cuda/`, `/audiolla/`, `/audiolla-cuda/`, `/flickies/`, `/flickies-cuda/`, `/mailbox/`, `/telethon/`, `/piston/`, `/storage/` (hybrids3), `/claudebox/`, `/pibox-zai/`, `/pibox/`, `/stealthy-auto-browse/`\n\n## Required env / keys\n\nCore, always needed regardless of which optional services you enable:\n\n| Variable | Purpose |\n| --- | --- |\n| `AIGATE_TOKEN` | Master bearer token. Every per-service token below defaults to this value when left unset — one token authenticates against LiteLLM, claudebox, pibox-zai, predictalot, decidealot, mcp_tools, stealthy-auto-browse, hybrids3, telethon, audiolla, flickies, talkies, talkies-cuda. Override a specific `*_AUTH_TOKEN` / `*_API_TOKEN` var to scope that service separately. |\n| `LITELLM_MASTER_KEY` | Optional override; defaults to `AIGATE_TOKEN` when unset. |\n| `POSTGRES_DB` / `POSTGRES_USER` / `POSTGRES_PASSWORD` / `DATABASE_URL` | LiteLLM's key/usage/budget store. |\n| `REDIS_PASSWORD` | Password of the `proxq` Redis ACL user (job queue in DB 1). The `default` user is disabled. No whitespace. |\n| `LITELLM_REDIS_PASSWORD` | Password of the `litellm` Redis ACL user, which only holds the resource manager's hardware locks. Falls back to `REDIS_PASSWORD`. |\n| `LITELLM_UI_BASIC_AUTH` | `user:pass` for nginx basic auth in front of `/ui/`. Leave empty to disable (LiteLLM's own login still applies). |\n| `LITELLM_USERNAME` / `LITELLM_PASSWORD` | LiteLLM's own admin UI login. |\n\nAll requests carry `Authorization: Bearer $AIGATE_TOKEN` (or the relevant per-service override):\n\n```bash\ncurl http://localhost:4000/chat/completions \\\n  -H \"Authorization: Bearer $AIGATE_TOKEN\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"model\": \"local-ollama-cpu-llama3.2-3b\", \"messages\": [{\"role\": \"user\", \"content\": \"hello\"}]}'\n```\n\nPer-service keys/tokens only matter once you flip that service's flag to `1`. Every variable is documented inline in `.env.example` — read the comment above each block before enabling. Notable ones:\n\n- Cloud providers: `GROQ=1`, `CEREBRAS=1`, `OPENROUTER=1`, `HUGGINGFACE=1`, `MISTRAL=1`, `COHERE=1`, `ANTHROPIC=1`, `OPENAI=1` each need their own API key var alongside the flag (e.g. `OPENAI_API_KEY`).\n- `CLAUDEBOX=1` needs Claude OAuth token or Anthropic API key; token defaults to `AIGATE_TOKEN` via `CLAUDEBOX_API_TOKEN`.\n- `PIBOX_ZAI=1` needs a z.ai key; token defaults via `PIBOX_ZAI_API_TOKEN`.\n- `PIBOX=1` needs no outside account; it runs pi on this stack's own models. Set `PIBOX_MODELS` to enabled models that can call tools, and keep agent models (`claudebox-*`, `pibox-*`) out of that list to avoid recursion.\n- `DECIDEALOT=1` starts CPU typed decisions. `DECIDEALOT_CUDA=1` starts its GPU sibling. Laya, Von, and CLM are on by default. `make run-bg` also enables the CUDA Qwen encoder profile. CLM calls `http://litellm:4000/v1/embeddings` on the internal network with model `local-llamacpp-cuda-qwen3-8b`. Set `DECIDEALOT_CLM_ENABLED=false` to run without that GPU dependency. Set `DECIDEALOT_TYPESAFE_API_KEY` in the gitignored `.env` to enable hosted Jev. The `list_models` tool returns TypeSafe's current selectors, and `system_one_batch` accepts independent requests.\n- `MAILBOX=1` needs `MAILBOX_CONFIG` pointing at an existing host YAML file (copy `mailbox/config.example.yaml`, fill IMAP/SMTP creds, put a token under `auth.tokens:`) and `MAILBOX_AUTH_TOKEN` mirroring that token.\n- `TELETHON=1` needs `TELETHON_API_ID`, `TELETHON_API_HASH`, `TELETHON_SESSION` (generate the string session once via the telethon-plus `login` command — see `docs/services/telethon.md` upstream).\n- `CLOUDFLARED=1` / `TAILSCALE=1` are the two supported ways to expose the gateway beyond localhost without opening host ports — prefer these over publishing `4000` directly.\n- CUDA variants of any service (`*_CUDA=1`) require `nvidia-container-toolkit` on the host.\n\n## Model routing\n\nLiteLLM regenerates its config on every `make run`/`make run-bg`, including only enabled providers. Fallback chains (`litellm/config/fallbacks.json`) are priority-ordered and filtered to what's actually enabled:\n\n1. Free cloud (Groq, Cerebras, OpenRouter, HuggingFace, Mistral, Cohere) — rate-limited/capped, not unlimited.\n2. Subscription (claudebox, pibox-zai) — no per-token billing, but the allowance is metered.\n3. Pay-per-token (Anthropic, OpenAI) — real money per token.\n4. Local (Ollama, talkies, sd.cpp, vLLM, llama.cpp) — no external limits, bounded only by local hardware.\n\nModel names encode the route, e.g. `groq-gpt-oss-120b`, `local-ollama-cpu-llama3.2-3b`, `local-sdcpp-cuda-sd-turbo`. On a rate-limit or failure, LiteLLM automatically retries the next model in that model's fallback chain; the response's `model` field reports who actually served it. Async/long-running calls can go through `/q/` (proxq) instead of the sync path — submit, get a job ID back immediately, poll `/q/__jobs/{id}`.\n\nLiteLLM's resource manager serializes local jobs per hardware class and asks competing services to unload. Aigate's Decidealot launcher shares that Redis lock for Laya and Von, including direct REST, MCP, and batch items. Before CUDA inference it unloads Aigate's local llama.cpp encoder. CLM takes no outer lock because its nested LiteLLM embeddings call owns admission. Remote embeddings remain the remote service's responsibility. This launcher does not evict every other resident GPU service. Direct audiolla, flickies, and predictalot HTTP requests still bypass LiteLLM admission. `POST /v1/unload/{cuda,cpu}` requests manual cleanup across that hardware class. See [resource management](../../../../docs/resource-management.md).\n\n## Data / persistence\n\nAll persistent state lives under `.data/` (bind mounts), overridable via `DATA_DIR` or per-service `DATA_DIR_*`. Contents are gitignored; the directory tree itself is tracked via `.gitkeep`.\n\nFile v8.0.0:skill-card.md\n\n## Description:\n\nGuides deployment and use of a self-hosted, OpenAI-compatible AI gateway that combines model routing with optional tools and services.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[psyb0t](https://clawhub.ai/user/psyb0t)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and operators use this skill to set up a Docker-based AI gateway and connect clients to model providers and optional tools through one endpoint.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: One bearer token can grant access to enabled code execution, browser sessions, messaging, storage, and model-provider accounts.\n\nMitigation: Use a strong unique master token, separate sensitive per-service tokens, and share credentials only with trusted agents for authorized tasks.\n\nRisk: Exposing the gateway or enabling unnecessary services increases the impact of unauthorized access.\n\nMitigation: Install on a trusted host or isolated VM, enable only needed services, and do not expose port 4000 directly to the internet.\n\n## Reference(s):\n\n- [aigate release on ClawHub](https://clawhub.ai/psyb0t/skills/aigate)\n- [aigate setup guide](artifact/references/setup.md)\n\n## Skill Output:\n\n**Output Type(s):** [Guidance, Shell commands, Configuration instructions]\n\n**Output Format:** [Markdown with shell examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Includes examples for configuring services and calling the gateway.]\n\n## Skill Version(s):\n\n8.0.0 (source: ClawHub release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v7.0.0: 4 files, 10655 bytes\n\nFiles: references/setup.md (7799b), skill-card.md (1901b), SKILL.md (11921b), _meta.json (125b)\n\nFile v7.0.0:SKILL.md\n\n---\nname: aigate\ndescription: Self-hosted AI platform — one `docker-compose up`, one OpenAI-compatible endpoint at http://localhost:4000. Bundles inference (Groq/Cerebras/OpenRouter/HuggingFace/Mistral/Cohere/Ollama/vLLM/llama.cpp/claudebox/pibox-zai/Anthropic/OpenAI), MCP tool use, a stealth browser cluster, image generation (FLUX/DALL-E/SD), speech synthesis (Kokoro/Qwen3-TTS/Chatterbox/OpenAI TTS), transcription (Whisper/Parakeet), S3-compatible object storage, agentic code execution (Claude Code + pi-coding-agent + sandboxed piston), web search (SearXNG), an email gateway (mailbox), a Telegram client (Telethon), time-series forecasting + tabular ML (predictalot), audio/video production (audiolla/flickies), an async job queue (proxq), and a web UI (LibreChat) — all reachable through one bearer token and automatic per-model fallback routing. Use when the user wants a one-command self-hosted OpenAI-compatible stack that aggregates many providers/tools behind a single endpoint instead of wiring each service up individually.\nhomepage: https://github.com/psyb0t/aigate\nuser-invocable: true\npermissions:\n  network:\n    - outbound HTTP to the aigate endpoint (default http://localhost:4000, or wherever it's deployed)\n    - aigate itself reaches out to model providers, MCP tool backends, and the open internet on the user's behalf (web search, browser automation, email, Telegram)\n  shell:\n    - docker / docker compose (bring the stack up/down, read logs)\n    - curl (call the OpenAI-compatible endpoint and direct service routes)\nmetadata:\n  openclaw:\n    emoji: \"🚪\"\n    primaryEnv: AIGATE_TOKEN\n    requires:\n      bins:\n        - docker\n        - curl\n---\n\n# aigate — self-hosted AI platform behind one endpoint\n\nA self-hosted AI platform. One `docker-compose up` stands up inference, tool use, browser automation, image generation, speech synthesis, transcription, object storage, agentic code execution, web search, an email gateway, a Telegram client, time-series forecasting, an async job queue, and a web UI — all behind a single OpenAI-compatible endpoint at `http://localhost:4000`. Point any existing OpenAI-client library or `curl` at it and it works. Everything else is opt-in via `.env` flags; the always-on core is nginx, LiteLLM, PostgreSQL, Redis, and the `proxq` async job queue (at `/q/`, no flag needed).\n\n## Security & safety\n\n**This is a very high-capability, very high-blast-radius stack. Treat the endpoint and its token like root on the host.** A single `AIGATE_TOKEN` bearer can, depending on what's enabled:\n\n- Hold API keys/credentials for many cloud model providers (Groq, Cerebras, OpenRouter, HuggingFace, Mistral, Cohere, Anthropic, OpenAI) plus subscription agent backends (Claude Code OAuth/API key, z.ai GLM Coding Plan).\n- Execute arbitrary code — three full agentic coding agents (claudebox, pibox-zai, pibox) with shell + file access, plus sandboxed multi-language execution (piston).\n- Drive a real browser (stealth Camoufox cluster) that can log into sites, fill forms, and act as the user across the open web.\n- Send email and Telegram messages on the user's behalf (mailbox, Telethon) — mailbox additionally holds plaintext IMAP/SMTP credentials in its YAML config.\n- Read/write S3-compatible object storage with a public-read bucket.\n\n**No per-tool scoping by default.** `AIGATE_TOKEN` is a single all-or-nothing capability grant — every per-service token (`CLAUDEBOX_API_TOKEN`, `PIBOX_ZAI_API_TOKEN`, `PREDICTALOT_AUTH_TOKEN`, `DECIDEALOT_AUTH_TOKEN`, `AUDIOLLA_AUTH_TOKEN`, `FLICKIES_AUTH_TOKEN`, `STEALTHY_AUTO_BROWSE_AUTH_TOKEN`, `HYBRIDS3_MASTER_KEY`, `MCP_TOOLS_AUTH_TOKEN`, `TELETHON_AUTH_KEY`, etc.) defaults to it unless the operator explicitly overrides each one separately. Handing an agent the token is not \"give it chat access\" — it's granting code execution, browser automation, and messaging in one shot, with no way to grant a narrower subset unless the operator has pre-split the per-service tokens. An agent must only be given `AIGATE_TOKEN` when it is fully trusted and only for the specific action the user explicitly requested — never pass it to an agent \"just in case it needs something.\"\n\nTreat aigate as a **trusted host only**. Concretely:\n- Never expose port `4000` directly to the public internet. Use Cloudflare Tunnel (`CLOUDFLARED=1`) or Tailscale (`TAILSCALE=1`) — both keep no ports open on the host — or put a real authenticating gateway/reverse-proxy in front of it.\n- Every request needs `Authorization: Bearer $AIGATE_TOKEN` (or a per-service override token) — there is no unauthenticated path once a service is enabled. Don't hardcode the token in scripts committed to a repo; source it from `.env`/environment.\n- Internal services (Postgres, Redis, LiteLLM, and most optional services) bind to no host ports at all — only nginx is exposed. Don't add host port mappings for internal services unless you specifically need direct access and understand you're widening the blast radius.\n- `piston` runs `privileged: true` (required for nsjail's own isolation, not a bypass of it) and lives on an internal-only network with no outbound internet — don't change that without understanding why.\n- Guard `.env` and any mailbox/Telethon config files — they hold plaintext secrets and are gitignored for a reason.\n\n## When to use\n\n- The user wants a single self-hosted endpoint that speaks the OpenAI API and routes across many providers with automatic fallback (free-tier cloud → subscription → pay-per-token → local).\n- The user wants bundled AI tooling (browser automation, image/speech/transcription, code execution, storage, search, email, Telegram, forecasting) reachable via MCP tools or REST without standing up each service by hand.\n- The user wants to run models fully locally (CPU or NVIDIA GPU) with no external calls, or mix local + cloud with automatic fallback between them.\n- The user needs a chat web UI (LibreChat) pre-wired to every enabled model and tool.\n\n## When NOT to use\n\n- The user only needs one specific provider's API directly — aigate is overhead if the goal is just \"call OpenAI\" with no routing/tooling/fallback need.\n- Untrusted/multi-tenant exposure without a real auth gateway in front — aigate's bearer-token model is not a substitute for per-user authz.\n- The user needs Windows-native or non-Docker deployment — this stack is Docker Compose only.\n\n## Quick start\n\n```bash\ngit clone https://github.com/psyb0t/aigate\ncd aigate\nmake bootstrap # creates .env from .env.example (any target does this)\n# edit .env: set AIGATE_TOKEN, flip the flags for the providers/services you want to 1\n# .env is gitignored. docker-compose.yml is tracked, so put compose changes in\n# docker-compose.override.yml, which is gitignored and merges last\nmake limits    # checks enabled services fit this machine, writes CPU caps to .env.limits\nmake run-bg    # start the stack in the background\n```\n\nGateway is at `http://localhost:4000`. Call it like any OpenAI-compatible endpoint:\n\n```bash\ncurl http://localhost:4000/chat/completions \\\n  -H \"Authorization: Bearer $AIGATE_TOKEN\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"model\": \"local-ollama-cpu-llama3.2-3b\", \"messages\": [{\"role\": \"user\", \"content\": \"hello\"}]}'\n```\n\n`model` selects the provider/route; LiteLLM handles fallback automatically if the requested one rate-limits or fails. See `references/setup.md` for the full env/routing story.\n\n## What's bundled and how to reach it\n\nEverything below sits behind the same `http://localhost:4000` endpoint and the same `AIGATE_TOKEN` bearer — aigate's job is exposing them, not reimplementing them. Enable each with its `.env` flag; disabled services are excluded from routing/fallback entirely.\n\n- **Inference + routing** — `/chat/completions`, `/embeddings`, `/images/generations`, `/audio/*` (OpenAI-compatible, via LiteLLM). Model name picks the provider: free-tier cloud (Groq, Cerebras, OpenRouter, HuggingFace, Mistral, Cohere), subscription agents (claudebox = Claude Code, pibox-zai = pi-coding-agent on a GLM Coding Plan), pay-per-token (Anthropic, OpenAI), or fully local CPU/CUDA (Ollama, vLLM, llama.cpp, talkies, sd.cpp). `pibox` runs the same pi agent on whichever of these models you point it at, so it costs whatever that model costs. Fallback chains retry the next provider automatically on 429/5xx.\n- **MCP tool use** — any function-calling model can autonomously invoke `generate_image`, `generate_tts`, `search_web`, `execute_code`, and per-service MCP tools (browser, storage, mailbox, Telethon, predictalot, decidealot, audiolla, flickies, claudebox/pibox-zai/pibox agent tools). Auto-enabled with the underlying service.\n- **Browser automation** — `stealthy-auto-browse`, 5-replica stealth Camoufox cluster behind HAProxy. REST + MCP (`BROWSER=1`).\n- **Agentic code execution** — claudebox (Claude Code) and pibox-zai (pi-coding-agent/z.ai) for full shell+file agentic tasks; piston at `/piston/` for sandboxed nsjail-isolated one-shot code execution (`CLAUDEBOX=1`, `PIBOX_ZAI=1`, `PISTON=1`).\n- **Object storage** — `hybrids3` at `/storage/`, S3-compatible, plain HTTP + boto3, public-read uploads, presigned URLs (`HYBRIDS3=1`).\n- **Image generation** — cloud (FLUX, DALL-E, SD) and local CPU/CUDA (`SDCPP=1` / `SDCPP_CUDA=1`) via `/images/generations` or MCP.\n- **Speech synthesis + transcription** — `talkies` unifies both under `/audio/speech` and `/audio/transcriptions` (Kokoro, Qwen3-TTS, Chatterbox Turbo, Whisper, Parakeet, Canary, Sherpa-ONNX, Vosk, wav2vec2/ZIPA phoneme ASR — `TALKIES=1` / `TALKIES_CUDA=1`); cloud TTS/ASR routes through the same endpoints.\n- **Web search** — SearXNG at `/searxng/`, plus MCP `search_web` (`SEARXNG=1`).\n- **Email gateway** — `mailbox` at `/mailbox/`, stateless IMAP+SMTP across N accounts from one YAML config, REST + MCP (`MAILBOX=1`, needs `MAILBOX_CONFIG` + `MAILBOX_AUTH_TOKEN`).\n- **Telegram client** — `telethon` at `/telethon/`, REST + MCP (`TELETHON=1`, needs API ID/hash + string session).\n- **Time-series forecasting + tabular ML** — `predictalot` at `/predictalot/` (CPU) and `/predictalot-cuda/` (GPU), REST + MCP (`PREDICTALOT=1` / `PREDICTALOT_CUDA=1`).\n- **Typed decisions**: `decidealot` at `/decidealot/` (CPU) and `/decidealot-cuda/` (GPU), REST + MCP (`DECIDEALOT=1` / `DECIDEALOT_CUDA=1`). Laya, Von, and CLM are enabled by default. `make run-bg` also starts CLM's CUDA Qwen3-8B encoder. CLM calls AIGate internally at `http://litellm:4000/v1/embeddings`, selecting `local-llamacpp-cuda-qwen3-8b`. Set `DECIDEALOT_CLM_ENABLED=false` to remove that GPU dependency. Set `DECIDEALOT_TYPESAFE_API_KEY` in the gitignored `.env` to enable hosted Jev, then select a name from `list_models`. Answers have `choice` / `score` / `noul` types and probabilities. `system_one_batch` accepts independent requests. The caller owns the threshold and the action that follows.\n- **Audio production** — `audiolla` at `/audiolla/` / `/audiolla-cuda/` — stem separation, mastering, MIDI, text-to-audio, REST + MCP (`AUDIOLLA=1` / `AUDIOLLA_CUDA=1`).\n- **Video toolkit** — `flickies` at `/flickies/` / `/flickies-cuda/` — lipsync, face restore, ffmpeg ops, REST + MCP (`FLICKIES=1` / `FLICKIES_CUDA=1`).\n- **Async job queue** — `proxq` at `/q/` — queue any OpenAI-path request, poll `/q/__jobs/{id}`, avoids client-side timeouts on long inference.\n\n## Web UI\n\nLibreChat at `/librechat/` (`LIBRECHAT=1`) — pre-configured with every enabled model and MCP tool, conversation history, file uploads, WebSocket streaming. Email/password auth; the first registered user becomes admin (then set `LIBRECHAT_ALLOW_REGISTRATION=false`). Admin UI for LiteLLM itself is at `/ui/`, optionally behind nginx basic auth (`LITELLM_UI_BASIC_AUTH`).\n\n## Setup details\n\nFor docker-compose bring-up, required env/keys, ports, and model/routing config, see `references/setup.md`.\n\nFile v7.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn79dhvmpjng4rp2jjk8k0v5xx80ccbk\",\n  \"slug\": \"aigate\",\n  \"version\": \"7.0.0\",\n  \"publishedAt\": 1791027832594\n}\n\nFile v7.0.0:references/setup.md\n\n# aigate setup\n\nAccurate to the aigate `README.md` / `docker-compose.yml` / `.env.example` at time of writing. Re-check those files if this drifts. `.env` and `docker-compose.override.yml` are the operator's local files and are gitignored.\n\n## Bring-up\n\n```bash\ngit clone https://github.com/psyb0t/aigate\ncd aigate\nmake bootstrap   # creates .env from .env.example (any target does this)\n# edit .env — see \"Required env\" below, then flip service flags to 1\n# .env is gitignored. docker-compose.yml is tracked, so put compose changes in\n# docker-compose.override.yml, which is gitignored and merges last\nmake limits      # checks enabled services fit this machine, writes CPU caps to .env.limits\nmake run-bg      # start detached\n# or: make run   # start in foreground with logs\n```\n\n`make run` / `make run-bg` regenerate `litellm/config.yaml` from fragments (only enabled providers + filtered fallback chains) and pre-flight-validate any file-path env vars (e.g. `MAILBOX_CONFIG`, `CLOUDFLARED_CONFIG`) actually exist before starting containers.\n\nOther Makefile targets: `make down`, `make restart`, `make logs`, `make build-config` (regenerate litellm config only), `make test` (stack must already be running).\n\n## Ports\n\nSingle exposed port: **`4000`** (nginx), hardcoded — not env-configurable. Everything else (Postgres, Redis, LiteLLM, and most optional services) binds to no host port at all; they're reached only through nginx's path-based routing on `:4000`. Do not add host port mappings for internal services unless you specifically need direct access.\n\n- Gateway (OpenAI-compatible): `http://localhost:4000`\n- LiteLLM admin UI: `http://localhost:4000/ui/`\n- LibreChat (if `LIBRECHAT=1`): `http://localhost:4000/librechat/`\n- SearXNG (if `SEARXNG=1`): `http://localhost:4000/searxng/`\n- Async queue (proxq): `http://localhost:4000/q/`\n- Direct-routed services (not via LiteLLM): `/predictalot/`, `/predictalot-cuda/`, `/decidealot/`, `/decidealot-cuda/`, `/audiolla/`, `/audiolla-cuda/`, `/flickies/`, `/flickies-cuda/`, `/mailbox/`, `/telethon/`, `/piston/`, `/storage/` (hybrids3), `/claudebox/`, `/pibox-zai/`, `/pibox/`, `/stealthy-auto-browse/`\n\n## Required env / keys\n\nCore, always needed regardless of which optional services you enable:\n\n| Variable | Purpose |\n| --- | --- |\n| `AIGATE_TOKEN` | Master bearer token. Every per-service token below defaults to this value when left unset — one token authenticates against LiteLLM, claudebox, pibox-zai, predictalot, decidealot, mcp_tools, stealthy-auto-browse, hybrids3, telethon, audiolla, flickies, talkies, talkies-cuda. Override a specific `*_AUTH_TOKEN` / `*_API_TOKEN` var to scope that service separately. |\n| `LITELLM_MASTER_KEY` | Optional override; defaults to `AIGATE_TOKEN` when unset. |\n| `POSTGRES_DB` / `POSTGRES_USER` / `POSTGRES_PASSWORD` / `DATABASE_URL` | LiteLLM's key/usage/budget store. |\n| `REDIS_PASSWORD` | Password of the `proxq` Redis ACL user (job queue in DB 1). The `default` user is disabled. No whitespace. |\n| `LITELLM_REDIS_PASSWORD` | Password of the `litellm` Redis ACL user, which only holds the resource manager's hardware locks. Falls back to `REDIS_PASSWORD`. |\n| `LITELLM_UI_BASIC_AUTH` | `user:pass` for nginx basic auth in front of `/ui/`. Leave empty to disable (LiteLLM's own login still applies). |\n| `LITELLM_USERNAME` / `LITELLM_PASSWORD` | LiteLLM's own admin UI login. |\n\nAll requests carry `Authorization: Bearer $AIGATE_TOKEN` (or the relevant per-service override):\n\n```bash\ncurl http://localhost:4000/chat/completions \\\n  -H \"Authorization: Bearer $AIGATE_TOKEN\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"model\": \"local-ollama-cpu-llama3.2-3b\", \"messages\": [{\"role\": \"user\", \"content\": \"hello\"}]}'\n```\n\nPer-service keys/tokens only matter once you flip that service's flag to `1`. Every variable is documented inline in `.env.example` — read the comment above each block before enabling. Notable ones:\n\n- Cloud providers: `GROQ=1`, `CEREBRAS=1`, `OPENROUTER=1`, `HUGGINGFACE=1`, `MISTRAL=1`, `COHERE=1`, `ANTHROPIC=1`, `OPENAI=1` each need their own API key var alongside the flag (e.g. `OPENAI_API_KEY`).\n- `CLAUDEBOX=1` needs Claude OAuth token or Anthropic API key; token defaults to `AIGATE_TOKEN` via `CLAUDEBOX_API_TOKEN`.\n- `PIBOX_ZAI=1` needs a z.ai key; token defaults via `PIBOX_ZAI_API_TOKEN`.\n- `PIBOX=1` needs no outside account; it runs pi on this stack's own models. Set `PIBOX_MODELS` to enabled models that can call tools, and keep agent models (`claudebox-*`, `pibox-*`) out of that list to avoid recursion.\n- `DECIDEALOT=1` starts CPU typed decisions. `DECIDEALOT_CUDA=1` starts its GPU sibling. Laya, Von, and CLM are on by default. `make run-bg` also enables the CUDA Qwen encoder profile. CLM calls `http://litellm:4000/v1/embeddings` on the internal network with model `local-llamacpp-cuda-qwen3-8b`. Set `DECIDEALOT_CLM_ENABLED=false` to run without that GPU dependency. Set `DECIDEALOT_TYPESAFE_API_KEY` in the gitignored `.env` to enable hosted Jev. The `list_models` tool returns TypeSafe's current selectors, and `system_one_batch` accepts independent requests.\n- `MAILBOX=1` needs `MAILBOX_CONFIG` pointing at an existing host YAML file (copy `mailbox/config.example.yaml`, fill IMAP/SMTP creds, put a token under `auth.tokens:`) and `MAILBOX_AUTH_TOKEN` mirroring that token.\n- `TELETHON=1` needs `TELETHON_API_ID`, `TELETHON_API_HASH`, `TELETHON_SESSION` (generate the string session once via the telethon-plus `login` command — see `docs/services/telethon.md` upstream).\n- `CLOUDFLARED=1` / `TAILSCALE=1` are the two supported ways to expose the gateway beyond localhost without opening host ports — prefer these over publishing `4000` directly.\n- CUDA variants of any service (`*_CUDA=1`) require `nvidia-container-toolkit` on the host.\n\n## Model routing\n\nLiteLLM regenerates its config on every `make run`/`make run-bg`, including only enabled providers. Fallback chains (`litellm/config/fallbacks.json`) are priority-ordered and filtered to what's actually enabled:\n\n1. Free cloud (Groq, Cerebras, OpenRouter, HuggingFace, Mistral, Cohere) — rate-limited/capped, not unlimited.\n2. Subscription (claudebox, pibox-zai) — no per-token billing, but the allowance is metered.\n3. Pay-per-token (Anthropic, OpenAI) — real money per token.\n4. Local (Ollama, talkies, sd.cpp, vLLM, llama.cpp) — no external limits, bounded only by local hardware.\n\nModel names encode the route, e.g. `groq-gpt-oss-120b`, `local-ollama-cpu-llama3.2-3b`, `local-sdcpp-cuda-sd-turbo`. On a rate-limit or failure, LiteLLM automatically retries the next model in that model's fallback chain; the response's `model` field reports who actually served it. Async/long-running calls can go through `/q/` (proxq) instead of the sync path — submit, get a job ID back immediately, poll `/q/__jobs/{id}`.\n\nLiteLLM's resource manager serializes local jobs per hardware class and asks competing services to unload. Aigate's Decidealot launcher shares that Redis lock for Laya and Von, including direct REST, MCP, and batch items. Before CUDA inference it unloads Aigate's local llama.cpp encoder. CLM takes no outer lock because its nested LiteLLM embeddings call owns admission. Remote embeddings remain the remote service's responsibility. This launcher does not evict every other resident GPU service. Direct audiolla, flickies, and predictalot HTTP requests still bypass LiteLLM admission. `POST /v1/unload/{cuda,cpu}` requests manual cleanup across that hardware class. See [resource management](../../../../docs/resource-management.md).\n\n## Data / persistence\n\nAll persistent state lives under `.data/` (bind mounts), overridable via `DATA_DIR` or per-service `DATA_DIR_*`. Contents are gitignored; the directory tree itself is tracked via `.gitkeep`.\n\nFile v7.0.0:skill-card.md\n\n## Description:\n\nGuides developers in deploying and using a self-hosted, OpenAI-compatible AI gateway that connects model providers and optional tools behind one endpoint.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[psyb0t](https://clawhub.ai/user/psyb0t)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and operators use this skill to set up a Docker Compose gateway and connect agents to model inference, automation, and other optional services through one endpoint.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: A single bearer token may grant broad access to code execution, browser sessions, messaging, storage, and configured provider accounts.\n\nMitigation: Give access only to trusted agents for explicitly requested tasks; set distinct per-service tokens and enable only needed services.\n\nRisk: Exposing the gateway publicly can give token holders control over high-privilege services.\n\nMitigation: Install on a trusted host and keep port 4000 behind localhost, Tailscale, Cloudflare Tunnel, or an authenticating gateway.\n\n## Reference(s):\n\n- [aigate on ClawHub](https://clawhub.ai/psyb0t/skills/aigate)\n- [Setup guide](artifact/references/setup.md)\n\n## Skill Output:\n\n**Output Type(s):** [Guidance, Shell commands, Configuration instructions]\n\n**Output Format:** [Markdown with shell commands and configuration examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Instructions cover optional services and authenticated requests to a self-hosted gateway.]\n\n## Skill Version(s):\n\n7.0.0 (source: ClawHub release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v6.3.0: 4 files, 10545 bytes\n\nFiles: references/setup.md (7579b), skill-card.md (2052b), SKILL.md (11788b), _meta.json (125b)\n\nFile v6.3.0:SKILL.md\n\n---\nname: aigate\ndescription: Self-hosted AI platform — one `docker-compose up`, one OpenAI-compatible endpoint at http://localhost:4000. Bundles inference (Groq/Cerebras/OpenRouter/HuggingFace/Mistral/Cohere/Ollama/vLLM/llama.cpp/claudebox/pibox-zai/Anthropic/OpenAI), MCP tool use, a stealth browser cluster, image generation (FLUX/DALL-E/SD), speech synthesis (Kokoro/Qwen3-TTS/Chatterbox/OpenAI TTS), transcription (Whisper/Parakeet), S3-compatible object storage, agentic code execution (Claude Code + pi-coding-agent + sandboxed piston), web search (SearXNG), an email gateway (mailbox), a Telegram client (Telethon), time-series forecasting + tabular ML (predictalot), audio/video production (audiolla/flickies), an async job queue (proxq), and a web UI (LibreChat) — all reachable through one bearer token and automatic per-model fallback routing. Use when the user wants a one-command self-hosted OpenAI-compatible stack that aggregates many providers/tools behind a single endpoint instead of wiring each service up individually.\nhomepage: https://github.com/psyb0t/aigate\nuser-invocable: true\npermissions:\n  network:\n    - outbound HTTP to the aigate endpoint (default http://localhost:4000, or wherever it's deployed)\n    - aigate itself reaches out to model providers, MCP tool backends, and the open internet on the user's behalf (web search, browser automation, email, Telegram)\n  shell:\n    - docker / docker compose (bring the stack up/down, read logs)\n    - curl (call the OpenAI-compatible endpoint and direct service routes)\nmetadata:\n  openclaw:\n    emoji: \"🚪\"\n    primaryEnv: AIGATE_TOKEN\n    requires:\n      bins:\n        - docker\n        - curl\n---\n\n# aigate — self-hosted AI platform behind one endpoint\n\nA self-hosted AI platform. One `docker-compose up` stands up inference, tool use, browser automation, image generation, speech synthesis, transcription, object storage, agentic code execution, web search, an email gateway, a Telegram client, time-series forecasting, an async job queue, and a web UI — all behind a single OpenAI-compatible endpoint at `http://localhost:4000`. Point any existing OpenAI-client library or `curl` at it and it works. Everything else is opt-in via `.env` flags; the always-on core is nginx, LiteLLM, PostgreSQL, Redis, and the `proxq` async job queue (at `/q/`, no flag needed).\n\n## Security & safety\n\n**This is a very high-capability, very high-blast-radius stack. Treat the endpoint and its token like root on the host.** A single `AIGATE_TOKEN` bearer can, depending on what's enabled:\n\n- Hold API keys/credentials for many cloud model providers (Groq, Cerebras, OpenRouter, HuggingFace, Mistral, Cohere, Anthropic, OpenAI) plus subscription agent backends (Claude Code OAuth/API key, z.ai GLM Coding Plan).\n- Execute arbitrary code — three full agentic coding agents (claudebox, pibox-zai, pibox) with shell + file access, plus sandboxed multi-language execution (piston).\n- Drive a real browser (stealth Camoufox cluster) that can log into sites, fill forms, and act as the user across the open web.\n- Send email and Telegram messages on the user's behalf (mailbox, Telethon) — mailbox additionally holds plaintext IMAP/SMTP credentials in its YAML config.\n- Read/write S3-compatible object storage with a public-read bucket.\n\n**No per-tool scoping by default.** `AIGATE_TOKEN` is a single all-or-nothing capability grant — every per-service token (`CLAUDEBOX_API_TOKEN`, `PIBOX_ZAI_API_TOKEN`, `PREDICTALOT_AUTH_TOKEN`, `DECIDEALOT_AUTH_TOKEN`, `AUDIOLLA_AUTH_TOKEN`, `FLICKIES_AUTH_TOKEN`, `STEALTHY_AUTO_BROWSE_AUTH_TOKEN`, `HYBRIDS3_MASTER_KEY`, `MCP_TOOLS_AUTH_TOKEN`, `TELETHON_AUTH_KEY`, etc.) defaults to it unless the operator explicitly overrides each one separately. Handing an agent the token is not \"give it chat access\" — it's granting code execution, browser automation, and messaging in one shot, with no way to grant a narrower subset unless the operator has pre-split the per-service tokens. An agent must only be given `AIGATE_TOKEN` when it is fully trusted and only for the specific action the user explicitly requested — never pass it to an agent \"just in case it needs something.\"\n\nTreat aigate as a **trusted host only**. Concretely:\n- Never expose port `4000` directly to the public internet. Use Cloudflare Tunnel (`CLOUDFLARED=1`) or Tailscale (`TAILSCALE=1`) — both keep no ports open on the host — or put a real authenticating gateway/reverse-proxy in front of it.\n- Every request needs `Authorization: Bearer $AIGATE_TOKEN` (or a per-service override token) — there is no unauthenticated path once a service is enabled. Don't hardcode the token in scripts committed to a repo; source it from `.env`/environment.\n- Internal services (Postgres, Redis, LiteLLM, and most optional services) bind to no host ports at all — only nginx is exposed. Don't add host port mappings for internal services unless you specifically need direct access and understand you're widening the blast radius.\n- `piston` runs `privileged: true` (required for nsjail's own isolation, not a bypass of it) and lives on an internal-only network with no outbound internet — don't change that without understanding why.\n- Guard `.env` and any mailbox/Telethon config files — they hold plaintext secrets and are gitignored for a reason.\n\n## When to use\n\n- The user wants a single self-hosted endpoint that speaks the OpenAI API and routes across many providers with automatic fallback (free-tier cloud → subscription → pay-per-token → local).\n- The user wants bundled AI tooling (browser automation, image/speech/transcription, code execution, storage, search, email, Telegram, forecasting) reachable via MCP tools or REST without standing up each service by hand.\n- The user wants to run models fully locally (CPU or NVIDIA GPU) with no external calls, or mix local + cloud with automatic fallback between them.\n- The user needs a chat web UI (LibreChat) pre-wired to every enabled model and tool.\n\n## When NOT to use\n\n- The user only needs one specific provider's API directly — aigate is overhead if the goal is just \"call OpenAI\" with no routing/tooling/fallback need.\n- Untrusted/multi-tenant exposure without a real auth gateway in front — aigate's bearer-token model is not a substitute for per-user authz.\n- The user needs Windows-native or non-Docker deployment — this stack is Docker Compose only.\n\n## Quick start\n\n```bash\ngit clone https://github.com/psyb0t/aigate\ncd aigate\nmake bootstrap # creates .env from .env.example (any target does this)\n# edit .env: set AIGATE_TOKEN, flip the flags for the providers/services you want to 1\n# .env is gitignored. docker-compose.yml is tracked, so put compose changes in\n# docker-compose.override.yml, which is gitignored and merges last\nmake limits    # checks enabled services fit this machine, writes CPU caps to .env.limits\nmake run-bg    # start the stack in the background\n```\n\nGateway is at `http://localhost:4000`. Call it like any OpenAI-compatible endpoint:\n\n```bash\ncurl http://localhost:4000/chat/completions \\\n  -H \"Authorization: Bearer $AIGATE_TOKEN\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"model\": \"local-ollama-cpu-llama3.2-3b\", \"messages\": [{\"role\": \"user\", \"content\": \"hello\"}]}'\n```\n\n`model` selects the provider/route; LiteLLM handles fallback automatically if the requested one rate-limits or fails. See `references/setup.md` for the full env/routing story.\n\n## What's bundled and how to reach it\n\nEverything below sits behind the same `http://localhost:4000` endpoint and the same `AIGATE_TOKEN` bearer — aigate's job is exposing them, not reimplementing them. Enable each with its `.env` flag; disabled services are excluded from routing/fallback entirely.\n\n- **Inference + routing** — `/chat/completions`, `/embeddings`, `/images/generations`, `/audio/*` (OpenAI-compatible, via LiteLLM). Model name picks the provider: free-tier cloud (Groq, Cerebras, OpenRouter, HuggingFace, Mistral, Cohere), subscription agents (claudebox = Claude Code, pibox-zai = pi-coding-agent on a GLM Coding Plan), pay-per-token (Anthropic, OpenAI), or fully local CPU/CUDA (Ollama, vLLM, llama.cpp, talkies, sd.cpp). `pibox` runs the same pi agent on whichever of these models you point it at, so it costs whatever that model costs. Fallback chains retry the next provider automatically on 429/5xx.\n- **MCP tool use** — any function-calling model can autonomously invoke `generate_image`, `generate_tts`, `search_web`, `execute_code`, and per-service MCP tools (browser, storage, mailbox, Telethon, predictalot, decidealot, audiolla, flickies, claudebox/pibox-zai/pibox agent tools). Auto-enabled with the underlying service.\n- **Browser automation** — `stealthy-auto-browse`, 5-replica stealth Camoufox cluster behind HAProxy. REST + MCP (`BROWSER=1`).\n- **Agentic code execution** — claudebox (Claude Code) and pibox-zai (pi-coding-agent/z.ai) for full shell+file agentic tasks; piston at `/piston/` for sandboxed nsjail-isolated one-shot code execution (`CLAUDEBOX=1`, `PIBOX_ZAI=1`, `PISTON=1`).\n- **Object storage** — `hybrids3` at `/storage/`, S3-compatible, plain HTTP + boto3, public-read uploads, presigned URLs (`HYBRIDS3=1`).\n- **Image generation** — cloud (FLUX, DALL-E, SD) and local CPU/CUDA (`SDCPP=1` / `SDCPP_CUDA=1`) via `/images/generations` or MCP.\n- **Speech synthesis + transcription** — `talkies` unifies both under `/audio/speech` and `/audio/transcriptions` (Kokoro, Qwen3-TTS, Chatterbox Turbo, Whisper, Parakeet, Canary, Sherpa-ONNX, Vosk, wav2vec2/ZIPA phoneme ASR — `TALKIES=1` / `TALKIES_CUDA=1`); cloud TTS/ASR routes through the same endpoints.\n- **Web search** — SearXNG at `/searxng/`, plus MCP `search_web` (`SEARXNG=1`).\n- **Email gateway** — `mailbox` at `/mailbox/`, stateless IMAP+SMTP across N accounts from one YAML config, REST + MCP (`MAILBOX=1`, needs `MAILBOX_CONFIG` + `MAILBOX_AUTH_TOKEN`).\n- **Telegram client** — `telethon` at `/telethon/`, REST + MCP (`TELETHON=1`, needs API ID/hash + string session).\n- **Time-series forecasting + tabular ML** — `predictalot` at `/predictalot/` (CPU) and `/predictalot-cuda/` (GPU), REST + MCP (`PREDICTALOT=1` / `PREDICTALOT_CUDA=1`).\n- **Typed decisions**: `decidealot` at `/decidealot/` (CPU) and `/decidealot-cuda/` (GPU), REST + MCP (`DECIDEALOT=1` / `DECIDEALOT_CUDA=1`). Local Laya and Von models return `choice` / `score` / `noul` answers with probabilities. Set `DECIDEALOT_CLM_ENABLED=true` with `LLAMACPP_CUDA=1` to add CLM through AIGate's local Qwen3-8B embeddings route. Set `DECIDEALOT_TYPESAFE_API_KEY` in the gitignored `.env` to enable hosted Jev, then select a name from `list_models`. `system_one_batch` accepts independent requests. The caller owns the threshold and the action that follows.\n- **Audio production** — `audiolla` at `/audiolla/` / `/audiolla-cuda/` — stem separation, mastering, MIDI, text-to-audio, REST + MCP (`AUDIOLLA=1` / `AUDIOLLA_CUDA=1`).\n- **Video toolkit** — `flickies` at `/flickies/` / `/flickies-cuda/` — lipsync, face restore, ffmpeg ops, REST + MCP (`FLICKIES=1` / `FLICKIES_CUDA=1`).\n- **Async job queue** — `proxq` at `/q/` — queue any OpenAI-path request, poll `/q/__jobs/{id}`, avoids client-side timeouts on long inference.\n\n## Web UI\n\nLibreChat at `/librechat/` (`LIBRECHAT=1`) — pre-configured with every enabled model and MCP tool, conversation history, file uploads, WebSocket streaming. Email/password auth; the first registered user becomes admin (then set `LIBRECHAT_ALLOW_REGISTRATION=false`). Admin UI for LiteLLM itself is at `/ui/`, optionally behind nginx basic auth (`LITELLM_UI_BASIC_AUTH`).\n\n## Setup details\n\nFor docker-compose bring-up, required env/keys, ports, and model/routing config, see `references/setup.md`.\n\nFile v6.3.0:_meta.json\n\n{\n  \"ownerId\": \"kn79dhvmpjng4rp2jjk8k0v5xx80ccbk\",\n  \"slug\": \"aigate\",\n  \"version\": \"6.3.0\",\n  \"publishedAt\": 1790972481959\n}\n\nFile v6.3.0:references/setup.md\n\n# aigate setup\n\nAccurate to the aigate `README.md` / `docker-compose.yml` / `.env.example` at time of writing. Re-check those files if this drifts. `.env` and `docker-compose.override.yml` are the operator's local files and are gitignored.\n\n## Bring-up\n\n```bash\ngit clone https://github.com/psyb0t/aigate\ncd aigate\nmake bootstrap   # creates .env from .env.example (any target does this)\n# edit .env — see \"Required env\" below, then flip service flags to 1\n# .env is gitignored. docker-compose.yml is tracked, so put compose changes in\n# docker-compose.override.yml, which is gitignored and merges last\nmake limits      # checks enabled services fit this machine, writes CPU caps to .env.limits\nmake run-bg      # start detached\n# or: make run   # start in foreground with logs\n```\n\n`make run` / `make run-bg` regenerate `litellm/config.yaml` from fragments (only enabled providers + filtered fallback chains) and pre-flight-validate any file-path env vars (e.g. `MAILBOX_CONFIG`, `CLOUDFLARED_CONFIG`) actually exist before starting containers.\n\nOther Makefile targets: `make down`, `make restart`, `make logs`, `make build-config` (regenerate litellm config only), `make test` (stack must already be running).\n\n## Ports\n\nSingle exposed port: **`4000`** (nginx), hardcoded — not env-configurable. Everything else (Postgres, Redis, LiteLLM, and most optional services) binds to no host port at all; they're reached only through nginx's path-based routing on `:4000`. Do not add host port mappings for internal services unless you specifically need direct access.\n\n- Gateway (OpenAI-compatible): `http://localhost:4000`\n- LiteLLM admin UI: `http://localhost:4000/ui/`\n- LibreChat (if `LIBRECHAT=1`): `http://localhost:4000/librechat/`\n- SearXNG (if `SEARXNG=1`): `http://localhost:4000/searxng/`\n- Async queue (proxq): `http://localhost:4000/q/`\n- Direct-routed services (not via LiteLLM): `/predictalot/`, `/predictalot-cuda/`, `/decidealot/`, `/decidealot-cuda/`, `/audiolla/`, `/audiolla-cuda/`, `/flickies/`, `/flickies-cuda/`, `/mailbox/`, `/telethon/`, `/piston/`, `/storage/` (hybrids3), `/claudebox/`, `/pibox-zai/`, `/pibox/`, `/stealthy-auto-browse/`\n\n## Required env / keys\n\nCore, always needed regardless of which optional services you enable:\n\n| Variable | Purpose |\n| --- | --- |\n| `AIGATE_TOKEN` | Master bearer token. Every per-service token below defaults to this value when left unset — one token authenticates against LiteLLM, claudebox, pibox-zai, predictalot, decidealot, mcp_tools, stealthy-auto-browse, hybrids3, telethon, audiolla, flickies, talkies, talkies-cuda. Override a specific `*_AUTH_TOKEN` / `*_API_TOKEN` var to scope that service separately. |\n| `LITELLM_MASTER_KEY` | Optional override; defaults to `AIGATE_TOKEN` when unset. |\n| `POSTGRES_DB` / `POSTGRES_USER` / `POSTGRES_PASSWORD` / `DATABASE_URL` | LiteLLM's key/usage/budget store. |\n| `REDIS_PASSWORD` | Password of the `proxq` Redis ACL user (job queue in DB 1). The `default` user is disabled. No whitespace. |\n| `LITELLM_REDIS_PASSWORD` | Password of the `litellm` Redis ACL user, which only holds the resource manager's hardware locks. Falls back to `REDIS_PASSWORD`. |\n| `LITELLM_UI_BASIC_AUTH` | `user:pass` for nginx basic auth in front of `/ui/`. Leave empty to disable (LiteLLM's own login still applies). |\n| `LITELLM_USERNAME` / `LITELLM_PASSWORD` | LiteLLM's own admin UI login. |\n\nAll requests carry `Authorization: Bearer $AIGATE_TOKEN` (or the relevant per-service override):\n\n```bash\ncurl http://localhost:4000/chat/completions \\\n  -H \"Authorization: Bearer $AIGATE_TOKEN\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"model\": \"local-ollama-cpu-llama3.2-3b\", \"messages\": [{\"role\": \"user\", \"content\": \"hello\"}]}'\n```\n\nPer-service keys/tokens only matter once you flip that service's flag to `1`. Every variable is documented inline in `.env.example` — read the comment above each block before enabling. Notable ones:\n\n- Cloud providers: `GROQ=1`, `CEREBRAS=1`, `OPENROUTER=1`, `HUGGINGFACE=1`, `MISTRAL=1`, `COHERE=1`, `ANTHROPIC=1`, `OPENAI=1` each need their own API key var alongside the flag (e.g. `OPENAI_API_KEY`).\n- `CLAUDEBOX=1` needs Claude OAuth token or Anthropic API key; token defaults to `AIGATE_TOKEN` via `CLAUDEBOX_API_TOKEN`.\n- `PIBOX_ZAI=1` needs a z.ai key; token defaults via `PIBOX_ZAI_API_TOKEN`.\n- `PIBOX=1` needs no outside account; it runs pi on this stack's own models. Set `PIBOX_MODELS` to enabled models that can call tools, and keep agent models (`claudebox-*`, `pibox-*`) out of that list to avoid recursion.\n- `DECIDEALOT=1` starts CPU typed decisions. `DECIDEALOT_CUDA=1` starts its GPU sibling. Laya and Von are on by default. To add CLM, set `DECIDEALOT_CLM_ENABLED=true` and `LLAMACPP_CUDA=1`; AIGate then routes CLM embeddings to its internal `local-llamacpp-cuda-qwen3-8b` model. Set `DECIDEALOT_TYPESAFE_API_KEY` in the gitignored `.env` to add hosted Jev. The `list_models` tool returns TypeSafe's current selectors, and `system_one_batch` accepts independent requests.\n- `MAILBOX=1` needs `MAILBOX_CONFIG` pointing at an existing host YAML file (copy `mailbox/config.example.yaml`, fill IMAP/SMTP creds, put a token under `auth.tokens:`) and `MAILBOX_AUTH_TOKEN` mirroring that token.\n- `TELETHON=1` needs `TELETHON_API_ID`, `TELETHON_API_HASH`, `TELETHON_SESSION` (generate the string session once via the telethon-plus `login` command — see `docs/services/telethon.md` upstream).\n- `CLOUDFLARED=1` / `TAILSCALE=1` are the two supported ways to expose the gateway beyond localhost without opening host ports — prefer these over publishing `4000` directly.\n- CUDA variants of any service (`*_CUDA=1`) require `nvidia-container-toolkit` on the host.\n\n## Model routing\n\nLiteLLM regenerates its config on every `make run`/`make run-bg`, including only enabled providers. Fallback chains (`litellm/config/fallbacks.json`) are priority-ordered and filtered to what's actually enabled:\n\n1. Free cloud (Groq, Cerebras, OpenRouter, HuggingFace, Mistral, Cohere) — rate-limited/capped, not unlimited.\n2. Subscription (claudebox, pibox-zai) — no per-token billing, but the allowance is metered.\n3. Pay-per-token (Anthropic, OpenAI) — real money per token.\n4. Local (Ollama, talkies, sd.cpp, vLLM, llama.cpp) — no external limits, bounded only by local hardware.\n\nModel names encode the route, e.g. `groq-gpt-oss-120b`, `local-ollama-cpu-llama3.2-3b`, `local-sdcpp-cuda-sd-turbo`. On a rate-limit or failure, LiteLLM automatically retries the next model in that model's fallback chain; the response's `model` field reports who actually served it. Async/long-running calls can go through `/q/` (proxq) instead of the sync path — submit, get a job ID back immediately, poll `/q/__jobs/{id}`.\n\nResource contention on local CUDA/CPU services (LLM vs image-gen vs TTS/STT all fighting for the same GPU) is handled automatically by a LiteLLM callback (`resource_manager.py`) — one job per hardware class at a time, idle models auto-unload, competing services get told to free VRAM/RAM before a request proceeds. No manual model management needed. The direct-routed services (audiolla, flickies, predictalot, decidealot) get evicted this way too, but a request sent straight to one of them does not evict LiteLLM-routed models. `POST /v1/unload/{cuda,cpu}` frees everything on one hardware class by hand.\n\n## Data / persistence\n\nAll persistent state lives under `.data/` (bind mounts), overridable via `DATA_DIR` or per-service `DATA_DIR_*`. Contents are gitignored; the directory tree itself is tracked via `.gitkeep`.\n\nFile v6.3.0:skill-card.md\n\n## Description:\n\nGuides developers in deploying a self-hosted, OpenAI-compatible AI gateway that brings optional models and tools behind one endpoint.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[psyb0t](https://clawhub.ai/user/psyb0t)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and operators use this skill to set up and access a self-hosted AI gateway with optional inference, agent tools, browser automation, and media services.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: One bearer token can grant broad access to code execution, browser automation, messaging, storage, and administration.\n\nMitigation: Give agents only the access required for an approved task; set separate per-service tokens before granting access.\n\nRisk: Exposing the gateway publicly can put high-privilege services within reach of unauthorized users.\n\nMitigation: Keep the gateway behind a VPN, SSO, or a strongly authenticated proxy; do not publish its port directly.\n\nRisk: Enabled services may hold provider keys and messaging account credentials.\n\nMitigation: Enable only needed services and protect the environment file and mailbox and Telegram configuration.\n\n## Reference(s):\n\n- [aigate setup guide](references/setup.md)\n- [ClawHub skill release](https://clawhub.ai/psyb0t/skills/aigate)\n- [Project homepage (listed in skill documentation)](https://github.com/psyb0t/aigate)\n\n## Skill Output:\n\n**Output Type(s):** [Shell commands, Configuration instructions, Guidance]\n\n**Output Format:** [Markdown with shell command examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Available services depend on operator configuration.]\n\n## Skill Version(s):\n\n6.3.0 (source: ClawHub release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v6.2.0: 4 files, 10371 bytes\n\nFiles: references/setup.md (7387b), skill-card.md (2039b), SKILL.md (11619b), _meta.json (125b)\n\nFile v6.2.0:SKILL.md\n\n---\nname: aigate\ndescription: Self-hosted AI platform — one `docker-compose up`, one OpenAI-compatible endpoint at http://localhost:4000. Bundles inference (Groq/Cerebras/OpenRouter/HuggingFace/Mistral/Cohere/Ollama/vLLM/llama.cpp/claudebox/pibox-zai/Anthropic/OpenAI), MCP tool use, a stealth browser cluster, image generation (FLUX/DALL-E/SD), speech synthesis (Kokoro/Qwen3-TTS/Chatterbox/OpenAI TTS), transcription (Whisper/Parakeet), S3-compatible object storage, agentic code execution (Claude Code + pi-coding-agent + sandboxed piston), web search (SearXNG), an email gateway (mailbox), a Telegram client (Telethon), time-series forecasting + tabular ML (predictalot), audio/video production (audiolla/flickies), an async job queue (proxq), and a web UI (LibreChat) — all reachable through one bearer token and automatic per-model fallback routing. Use when the user wants a one-command self-hosted OpenAI-compatible stack that aggregates many providers/tools behind a single endpoint instead of wiring each service up individually.\nhomepage: https://github.com/psyb0t/aigate\nuser-invocable: true\npermissions:\n  network:\n    - outbound HTTP to the aigate endpoint (default http://localhost:4000, or wherever it's deployed)\n    - aigate itself reaches out to model providers, MCP tool backends, and the open internet on the user's behalf (web search, browser automation, email, Telegram)\n  shell:\n    - docker / docker compose (bring the stack up/down, read logs)\n    - curl (call the OpenAI-compatible endpoint and direct service routes)\nmetadata:\n  openclaw:\n    emoji: \"🚪\"\n    primaryEnv: AIGATE_TOKEN\n    requires:\n      bins:\n        - docker\n        - curl\n---\n\n# aigate — self-hosted AI platform behind one endpoint\n\nA self-hosted AI platform. One `docker-compose up` stands up inference, tool use, browser automation, image generation, speech synthesis, transcription, object storage, agentic code execution, web search, an email gateway, a Telegram client, time-series forecasting, an async job queue, and a web UI — all behind a single OpenAI-compatible endpoint at `http://localhost:4000`. Point any existing OpenAI-client library or `curl` at it and it works. Everything else is opt-in via `.env` flags; the always-on core is nginx, LiteLLM, PostgreSQL, Redis, and the `proxq` async job queue (at `/q/`, no flag needed).\n\n## Security & safety\n\n**This is a very high-capability, very high-blast-radius stack. Treat the endpoint and its token like root on the host.** A single `AIGATE_TOKEN` bearer can, depending on what's enabled:\n\n- Hold API keys/credentials for many cloud model providers (Groq, Cerebras, OpenRouter, HuggingFace, Mistral, Cohere, Anthropic, OpenAI) plus subscription agent backends (Claude Code OAuth/API key, z.ai GLM Coding Plan).\n- Execute arbitrary code — three full agentic coding agents (claudebox, pibox-zai, pibox) with shell + file access, plus sandboxed multi-language execution (piston).\n- Drive a real browser (stealth Camoufox cluster) that can log into sites, fill forms, and act as the user across the open web.\n- Send email and Telegram messages on the user's behalf (mailbox, Telethon) — mailbox additionally holds plaintext IMAP/SMTP credentials in its YAML config.\n- Read/write S3-compatible object storage with a public-read bucket.\n\n**No per-tool scoping by default.** `AIGATE_TOKEN` is a single all-or-nothing capability grant — every per-service token (`CLAUDEBOX_API_TOKEN`, `PIBOX_ZAI_API_TOKEN`, `PREDICTALOT_AUTH_TOKEN`, `DECIDEALOT_AUTH_TOKEN`, `AUDIOLLA_AUTH_TOKEN`, `FLICKIES_AUTH_TOKEN`, `STEALTHY_AUTO_BROWSE_AUTH_TOKEN`, `HYBRIDS3_MASTER_KEY`, `MCP_TOOLS_AUTH_TOKEN`, `TELETHON_AUTH_KEY`, etc.) defaults to it unless the operator explicitly overrides each one separately. Handing an agent the token is not \"give it chat access\" — it's granting code execution, browser automation, and messaging in one shot, with no way to grant a narrower subset unless the operator has pre-split the per-service tokens. An agent must only be given `AIGATE_TOKEN` when it is fully trusted and only for the specific action the user explicitly requested — never pass it to an agent \"just in case it needs something.\"\n\nTreat aigate as a **trusted host only**. Concretely:\n- Never expose port `4000` directly to the public internet. Use Cloudflare Tunnel (`CLOUDFLARED=1`) or Tailscale (`TAILSCALE=1`) — both keep no ports open on the host — or put a real authenticating gateway/reverse-proxy in front of it.\n- Every request needs `Authorization: Bearer $AIGATE_TOKEN` (or a per-service override token) — there is no unauthenticated path once a service is enabled. Don't hardcode the token in scripts committed to a repo; source it from `.env`/environment.\n- Internal services (Postgres, Redis, LiteLLM, and most optional services) bind to no host ports at all — only nginx is exposed. Don't add host port mappings for internal services unless you specifically need direct access and understand you're widening the blast radius.\n- `piston` runs `privileged: true` (required for nsjail's own isolation, not a bypass of it) and lives on an internal-only network with no outbound internet — don't change that without understanding why.\n- Guard `.env` and any mailbox/Telethon config files — they hold plaintext secrets and are gitignored for a reason.\n\n## When to use\n\n- The user wants a single self-hosted endpoint that speaks the OpenAI API and routes across many providers with automatic fallback (free-tier cloud → subscription → pay-per-token → local).\n- The user wants bundled AI tooling (browser automation, image/speech/transcription, code execution, storage, search, email, Telegram, forecasting) reachable via MCP tools or REST without standing up each service by hand.\n- The user wants to run models fully locally (CPU or NVIDIA GPU) with no external calls, or mix local + cloud with automatic fallback between them.\n- The user needs a chat web UI (LibreChat) pre-wired to every enabled model and tool.\n\n## When NOT to use\n\n- The user only needs one specific provider's API directly — aigate is overhead if the goal is just \"call OpenAI\" with no routing/tooling/fallback need.\n- Untrusted/multi-tenant exposure without a real auth gateway in front — aigate's bearer-token model is not a substitute for per-user authz.\n- The user needs Windows-native or non-Docker deployment — this stack is Docker Compose only.\n\n## Quick start\n\n```bash\ngit clone https://github.com/psyb0t/aigate\ncd aigate\nmake bootstrap # creates .env from .env.example (any target does this)\n# edit .env: set AIGATE_TOKEN, flip the flags for the providers/services you want to 1\n# .env is gitignored. docker-compose.yml is tracked, so put compose changes in\n# docker-compose.override.yml, which is gitignored and merges last\nmake limits    # checks enabled services fit this machine, writes CPU caps to .env.limits\nmake run-bg    # start the stack in the background\n```\n\nGateway is at `http://localhost:4000`. Call it like any OpenAI-compatible endpoint:\n\n```bash\ncurl http://localhost:4000/chat/completions \\\n  -H \"Authorization: Bearer $AIGATE_TOKEN\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"model\": \"local-ollama-cpu-llama3.2-3b\", \"messages\": [{\"role\": \"user\", \"content\": \"hello\"}]}'\n```\n\n`model` selects the provider/route; LiteLLM handles fallback automatically if the requested one rate-limits or fails. See `references/setup.md` for the full env/routing story.\n\n## What's bundled and how to reach it\n\nEverything below sits behind the same `http://localhost:4000` endpoint and the same `AIGATE_TOKEN` bearer — aigate's job is exposing them, not reimplementing them. Enable each with its `.env` flag; disabled services are excluded from routing/fallback entirely.\n\n- **Inference + routing** — `/chat/completions`, `/embeddings`, `/images/generations`, `/audio/*` (OpenAI-compatible, via LiteLLM). Model name picks the provider: free-tier cloud (Groq, Cerebras, OpenRouter, HuggingFace, Mistral, Cohere), subscription agents (claudebox = Claude Code, pibox-zai = pi-coding-agent on a GLM Coding Plan), pay-per-token (Anthropic, OpenAI), or fully local CPU/CUDA (Ollama, vLLM, llama.cpp, talkies, sd.cpp). `pibox` runs the same pi agent on whichever of\n\nArchive v6.1.0: 4 files, 10141 bytes\n\nFiles: references/setup.md (7109b), skill-card.md (1966b), SKILL.md (11501b), _meta.json (125b)\n\nArchive v6.0.0: 4 files, 10109 bytes\n\nFiles: references/setup.md (7109b), skill-card.md (1899b), SKILL.md (11501b), _meta.json (125b)\n\nArchive v5.7.0: 4 files, 10074 bytes\n\nFiles: references/setup.md (6889b), skill-card.md (1994b), SKILL.md (11479b), _meta.json (125b)","readmeExcerpt":"Skill: aigate Owner: psyb0t Summary: Self-hosted AI platform — one docker-compose up, one OpenAI-compatible endpoint at http://localhost:4000. Bundles inference (Groq/Cerebras/OpenRouter/HuggingFace/Mistral/Cohere/Ollama/vLLM/llama.cpp/claudebox/pibox-zai/Anthropic/OpenAI), MCP tool use, a stealth browser cluster, image generation (FLUX/DALL-E/SD), speech synthesis (Kokoro/Qwen3-TTS/Chatterbox/OpenAI TTS), transcript","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"git clone https://github.com/psyb0t/aigate\ncd aigate\nmake bootstrap # creates .env from .env.example (any target does this)\n# edit .env: set AIGATE_TOKEN, flip the flags for the providers/services you want to 1\n# .env is gitignored. docker-compose.yml is tracked, so put compose changes in\n# docker-compose.override.yml, which is gitignored and merges last\nmake limits    # checks enabled services fit this machine, writes CPU caps to .env.limits\nmake run-bg    # start the stack in the background"},{"language":"bash","snippet":"curl http://localhost:4000/chat/completions \\\n  -H \"Authorization: Bearer $AIGATE_TOKEN\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"model\": \"local-ollama-cpu-llama3.2-3b\", \"messages\": [{\"role\": \"user\", \"content\": \"hello\"}]}'"},{"language":"bash","snippet":"curl http://localhost:4000/chat/completions \\\n  -H \"Authorization: Bearer $AIGATE_TOKEN\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"model\": \"local-ollama-cpu-llama3.2-3b\", \"messages\": [{\"role\": \"user\", \"content\": \"hello\"}]}'"},{"language":"bash","snippet":"git clone https://github.com/psyb0t/aigate\ncd aigate\nmake bootstrap   # creates .env from .env.example (any target does this)\n# edit .env — see \"Required env\" below, then flip service flags to 1\n# .env is gitignored. docker-compose.yml is tracked, so put compose changes in\n# docker-compose.override.yml, which is gitignored and merges last\nmake limits      # checks enabled services fit this machine, writes CPU caps to .env.limits\nmake run-bg      # start detached\n# or: make run   # start in foreground with logs"},{"language":"bash","snippet":"curl http://localhost:4000/chat/completions \\\n  -H \"Authorization: Bearer $AIGATE_TOKEN\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"model\": \"local-ollama-cpu-llama3.2-3b\", \"messages\": [{\"role\": \"user\", \"content\": \"hello\"}]}'"},{"language":"bash","snippet":"curl http://localhost:4000/chat/completions \\\n  -H \"Authorization: Bearer $AIGATE_TOKEN\" \\\n  -H \"Content-Type: application/json\" \\\n  -d '{\"model\": \"local-ollama-cpu-llama3.2-3b\", \"messages\": [{\"role\": \"user\", \"content\": \"hello\"}]}'"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: aigate\ndescription: Self-hosted AI platform — one `docker-compose up`, one OpenAI-compatible endpoint at http://localhost:4000. Bundles inference (Groq/Cerebras/OpenRouter/HuggingFace/Mistral/Cohere/Ollama/vLLM/llama.cpp/claudebox/pibox-zai/Anthropic/OpenAI), MCP tool use, a stealth browser cluster, image generation (FLUX/DALL-E/SD), speech synthesis (Kokoro/Qwen3-TTS/Chatterbox/OpenAI TTS), transcription (Whisper/Parakeet), S3-compatible object storage, agentic code execution (Claude Code + pi-coding-agent + sandboxed piston), web search (SearXNG), an email gateway (mailbox), a Telegram client (Telethon), time-series forecasting + tabular ML (predictalot), audio/video production (audiolla/flickies), an async job queue (proxq), and a web UI (LibreChat) — all reachable through one bearer token and automatic per-model fallback routing. Use when the user wants a one-command self-hosted OpenAI-compatible stack that aggregates many providers/tools behind a single endpoint instead of wiring each service up individually.\nhomepage: https://github.com/psyb0t/aigate\nuser-invocable: true\npermissions:\n  network:\n    - outbound HTTP to the aigate endpoint (default http://localhost:4000, or wherever it's deployed)\n    - aigate itself reaches out to model providers, MCP tool backends, and the open internet on the user's behalf (web search, browser automation, email, Telegram)\n  shell:\n    - docker / docker compose (bring the stack up/down, read logs)\n    - curl (call the OpenAI-compatible endpoint and direct service routes)\nmetadata:\n  openclaw:\n    emoji: \"🚪\"\n    primaryEnv: AIGATE_TOKEN\n    requires:\n      bins:\n        - docker\n        - curl\n---\n\n# aigate — self-hosted AI platform behind one endpoint\n\nA self-hosted AI platform. One `docker-compose up` stands up inference, tool use, browser automation, image generation, speech synthesis, transcription, object storage, agentic code execution, web search, an email gateway, a Telegram client, time-series forecasting, an async job queue, and a web UI — all behind a single OpenAI-compatible endpoint at `http://localhost:4000`. Point any existing OpenAI-client library or `curl` at it and it works. Everything else is opt-in via `.env` flags; the always-on core is nginx, LiteLLM, PostgreSQL, Redis, and the `proxq` async job queue (at `/q/`, no flag needed).\n\n## Security & safety\n\n**This is a very high-capability, very high-blast-radius stack. Treat the endpoint and its token like root on the host.** A single `AIGATE_TOKEN` bearer can, depending on what's enabled:\n\n- Hold API keys/credentials for many cloud model providers (Groq, Cerebras, OpenRouter, HuggingFace, Mistral, Cohere, Anthropic, OpenAI) plus subscription agent backends (Claude Code OAuth/API key, z.ai GLM Coding Plan).\n- Execute arbitrary code — three full agentic coding agents (claudebox, pibox-zai, pibox) with shell + file access, plus sandboxed multi-language execution (piston).\n- Drive a real browser (stealth Camoufox cluster) that can log"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn79dhvmpjng4rp2jjk8k0v5xx80ccbk\",\n  \"slug\": \"aigate\",\n  \"version\": \"10.0.0\",\n  \"publishedAt\": 1791475477433\n}"},{"path":"references/setup.md","content":"# aigate setup\n\nAccurate to the aigate `README.md` / `docker-compose.yml` / `.env.example` at time of writing. Re-check those files if this drifts. `.env` and `docker-compose.override.yml` are the operator's local files and are gitignored.\n\n## Bring-up\n\n```bash\ngit clone https://github.com/psyb0t/aigate\ncd aigate\nmake bootstrap   # creates .env from .env.example (any target does this)\n# edit .env — see \"Required env\" below, then flip service flags to 1\n# .env is gitignored. docker-compose.yml is tracked, so put compose changes in\n# docker-compose.override.yml, which is gitignored and merges last\nmake limits      # checks enabled services fit this machine, writes CPU caps to .env.limits\nmake run-bg      # start detached\n# or: make run   # start in foreground with logs\n```\n\n`make restart-audiolla` recreates only enabled Audiolla CPU/CUDA variants from their locally available pinned images. It does not rebuild or pull images and leaves other services running.\n\n`make run` / `make run-bg` regenerate `litellm/config.yaml` from fragments (only enabled providers + filtered fallback chains) and pre-flight-validate any file-path env vars (e.g. `MAILBOX_CONFIG`, `CLOUDFLARED_CONFIG`) actually exist before starting containers.\n\nOther Makefile targets: `make down`, `make restart`, `make logs`, `make build-config` (regenerate litellm config only), `make test` (stack must already be running).\n\n## Ports\n\nSingle exposed port: **`4000`** (nginx), hardcoded — not env-configurable. Everything else (Postgres, Redis, LiteLLM, and most optional services) binds to no host port at all; they're reached only through nginx's path-based routing on `:4000`. Do not add host port mappings for internal services unless you specifically need direct access.\n\n- Gateway (OpenAI-compatible): `http://localhost:4000`\n- LiteLLM admin UI: `http://localhost:4000/ui/`\n- LibreChat (if `LIBRECHAT=1`): `http://localhost:4000/librechat/`\n- SearXNG (if `SEARXNG=1`): `http://localhost:4000/searxng/`\n- Async queue (proxq): `http://localhost:4000/q/`\n- Direct-routed services (not via LiteLLM): `/predictalot/`, `/predictalot-cuda/`, `/decidealot/`, `/decidealot-cuda/`, `/audiolla/`, `/audiolla-cuda/`, `/flickies/`, `/flickies-cuda/`, `/mailbox/`, `/telethon/`, `/piston/`, `/storage/` (hybrids3), `/claudebox/`, `/pibox-zai/`, `/pibox/`, `/stealthy-auto-browse/`\n\n## Required env / keys\n\nCore, always needed regardless of which optional services you enable:\n\n| Variable | Purpose |\n| --- | --- |\n| `AIGATE_TOKEN` | Master bearer token. Every per-service token below defaults to this value when left unset — one token authenticates against LiteLLM, claudebox, pibox-zai, predictalot, decidealot, mcp_tools, stealthy-auto-browse, hybrids3, telethon, audiolla, flickies, talkies, talkies-cuda. Override a specific `*_AUTH_TOKEN` / `*_API_TOKEN` var to scope that service separately. |\n| `LITELLM_MASTER_KEY` | Optional override; defaults to `AIGATE_TOKEN` when unset. |\n| `POSTGRES_DB` / `POSTGRES_USER` / `POSTGRES_P"},{"path":"skill-card.md","content":"## Description:\n\nGuides developers in deploying and using a self-hosted, OpenAI-compatible gateway for model routing and optional AI tools.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[psyb0t](https://clawhub.ai/user/psyb0t)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and engineers use this skill to set up a Docker-based AI gateway and connect clients to model providers and optional tools through one endpoint.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: One bearer token may grant access to code execution, browser automation, messaging, and storage across enabled services.\n\nMitigation: Enable only necessary services, split per-service tokens, and grant agent access only for explicitly authorized tasks.\n\nRisk: Direct internet exposure of the gateway could expose high-privilege capabilities.\n\nMitigation: Install only on a trusted host and avoid exposing port 4000 directly to the internet; use a protected access path.\n\nRisk: Gateway, mailbox, and Telegram configuration may contain sensitive credentials.\n\nMitigation: Keep AIGATE_TOKEN and mailbox/Telethon configuration private and out of committed files.\n\n## Reference(s):\n\n- [aigate ClawHub release](https://clawhub.ai/psyb0t/skills/aigate)\n- [aigate setup guide](references/setup.md)\n\n## Skill Output:\n\n**Output Type(s):** [Guidance, Shell commands, Configuration instructions]\n\n**Output Format:** [Markdown with shell and configuration examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Deployment and API usage depend on enabled services and operator-provided credentials.]\n\n## Skill Version(s):\n\n10.0.0 (source: ClawHub release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":1662,"uniquenessScore":42,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T15:17:18.986Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T15:17:18.986Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T06:43:08.340Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}