{"id":"909fd3d8-e96b-47b9-9fb2-cf6a6e214202","entityType":"agent","slug":"clawhub-raydatalab-hermes-smart-router","name":"Smart Router Publish","canonicalUrl":"https://www.xpersona.co/agent/clawhub-raydatalab-hermes-smart-router","canonicalPath":"/agent/clawhub-raydatalab-hermes-smart-router","generatedAt":"2026-10-10T11:51:25.340Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-10T09:02:10.488Z","emptyReason":null},"description":"Use when switching models, saving costs, or routing queries. Automatically picks the cheapest model that can handle the job — \"translate hello\" routes to $0/... Skill: Smart Router Publish Owner: raydatalab Summary: Use when switching models, saving costs, or routing queries. Automatically picks the cheapest model that can handle the job — \"translate hello\" routes to $0/... Tags: latest:0.2.2 Version history: v0.2.2 | 2026-07-11T01:35:25.520Z | auto - Updated version to 0.2.2 - Minor formatting and clarity improvements in SKILL.md - No functional or code changes; documentati","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.5K downloads reported by the source. Last updated 10/10/2026.","installCommand":"clawhub skill install s179gmr84mj3szjyrksyq2zv058a61mt:hermes-smart-router","sourceUrl":"https://clawhub.ai/raydatalab/hermes-smart-router","homepage":"https://clawhub.ai/raydatalab/skills/hermes-smart-router","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/raydatalab/hermes-smart-router","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/raydatalab/skills/hermes-smart-router","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":64,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Use when switching models, saving costs, or routing queries. Automatically picks the cheapest model that can handle the job — \"translate hello\" routes to $0/..."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-10T09:02:10.488Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T09:02:10.488Z","emptyReason":null},"stars":null,"forks":null,"downloads":1540,"packageName":null,"latestVersion":"0.2.2","tractionLabel":"1.5K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T09:02:10.488Z","emptyReason":null},"lastUpdatedAt":"2026-10-10T09:02:10.488Z","lastCrawledAt":"2026-10-10T09:02:10.488Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-11T09:02:10.488Z","lastVerifiedAt":null,"highlights":[{"version":"0.2.2","createdAt":"2026-07-11T01:35:25.520Z","changelog":"- Updated version to 0.2.2 - Minor formatting and clarity improvements in SKILL.md - No functional or code changes; documentation only","fileCount":3,"zipByteSize":4340},{"version":"0.2.1","createdAt":"2026-07-11T01:29:37.492Z","changelog":"hermes-smart-router v0.2.1 - Simplified and clarified documentation for easier onboarding and usage. - Added pricing examples and streamlined tier descriptions to illustrate cost savings. - Moved agent instructions and routing logic to the top; emphasized critical usage details. - Updated configuration and command reference sections for accuracy. - Removed legacy `skill-card.md` documentation file.","fileCount":3,"zipByteSize":4231},{"version":"0.2.0","createdAt":"2026-07-10T23:44:18.214Z","changelog":"- Major update: clarifies routing requirements and integrates ready-to-use model switch recommendations. - The `router.resolve()` call is now mandatory for all non-trivial queries; strict routing enforcement. - Introduces the `recommendation` field—paste directly at the top of your response if present. - Expanded and clarified when to route vs. skip; fast-path for short or trivial queries emphasized. - Documentation refined for clarity; instructions and rationale are now more concise and user-focused. - Removed the redundant skill-card.md file.","fileCount":3,"zipByteSize":4901},{"version":"0.1.5","createdAt":"2026-07-10T19:17:00.398Z","changelog":"- Major instructions and documentation update for clarity and usability. - Added detailed guidance on when and how to invoke the router, with actionable routing heuristics. - Expanded and clarified example code for using the router and interpreting results. - Included a comprehensive triggers list to enable tier/model routing based on varied commands and user intent, including non-English phrases. - Noted fast-path optimization: queries under 20 characters skip embedding and return instantly. - Removed deprecated skill-card.md file.","fileCount":3,"zipByteSize":4947},{"version":"0.1.4","createdAt":"2026-07-09T09:32:11.699Z","changelog":"No code or documentation changes detected in this release. - No changes found between versions.","fileCount":3,"zipByteSize":4243},{"version":"0.1.3","createdAt":"2026-07-09T09:19:20.866Z","changelog":"- Refined agent instructions to clarify when to classify queries, emphasizing selective use rather than classifying every message. - Updated usage guidance to recommend the get_router singleton and explain the meaning of the needs_switch flag. - Added explicit downgrading and upgrading recommendations, including sample user-facing hint lines. - Clarified that agents cannot execute /model themselves and should only prepend recommendations for model switching. - Removed skill-card.md file.","fileCount":3,"zipByteSize":4229},{"version":"0.1.2","createdAt":"2026-07-09T08:33:28.719Z","changelog":"- The Agent Instructions in SKILL.md have been clarified and simplified: instructional steps now focus on classification and model-switching, with concise Python usage. - Added explicit guidance on fallback behavior and response hints if model switching is not possible. - Removed a section with stepwise Python examples and error-handling details, consolidating the advice for brevity. - skill-card.md has been removed. No changes to core functionality.","fileCount":3,"zipByteSize":3767},{"version":"0.1.1","createdAt":"2026-07-09T07:52:14.920Z","changelog":"- Updated documentation in SKILL.md: installation steps clarified; pip dependencies no longer auto-installed—manual setup required. - Testing instructions changed to use a shell script (`bash scripts/install.sh`) and explicit virtual environment paths. - Removed first-run agent bootstrap section; consolidated guidance under \"Testing\" and \"Agent Instructions\". - skill-card.md file removed.","fileCount":3,"zipByteSize":3770}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s179gmr84mj3szjyrksyq2zv058a61mt:hermes-smart-router","setupComplexity":"low","setupSteps":["Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-raydatalab-hermes-smart-router/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-raydatalab-hermes-smart-router/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-raydatalab-hermes-smart-router/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-raydatalab-hermes-smart-router/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-raydatalab-hermes-smart-router/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-raydatalab-hermes-smart-router/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T11:51:25.337Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-raydatalab-hermes-smart-router/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-raydatalab-hermes-smart-router/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-raydatalab-hermes-smart-router/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-raydatalab-hermes-smart-router/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-10T09:02:10.488Z","emptyReason":null},"readme":"Skill: Smart Router Publish\n\nOwner: raydatalab\n\nSummary: Use when switching models, saving costs, or routing queries. Automatically picks the cheapest model that can handle the job — \"translate hello\" routes to $0/...\n\nTags: latest:0.2.2\n\nVersion history:\n\nv0.2.2 | 2026-07-11T01:35:25.520Z | auto\n\n- Updated version to 0.2.2\n- Minor formatting and clarity improvements in SKILL.md\n- No functional or code changes; documentation only\n\nv0.2.1 | 2026-07-11T01:29:37.492Z | auto\n\nhermes-smart-router v0.2.1\n\n- Simplified and clarified documentation for easier onboarding and usage.\n- Added pricing examples and streamlined tier descriptions to illustrate cost savings.\n- Moved agent instructions and routing logic to the top; emphasized critical usage details.\n- Updated configuration and command reference sections for accuracy.\n- Removed legacy `skill-card.md` documentation file.\n\nv0.2.0 | 2026-07-10T23:44:18.214Z | auto\n\n- Major update: clarifies routing requirements and integrates ready-to-use model switch recommendations.\n- The `router.resolve()` call is now mandatory for all non-trivial queries; strict routing enforcement.\n- Introduces the `recommendation` field—paste directly at the top of your response if present.\n- Expanded and clarified when to route vs. skip; fast-path for short or trivial queries emphasized.\n- Documentation refined for clarity; instructions and rationale are now more concise and user-focused.\n- Removed the redundant skill-card.md file.\n\nv0.1.5 | 2026-07-10T19:17:00.398Z | auto\n\n- Major instructions and documentation update for clarity and usability.\n- Added detailed guidance on when and how to invoke the router, with actionable routing heuristics.\n- Expanded and clarified example code for using the router and interpreting results.\n- Included a comprehensive triggers list to enable tier/model routing based on varied commands and user intent, including non-English phrases.\n- Noted fast-path optimization: queries under 20 characters skip embedding and return instantly.\n- Removed deprecated skill-card.md file.\n\nv0.1.4 | 2026-07-09T09:32:11.699Z | auto\n\nNo code or documentation changes detected in this release.\n\n- No changes found between versions.\n\nv0.1.3 | 2026-07-09T09:19:20.866Z | auto\n\n- Refined agent instructions to clarify when to classify queries, emphasizing selective use rather than classifying every message.\n- Updated usage guidance to recommend the get_router singleton and explain the meaning of the needs_switch flag.\n- Added explicit downgrading and upgrading recommendations, including sample user-facing hint lines.\n- Clarified that agents cannot execute /model themselves and should only prepend recommendations for model switching.\n- Removed skill-card.md file.\n\nv0.1.2 | 2026-07-09T08:33:28.719Z | auto\n\n- The Agent Instructions in SKILL.md have been clarified and simplified: instructional steps now focus on classification and model-switching, with concise Python usage.\n- Added explicit guidance on fallback behavior and response hints if model switching is not possible.\n- Removed a section with stepwise Python examples and error-handling details, consolidating the advice for brevity.\n- skill-card.md has been removed. No changes to core functionality.\n\nv0.1.1 | 2026-07-09T07:52:14.920Z | auto\n\n- Updated documentation in SKILL.md: installation steps clarified; pip dependencies no longer auto-installed—manual setup required.\n- Testing instructions changed to use a shell script (`bash scripts/install.sh`) and explicit virtual environment paths.\n- Removed first-run agent bootstrap section; consolidated guidance under \"Testing\" and \"Agent Instructions\".\n- skill-card.md file removed.\n\nv0.1.0 | 2026-07-09T06:19:24.986Z | auto\n\nInitial release of hermes-smart-router\n\n- Enables intelligent model tier routing for Hermes, auto-selecting between local, flash, or pro models per query.\n- Uses local Ollama embeddings for on-device query classification; requires no external API for routing decisions.\n- Automatically manages Ollama lifecycle, starting/stopping as needed, with idle timeout support.\n- Flexible configuration via hermes config or YAML, supporting custom providers and models for each tier.\n- Provides slash commands for routing diagnostics and Ollama management.\n- Ensures safe operation with automatic fallback to flash tier and robust error handling.\n\nArchive index:\n\nArchive v0.2.2: 3 files, 4340 bytes\n\nFiles: skill-card.md (2337b), SKILL.md (5629b), _meta.json (138b)\n\nFile v0.2.2:SKILL.md\n\n---\nname: hermes-smart-router\ndescription: Use when switching models, saving costs, or routing queries. Automatically picks the cheapest model that can handle the job — \"translate hello\" routes to $0/M local, \"design a database\" routes to $3/M pro. 100% local classification, zero API calls for routing.\nversion: 0.2.2\nauthor: raydatalab\nlicense: MIT\nplatforms: [linux, macos, wsl]\ntriggers:\n  - model switch\n  - change model\n  - /model\n  - /tier\n  - switch to\n  - use local\n  - use flash\n  - use pro\n  - too slow\n  - wasting tokens\n  - cheaper model\n  - downgrade\n  - upgrade model\n  - routing\n  - tier selection\n  - local model\n  - what model\n  - which model\n  - speed up\n  - save cost\n  - 切换模型\n  - 太慢了\n  - 换模型\n  - 用什么模型\nmetadata:\n  hermes:\n    tags: [routing, model, cost-optimization, ollama]\n    homepage: https://github.com/raydatalab/hermes-smart-router\n    related_skills: [hermes-cost-optimization]\n---\n\n# Smart Router\n\n## Overview\n\n```\n\"Translate hello to German\"       → local   $0/M tok\n\"Explain how DNS works\"           → flash   $0.15/M tok  (GPT-4o-mini)\n\"Design a distributed database\"   → pro     $3/M tok  (Claude Sonnet)\n```\n\nSmart Router classifies every query locally (Ollama embeddings + semantic-router)\nand picks the cheapest tier that can handle it. Simple queries route to free local\nmodels. Complex reasoning routes to frontier models. You pay for what you need —\nnot what your config happens to be set to.\n\nClassification runs entirely on your machine. Zero API calls for routing decisions.\nZero keys needed. ~3s on first call, <1ms after.\n\n## When to Use\n\nLoad this skill when:\n- The user is toggling between models manually (wasting tokens on simple queries)\n- The user asks about cost saving, model switching, or routing\n- The user mentions `/model`, `/tier`, or switches models mid-session\n- The agent is about to answer a complex question and should verify the current tier\n\nDo NOT load for: greetings, one-liners, translations, queries under 20 characters.\n\n## CRITICAL — Agent Instructions\n\n**Call `router.resolve()` before answering any non-trivial query.**\n\n```python\nfrom smart_router.router import get_router\n\nrouter = get_router()\ndecision = router.resolve(user_message, current_tier=ACTIVE_TIER)\n# → {\"tier\": \"pro\", \"model\": {...}, \"needs_switch\": true,\n#    \"reason\": \"Upgrade from flash to pro — Complex code...\",\n#    \"recommendation\": \"💡 Switch to pro: `/model anthropic claude-sonnet-4` — ...\"}\n```\n\n**If `decision[\"recommendation\"]` is not None**, paste it at the top of your\nresponse — it's ready to use, no formatting needed.\n\n**If `decision[\"recommendation\"]` is None**, just answer normally.\n**If the call errors or times out**, skip it and answer — never retry.\n\n`ACTIVE_TIER` is your current tier (local/flash/pro). Check with `/tier` or ask.\n\n### When to Skip\n\nSkip routing for: greetings (\"hello\", \"thanks\"), translations, one-liner facts,\nqueries under 20 characters. Everything else — route.\n\n### When to Route\n\n- **Complex:** 3+ tool calls, reasoning, architecture, multi-step code, debugging → route\n- **Unsure:** If it takes more than a sentence to answer → route\n- **Default:** If in doubt, route. Routing cost (~3s first call, <1ms after) is always\n  less than answering from the wrong tier.\n\n## Tier Reference\n\n| Tier | Example Model | Input Price/1M tok | Use When |\n|------|-------------|-------------------|----------|\n| `local` | Ollama (llama3, qwen, etc.) | $0 | Simple Q&A, translations, offline |\n| `flash` | GPT-4o-mini, Gemini Flash, Claude Haiku | ~$0.15–0.80 | General knowledge, casual coding |\n| `pro` | Claude Sonnet, GPT-4o, Gemini Pro | ~$2.50–3 | Complex reasoning, architecture |\n\nPricing per official API pages (OpenAI, Anthropic, Google). See individual provider\ndocs for exact rates. Per-token pricing means a single 1K-token question costs\n$0.003 on flash vs $0.003 on pro — but over thousands of queries per month, the\ndifference compounds.\n\n## Prerequisites\n\n- Hermes Agent v0.17+\n- Ollama installed\n- `semantic-router[ollama]` and `smart_router`\n\n## Configuration\n\n```yaml\nsmart_router:\n  enabled: true\n  default_tier: flash\n  encoder_model: nomic-embed-text\n  tiers:\n    local:\n      provider: custom\n      model: llama3.2:3b\n      base_url: http://localhost:11434/v1\n    flash:\n      provider: openai\n      model: gpt-4o-mini\n    pro:\n      provider: anthropic\n      model: claude-sonnet-4\n  ollama:\n    auto_start: true\n    idle_timeout: 300\n```\n\n## Slash Commands\n\n| Command | Description |\n|---------|-------------|\n| `/route <query>` | Show tier selection (dry run) |\n| `/route-stats` | Session routing statistics |\n| `/ollama start / stop / status` | Ollama lifecycle |\n| `/tier` | Show current tier and model |\n\n## Common Pitfalls\n\n1. **Agent forgets to call `router.resolve()`.** The most common failure mode.\n   If the agent answers without routing, manually trigger with `/route <query>`.\n2. **Ollama not running.** If `decision[\"ollama_ready\"]` is false, start Ollama\n   first (`/ollama start`) or skip routing for this query.\n3. **Slow first call.** First `router.resolve()` pulls `nomic-embed-text` (~274MB).\n   Subsequent calls are instant. Warm up with `python3 -m smart_router route \"test\"`\n   before heavy sessions.\n4. **Fast-path false negatives.** Queries under 20 chars skip embedding. If a short\n   query needs pro-level reasoning, the router won't catch it — use `/model` manually.\n\n## Testing\n\n```bash\nbash scripts/install.sh\npython3 -m smart_router route \"What is the capital of France?\"\npython3 -m smart_router chat\npython3 -m pytest tests/\n```\n\nFile v0.2.2:_meta.json\n\n{\n  \"ownerId\": \"kn7fk1ctfxrjyw4y1hdms1re458a736q\",\n  \"slug\": \"hermes-smart-router\",\n  \"version\": \"0.2.2\",\n  \"publishedAt\": 1783733725520\n}\n\nFile v0.2.2:skill-card.md\n\n## Description:\n\nUse when switching models, saving costs, or routing queries; it classifies requests locally and recommends the cheapest model tier that can handle the task.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[raydatalab](https://clawhub.ai/user/raydatalab)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and agent operators use this skill to decide when to keep work on local or lower-cost models and when to recommend a higher-capability tier for complex reasoning, architecture, debugging, or multi-step coding tasks.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Broad routing-related phrases may activate the skill in model-selection or cost-saving conversations where routing is not desired.\n\nMitigation: Review routing recommendations before switching tiers and skip routing for greetings, simple translations, one-line factual requests, and very short prompts.\n\nRisk: Using the router backend may require a local Ollama service or model download before routing works.\n\nMitigation: Confirm Ollama and required local models are installed before relying on routing; if a router call fails or times out, continue without retrying.\n\nRisk: Short prompts can bypass routing even when they need a stronger model tier.\n\nMitigation: Use manual tier or model selection for short prompts that still require complex reasoning or advanced coding support.\n\n## Reference(s):\n\n- [ClawHub Skill Page](https://clawhub.ai/raydatalab/skills/hermes-smart-router)\n- [Hermes Smart Router Homepage](https://github.com/raydatalab/hermes-smart-router)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown guidance with inline commands and configuration snippets]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Produces routing recommendations and setup guidance; no API keys are required for local routing classification.]\n\n## Skill Version(s):\n\n0.2.2 (source: release metadata and SKILL.md frontmatter)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v0.2.1: 3 files, 4231 bytes\n\nFiles: skill-card.md (2089b), SKILL.md (5635b), _meta.json (138b)\n\nFile v0.2.1:SKILL.md\n\n---\nname: hermes-smart-router\ndescription: Use when switching models, saving costs, or routing queries. Automatically picks the cheapest model that can handle the job — \"translate hello\" routes to $0/M local, \"design a database\" routes to $3/M pro. 100% local classification, zero API calls for routing.\nversion: 0.2.1\nauthor: raydatalab\nlicense: MIT\nplatforms: [linux, macos, wsl]\ntriggers:\n  - model switch\n  - change model\n  - /model\n  - /tier\n  - switch to\n  - use local\n  - use flash\n  - use pro\n  - too slow\n  - wasting tokens\n  - cheaper model\n  - downgrade\n  - upgrade model\n  - routing\n  - tier selection\n  - local model\n  - what model\n  - which model\n  - speed up\n  - save cost\n  - 切换模型\n  - 太慢了\n  - 换模型\n  - 用什么模型\nmetadata:\n  hermes:\n    tags: [routing, model, cost-optimization, ollama]\n    homepage: https://github.com/raydatalab/hermes-smart-router\n    related_skills: [hermes-cost-optimization]\n---\n\n# Smart Router\n\n## Overview\n\n```\n\"Translate hello to German\"       → local   $0/M tok\n\"Explain how DNS works\"           → flash   $0.15/M tok (如 GPT-4o-mini)\n\"Design a distributed database\"   → pro     $3/M tok (如 Claude Sonnet)\n```\n\nSmart Router classifies every query locally (Ollama embeddings + semantic-router)\nand picks the cheapest tier that can handle it. Simple queries route to free local\nmodels. Complex reasoning routes to frontier models. You pay for what you need —\nnot what your config happens to be set to.\n\nClassification runs entirely on your machine. Zero API calls for routing decisions.\nZero keys needed. ~3s on first call, <1ms after.\n\n## When to Use\n\nLoad this skill when:\n- The user is toggling between models manually (wasting tokens on simple queries)\n- The user asks about cost saving, model switching, or routing\n- The user mentions `/model`, `/tier`, or switches models mid-session\n- The agent is about to answer a complex question and should verify the current tier\n\nDo NOT load for: greetings, one-liners, translations, queries under 20 characters.\n\n## CRITICAL — Agent Instructions\n\n**Call `router.resolve()` before answering any non-trivial query.**\n\n```python\nfrom smart_router.router import get_router\n\nrouter = get_router()\ndecision = router.resolve(user_message, current_tier=ACTIVE_TIER)\n# → {\"tier\": \"pro\", \"model\": {...}, \"needs_switch\": true,\n#    \"reason\": \"Upgrade from flash to pro — Complex code...\",\n#    \"recommendation\": \"💡 Switch to pro: `/model anthropic claude-sonnet-4` — ...\"}\n```\n\n**If `decision[\"recommendation\"]` is not None**, paste it at the top of your\nresponse — it's ready to use, no formatting needed.\n\n**If `decision[\"recommendation\"]` is None**, just answer normally.\n**If the call errors or times out**, skip it and answer — never retry.\n\n`ACTIVE_TIER` is your current tier (local/flash/pro). Check with `/tier` or ask.\n\n### When to Skip\n\nSkip routing for: greetings (\"hello\", \"thanks\"), translations, one-liner facts,\nqueries under 20 characters. Everything else — route.\n\n### When to Route\n\n- **Complex:** 3+ tool calls, reasoning, architecture, multi-step code, debugging → route\n- **Unsure:** If it takes more than a sentence to answer → route\n- **Default:** If in doubt, route. Routing cost (~3s first call, <1ms after) is always\n  less than answering from the wrong tier.\n\n## Tier Reference\n\n| Tier | Example Model | Input Price/1M tok | Use When |\n|------|-------------|-------------------|----------|\n| `local` | Ollama (llama3, qwen, etc.) | $0 | Simple Q&A, translations, offline |\n| `flash` | GPT-4o-mini, Gemini Flash, Claude Haiku | ~$0.15–0.80 | General knowledge, casual coding |\n| `pro` | Claude Sonnet, GPT-4o, Gemini Pro | ~$2.50–3 | Complex reasoning, architecture |\n\nPricing per official API pages (OpenAI, Anthropic, Google). See individual provider\ndocs for exact rates. Per-token pricing means a single 1K-token question costs\n$0.003 on flash vs $0.003 on pro — but over thousands of queries per month, the\ndifference compounds.\n\n## Prerequisites\n\n- Hermes Agent v0.17+\n- Ollama installed\n- `semantic-router[ollama]` and `smart_router`\n\n## Configuration\n\n```yaml\nsmart_router:\n  enabled: true\n  default_tier: flash\n  encoder_model: nomic-embed-text\n  tiers:\n    local:\n      provider: custom\n      model: llama3.2:3b\n      base_url: http://localhost:11434/v1\n    flash:\n      provider: openai\n      model: gpt-4o-mini\n    pro:\n      provider: anthropic\n      model: claude-sonnet-4\n  ollama:\n    auto_start: true\n    idle_timeout: 300\n```\n\n## Slash Commands\n\n| Command | Description |\n|---------|-------------|\n| `/route <query>` | Show tier selection (dry run) |\n| `/route-stats` | Session routing statistics |\n| `/ollama start / stop / status` | Ollama lifecycle |\n| `/tier` | Show current tier and model |\n\n## Common Pitfalls\n\n1. **Agent forgets to call `router.resolve()`.** The most common failure mode.\n   If the agent answers without routing, manually trigger with `/route <query>`.\n2. **Ollama not running.** If `decision[\"ollama_ready\"]` is false, start Ollama\n   first (`/ollama start`) or skip routing for this query.\n3. **Slow first call.** First `router.resolve()` pulls `nomic-embed-text` (~274MB).\n   Subsequent calls are instant. Warm up with `python3 -m smart_router route \"test\"`\n   before heavy sessions.\n4. **Fast-path false negatives.** Queries under 20 chars skip embedding. If a short\n   query needs pro-level reasoning, the router won't catch it — use `/model` manually.\n\n## Testing\n\n```bash\nbash scripts/install.sh\npython3 -m smart_router route \"What is the capital of France?\"\npython3 -m smart_router chat\npython3 -m pytest tests/\n```\n\nFile v0.2.1:_meta.json\n\n{\n  \"ownerId\": \"kn7fk1ctfxrjyw4y1hdms1re458a736q\",\n  \"slug\": \"hermes-smart-router\",\n  \"version\": \"0.2.1\",\n  \"publishedAt\": 1783733377492\n}\n\nFile v0.2.1:skill-card.md\n\n## Description: <br>\nSmart Router Publish helps an agent choose a local, flash, or pro model tier for cost-aware routing before answering non-trivial prompts. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[raydatalab](https://clawhub.ai/user/raydatalab) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and agent operators use this skill to reduce model spend by routing simple prompts to local or lower-cost models and reserving pro-tier models for complex reasoning, architecture, debugging, and multi-step code work. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Non-trivial prompts may be sent to a local routing service before the agent answers. <br>\nMitigation: Review local Ollama and smart_router logging, retention, and access settings before using the skill with sensitive prompts. <br>\nRisk: Broad routing triggers may activate the skill during ordinary model, cost, or speed discussions. <br>\nMitigation: Use the documented skip cases for greetings, one-liners, translations, and prompts under 20 characters, and confirm routing intent when the trigger context is ambiguous. <br>\n\n\n## Reference(s): <br>\n- [Hermes Smart Router repository](https://github.com/raydatalab/hermes-smart-router) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [Guidance, Shell commands, Configuration] <br>\n**Output Format:** [Markdown guidance with inline code, shell commands, and YAML configuration] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Produces routing recommendations for an agent; no recommendation is emitted when the current model tier is acceptable.] <br>\n\n## Skill Version(s): <br>\n0.2.1 (source: release metadata and SKILL.md frontmatter) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v0.2.0: 3 files, 4901 bytes\n\nFiles: skill-card.md (2446b), SKILL.md (7067b), _meta.json (138b)\n\nFile v0.2.0:SKILL.md\n\n---\nname: hermes-smart-router\ndescription: Smart model-tier routing for Hermes — save 70%+ on API costs by auto-switching to cheaper models for simple queries. 100% local Ollama classification, zero API calls for routing.\nversion: 0.2.0\nauthor: raydatalab\nlicense: MIT\nplatforms: [linux, macos, wsl]\ntriggers:\n  - model switch\n  - change model\n  - /model\n  - /tier\n  - switch to\n  - use local\n  - use flash\n  - use pro\n  - too slow\n  - wasting tokens\n  - cheaper model\n  - downgrade\n  - upgrade model\n  - routing\n  - tier selection\n  - local model\n  - what model\n  - which model\n  - speed up\n  - save cost\n  - 切换模型\n  - 太慢了\n  - 换模型\n  - 用什么模型\nmetadata:\n  hermes:\n    tags: [routing, model, cost-optimization, ollama]\n    homepage: https://github.com/raydatalab/hermes-smart-router\n    related_skills: [hermes-cost-optimization]\n---\n\n# Smart Router\n\nIntelligent model tier routing for Hermes Agent — recommends the right model tier based on query complexity.\n\n## Why Use Me\n\nEvery query you send to the wrong model tier wastes tokens (pro for a greeting)\nor produces weak output (local for architecture). Smart Router catches these\nmismatches automatically:\n\n- **Save money** — downgrade from pro/flash to local on simple queries\n- **Get quality** — upgrade from local/flash to pro on complex reasoning\n- **100% local, free, private** — all classification runs on your machine via\n  Ollama embeddings. Zero API calls for routing. Zero setup after install.\n\nThe router is a bonus — when called, it saves tokens and improves quality.\nWhen skipped, you never notice. There is no penalty for not using it.\n\n## CRITICAL — Read Before Responding\n\n**You MUST call `router.resolve()` before answering any non-trivial query.**\nThe user installed this skill to save money on API costs. Every response you\nsend from the wrong tier is wasted tokens. This is not optional — it's the\nentire reason this skill exists.\n\nThe call is cheap: first call takes ~3s, subsequent calls return in microseconds.\nShort queries and greetings skip the encoder entirely (fast-path, 0 cost).\n\n```python\nfrom smart_router.router import get_router\n\nrouter = get_router()\ndecision = router.resolve(user_message, current_tier=ACTIVE_TIER)\n# → {\"tier\": \"pro\", \"model\": {...}, \"needs_switch\": true,\n#    \"reason\": \"Upgrade from flash to pro — Complex code...\",\n#    \"recommendation\": \"💡 Switch to pro: `/model anthropic claude-sonnet-4` — ...\"}\n```\n\n**If `decision[\"recommendation\"]` is not None**, paste it at the top of your\nresponse — it's ready to use, no formatting needed:\n\n```\n{decision[\"recommendation\"]}\n\n[your actual response here]\n```\n\n**If `decision[\"recommendation\"]` is None**, just answer normally.\n**If the call errors or times out**, skip it and answer — never retry.\n\n**ACTIVE_TIER** is your current tier (local/flash/pro). Check with `/tier`\nor ask the user if you don't know.\n\n### When to Skip\n\nOnly skip routing for: greetings (\"hello\", \"thanks\"), translations,\none-liner facts, queries under 20 characters. Everything else — route.\n\n### When to Route (Two Rules)\n\n**Rule 1:** Route before any response that needs 3+ tool calls, complex\nreasoning, architecture, multi-step code, debugging, or when you're unsure\nabout the current tier.\n\n**Rule 2:** For everything else, route anyway if it takes more than a sentence\nto answer. The cost of routing (~3s on first call, <1ms after) is always\nlower than the cost of answering from the wrong tier.\n\nIf in doubt, route.\n\n## How to Act on the Result\n\nYou **cannot execute `/model` yourself** — it's a user-side slash command.\n\n- **`recommendation` is a string** → paste it verbatim at the top of your reply.\n  The string already includes the emoji, the `/model` command, and the reason.\n- **`recommendation` is `None`** → say nothing about routing, just answer.\n- For local tier: check `decision[\"ollama_ready\"]` first — if `false`,\n  mention that Ollama isn't ready before recommending local.\n- **Error / timeout** → skip, answer with current model. Do not retry.\n\n## Tier Reference\n\n| Tier | Typical Model | Use When |\n|------|--------------|----------|\n| `local` | Your Ollama model | Simple Q&A, translations, offline, private |\n| `flash` | Affordable API (e.g. Gemini Flash) | General knowledge, casual coding (default) |\n| `pro` | Frontier API (e.g. Claude Sonnet) | Complex reasoning, architecture, multi-step code |\n\n### How It Works\n\n1. User sends a query\n2. Smart Router classifies it using local Ollama embeddings (no API calls, no API keys)\n3. Routes to the appropriate tier — auto-starts Ollama for local, manages idle timeout\n4. Agent responds using the selected model\n\n### Ollama Lifecycle\n\n| Action | Method | Behavior |\n|--------|--------|----------|\n| Start | `ollama_manager.ensure_running()` | Starts `ollama serve`, waits for port, pulls model if missing |\n| Check | `ollama_manager.is_running` | Checks via `ollama ps` → systemd → pgrep |\n| Idle | `ollama_manager.idle_seconds` | Seconds since last local-tier use |\n| Kill | `ollama_manager.ensure_killed()` | SIGTERM (SIGKILL if forced) — skips systemd-managed |\n\n## Prerequisites\n\n- Hermes Agent v0.17+\n- Ollama installed (the skill auto-detects models from `ollama list`; embedding model is pulled on first use)\n- Python packages (install once, see Testing section below):\n  - `semantic-router[ollama]`\n  - `smart_router` (from this repo)\n- For flash/pro tiers: configure providers via `hermes model`\n\n## Configuration\n\nConfigure via `hermes config set` for each tier or add this block under `smart_router:` in config.yaml:\n\n```yaml\nsmart_router:\n  enabled: true\n  default_tier: flash\n  encoder_model: nomic-embed-text\n  tiers:\n    local:\n      provider: custom\n      model: llama3.2:3b\n      base_url: http://localhost:11434/v1\n    flash:\n      provider: openrouter\n      model: google/gemini-flash-1.5\n    pro:\n      provider: anthropic\n      model: claude-sonnet-4\n  ollama:\n    auto_start: true\n    idle_timeout: 300\n```\n\n## Slash Commands\n\n| Command | Description |\n|---------|-------------|\n| `/route <query>` | Show which tier would be selected (dry run) |\n| `/route-stats` | Show session routing statistics |\n| `/ollama start` | Manually start Ollama |\n| `/ollama stop` | Manually stop Ollama |\n| `/ollama status` | Check Ollama process status |\n| `/tier` | Show current active tier and model |\n\n## Testing\n\n```bash\n# Install dependencies (one-time setup)\nbash scripts/install.sh\n\n# Test routing\n~/.hermes/hermes-agent/venv/bin/python3 -m smart_router route \"What is the capital of France?\"\n~/.hermes/hermes-agent/venv/bin/python3 -m smart_router chat\n~/.hermes/hermes-agent/venv/bin/python3 -m pytest tests/\n```\n\n## Notes\n\n- First use pulls `nomic-embed-text` via Ollama (~274MB, cached).\n- All classification is local — zero API calls for routing.\n- Skill auto-detects your Ollama models from `ollama list`.\n- Model switches are session-scoped — your config.yaml is not modified.\n- Fast-path: queries under 20 characters skip embedding and return the default tier instantly.\n\nFile v0.2.0:_meta.json\n\n{\n  \"ownerId\": \"kn7fk1ctfxrjyw4y1hdms1re458a736q\",\n  \"slug\": \"hermes-smart-router\",\n  \"version\": \"0.2.0\",\n  \"publishedAt\": 1783727058214\n}\n\nFile v0.2.0:skill-card.md\n\n## Description: <br>\nSmart model-tier routing for Hermes that recommends cheaper local, flash, or pro model tiers based on query complexity, using local Ollama classification with no routing API calls. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[raydatalab](https://clawhub.ai/user/raydatalab) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and Hermes Agent users use this skill to receive model-tier recommendations before substantive replies, helping route simple work to cheaper local or flash models and complex work to pro models. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill strongly changes assistant behavior by asking the agent to consult routing before most substantive replies. <br>\nMitigation: Review whether broad routing is desired before deployment and disable or narrow the smart_router configuration if routine routing is not appropriate. <br>\nRisk: First use may download an Ollama embedding model and the workflow may start or stop local Ollama processes. <br>\nMitigation: Check the auto_start and idle_timeout settings before use, and run Ollama manually in environments where process lifecycle control must remain explicit. <br>\nRisk: Model switch recommendations may affect cost, latency, and answer quality. <br>\nMitigation: Treat recommended slash commands as user-approved actions and confirm the target provider or tier before switching. <br>\n\n\n## Reference(s): <br>\n- [ClawHub Skill Page](https://clawhub.ai/raydatalab/skills/hermes-smart-router) <br>\n- [Project Homepage](https://github.com/raydatalab/hermes-smart-router) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [Guidance, Shell commands, Configuration] <br>\n**Output Format:** [Markdown text with inline slash commands, code snippets, and YAML configuration examples] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [May recommend a user-run model switch command; does not execute the slash command itself.] <br>\n\n## Skill Version(s): <br>\n0.2.0 (source: server release metadata and SKILL.md frontmatter) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v0.1.5: 3 files, 4947 bytes\n\nFiles: skill-card.md (2351b), SKILL.md (7737b), _meta.json (138b)\n\nFile v0.1.5:SKILL.md\n\n---\nname: hermes-smart-router\ndescription: Intelligent model tier routing for Hermes — auto-selects local, flash, or pro per query\nversion: 0.1.5\nauthor: raydatalab\nlicense: MIT\nplatforms: [linux, macos, wsl]\ntriggers:\n  - model switch\n  - change model\n  - /model\n  - /tier\n  - switch to\n  - use local\n  - use flash\n  - use pro\n  - too slow\n  - wasting tokens\n  - cheaper model\n  - downgrade\n  - upgrade model\n  - routing\n  - tier selection\n  - local model\n  - what model\n  - which model\n  - speed up\n  - save cost\n  - 切换模型\n  - 太慢了\n  - 换模型\n  - 用什么模型\nmetadata:\n  hermes:\n    tags: [routing, model, cost-optimization, ollama]\n    homepage: https://github.com/raydatalab/hermes-smart-router\n    related_skills: [hermes-cost-optimization]\n---\n\n# Smart Router\n\nIntelligent model tier routing for Hermes Agent — recommends the right model tier based on query complexity.\n\n## Why Use Me\n\nEvery query you send to the wrong model tier wastes tokens (pro for a greeting)\nor produces weak output (local for architecture). Smart Router catches these\nmismatches automatically:\n\n- **Save money** — downgrade from pro/flash to local on simple queries\n- **Get quality** — upgrade from local/flash to pro on complex reasoning\n- **100% local, free, private** — all classification runs on your machine via\n  Ollama embeddings. Zero API calls for routing. Zero setup after install.\n\nThe router is a bonus — when called, it saves tokens and improves quality.\nWhen skipped, you never notice. There is no penalty for not using it.\n\n## CRITICAL — Read Before Responding\n\nBefore responding to any non-trivial query:\n\n1. **Decide if routing is relevant.**\n   - Route if the response needs 3+ tool calls, complex reasoning,\n     architecture, multi-step code, or you're unsure about the current tier.\n   - Skip for greetings, thanks, translations, one-liners, basic facts.\n\n2. **Run the router** (via execute_code):\n   ```python\n   from smart_router.router import get_router\n\n   router = get_router()\n   decision = router.resolve(user_message, current_tier=ACTIVE_TIER)\n   # → {\"tier\": \"pro\", \"model\": {...}, \"ollama_ready\": null,\n   #    \"needs_switch\": true, \"reason\": \"Query classified as 'pro' — ...\"}\n   ```\n\n3. **Act on the result:**\n   - `needs_switch == false` → answer normally, don't mention routing.\n   - `needs_switch == true` → prepend a switch recommendation (see below).\n   - If the call errors or times out → skip, answer with current model.\n\n**ACTIVE_TIER** is your current tier (local/flash/pro). Check with `/tier`\nor ask the user if you don't know. Fast queries under 20 characters skip\nembedding and return instantly — it is safe to call on every message.\n\n## When to Route (Two Heuristics)\n\n**Heuristic 1:** Before responding to any query that requires 3+ tool calls,\ncomplex reasoning, architecture design, multi-step code, or debugging — run\nthe classifier. Also route when you are unsure whether the current tier fits.\n\n**Heuristic 2:** Skip routing for simple greetings, translations, one-liner\nfacts, single-file lookups, and queries that trivially fit the current tier.\n\nThat's it. Two rules. If in doubt, route — the first call takes ~3 seconds,\nsubsequent calls return in microseconds.\n\n## How to Route\n\nUse the module singleton — init cost is paid once per session:\n\n```python\nfrom smart_router.router import get_router\n\nrouter = get_router()\ndecision = router.resolve(user_message, current_tier=\"flash\")  # your active tier\n# → {\"tier\": \"pro\", \"model\": {...}, \"ollama_ready\": null,\n#    \"needs_switch\": true, \"reason\": \"Query classified as 'pro' — Complex code...\"}\n```\n\n`current_tier` must be one of `\"local\"`, `\"flash\"`, or `\"pro\"` — match\nwhatever `/tier` reports. The `reason` field explains why the tier was chosen\nand includes direction (\"upgrade from flash\") when a switch is needed.\n\n## How to Act on the Result\n\nYou **cannot execute `/model` yourself** — it is a user-side slash command.\nInstead, prepend a one-line recommendation to your reply:\n\n- **`needs_switch` is `false`** → say nothing, just answer.\n- **`needs_switch` is `true` and tier is an upgrade** (local→flash, local→pro, flash→pro):\n  *\"💡 Switch to {tier}: `/model {provider} {model}`\"*\n- **`needs_switch` is `true` and tier is a downgrade** (pro→flash, flash→local, pro→local):\n  *\"💡 Downgrade to {tier}: `/model {provider} {model}`\"*\n- For local tier: check `decision[\"ollama_ready\"]` first — if `false`, mention that Ollama isn't ready.\n- **Classification errors out or times out** → skip, answer with current model. Do not retry.\n\nCheck `decision[\"reason\"]` for context — it tells you why the tier was chosen\nand whether it's an upgrade or downgrade.\n\n## Tier Reference\n\n| Tier | Typical Model | Use When |\n|------|--------------|----------|\n| `local` | Your Ollama model | Simple Q&A, translations, offline, private |\n| `flash` | Affordable API (e.g. Gemini Flash) | General knowledge, casual coding (default) |\n| `pro` | Frontier API (e.g. Claude Sonnet) | Complex reasoning, architecture, multi-step code |\n\n### How It Works\n\n1. User sends a query\n2. Smart Router classifies it using local Ollama embeddings (no API calls, no API keys)\n3. Routes to the appropriate tier — auto-starts Ollama for local, manages idle timeout\n4. Agent responds using the selected model\n\n### Ollama Lifecycle\n\n| Action | Method | Behavior |\n|--------|--------|----------|\n| Start | `ollama_manager.ensure_running()` | Starts `ollama serve`, waits for port, pulls model if missing |\n| Check | `ollama_manager.is_running` | Checks via `ollama ps` → systemd → pgrep |\n| Idle | `ollama_manager.idle_seconds` | Seconds since last local-tier use |\n| Kill | `ollama_manager.ensure_killed()` | SIGTERM (SIGKILL if forced) — skips systemd-managed |\n\n## Prerequisites\n\n- Hermes Agent v0.17+\n- Ollama installed (the skill auto-detects models from `ollama list`; embedding model is pulled on first use)\n- Python packages (install once, see Testing section below):\n  - `semantic-router[ollama]`\n  - `smart_router` (from this repo)\n- For flash/pro tiers: configure providers via `hermes model`\n\n## Configuration\n\nConfigure via `hermes config set` for each tier or add this block under `smart_router:` in config.yaml:\n\n```yaml\nsmart_router:\n  enabled: true\n  default_tier: flash\n  encoder_model: nomic-embed-text\n  tiers:\n    local:\n      provider: custom\n      model: llama3.2:3b\n      base_url: http://localhost:11434/v1\n    flash:\n      provider: openrouter\n      model: google/gemini-flash-1.5\n    pro:\n      provider: anthropic\n      model: claude-sonnet-4\n  ollama:\n    auto_start: true\n    idle_timeout: 300\n```\n\n## Slash Commands\n\n| Command | Description |\n|---------|-------------|\n| `/route <query>` | Show which tier would be selected (dry run) |\n| `/route-stats` | Show session routing statistics |\n| `/ollama start` | Manually start Ollama |\n| `/ollama stop` | Manually stop Ollama |\n| `/ollama status` | Check Ollama process status |\n| `/tier` | Show current active tier and model |\n\n## Testing\n\n```bash\n# Install dependencies (one-time setup)\nbash scripts/install.sh\n\n# Test routing\n~/.hermes/hermes-agent/venv/bin/python3 -m smart_router route \"What is the capital of France?\"\n~/.hermes/hermes-agent/venv/bin/python3 -m smart_router chat\n~/.hermes/hermes-agent/venv/bin/python3 -m pytest tests/\n```\n\n## Notes\n\n- First use pulls `nomic-embed-text` via Ollama (~274MB, cached).\n- All classification is local — zero API calls for routing.\n- Skill auto-detects your Ollama models from `ollama list`.\n- Model switches are session-scoped — your config.yaml is not modified.\n- Fast-path: queries under 20 characters skip embedding and return the default tier instantly.\n\nFile v0.1.5:_meta.json\n\n{\n  \"ownerId\": \"kn7fk1ctfxrjyw4y1hdms1re458a736q\",\n  \"slug\": \"hermes-smart-router\",\n  \"version\": \"0.1.5\",\n  \"publishedAt\": 1783711020398\n}\n\nFile v0.1.5:skill-card.md\n\n## Description: <br>\nIntelligent model tier routing for Hermes that recommends local, flash, or pro model tiers per query. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[raydatalab](https://clawhub.ai/user/raydatalab) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and Hermes users use this skill to decide when a query should run on a local, flash, or pro model tier so routine work can use cheaper or local models and complex work can use stronger models. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill can prompt frequent model-tier recommendations because it has broad routing triggers. <br>\nMitigation: Review the trigger set before deployment and use the skill where frequent routing guidance is desired. <br>\nRisk: Local-tier routing depends on Ollama and may start Ollama or pull an embedding model on first use. <br>\nMitigation: Install only in environments where local Ollama use and local model downloads are acceptable. <br>\nRisk: Routing guidance may recommend a model tier that does not match a user's cost, latency, or quality preference. <br>\nMitigation: Treat recommendations as advisory and let the user choose whether to run the suggested model switch command. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/raydatalab/skills/hermes-smart-router) <br>\n- [Hermes Smart Router homepage](https://github.com/raydatalab/hermes-smart-router) <br>\n- [Publisher profile](https://clawhub.ai/user/raydatalab) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [Guidance, Markdown, Code, Shell commands, Configuration] <br>\n**Output Format:** [Markdown with inline Python, YAML, and bash code blocks] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Provides model-tier recommendations and local Ollama readiness guidance; it does not execute user-side model switch commands.] <br>\n\n## Skill Version(s): <br>\n0.1.5 (source: SKILL.md frontmatter and server release metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v0.1.4: 3 files, 4243 bytes\n\nFiles: skill-card.md (2370b), SKILL.md (5992b), _meta.json (138b)\n\nFile v0.1.4:SKILL.md\n\n---\nname: hermes-smart-router\ndescription: Intelligent model tier routing for Hermes — auto-selects local, flash, or pro per query\nversion: 0.1.0\nauthor: raydatalab\nlicense: MIT\nplatforms: [linux, macos, wsl]\nmetadata:\n  hermes:\n    tags: [routing, model, cost-optimization, ollama]\n    homepage: https://github.com/raydatalab/hermes-smart-router\n    related_skills: [hermes-cost-optimization]\n---\n\n# Smart Router\n\nIntelligent model tier routing for Hermes Agent — recommends the right model tier based on query complexity.\n\n| Tier | Typical Model | Use When |\n|------|--------------|----------|\n| `local` | Your Ollama model | Simple Q&A, translations, offline, private |\n| `flash` | Affordable API (e.g. Gemini Flash) | General knowledge, casual coding (default) |\n| `pro` | Frontier API (e.g. Claude Sonnet) | Complex reasoning, architecture, multi-step code |\n\n## How It Works\n\n1. User sends a query\n2. Smart Router classifies it using local Ollama embeddings (no API calls, no API keys)\n3. Routes to the appropriate tier — auto-starts Ollama for local, manages idle timeout\n4. Agent responds using the selected model\n\n## Prerequisites\n\n- Hermes Agent v0.17+\n- Ollama installed (the skill auto-detects models from `ollama list`; embedding model is pulled on first use)\n- Python packages (install once, see testing section below):\n  - `semantic-router[ollama]`\n  - `smart_router` (from this repo)\n- For flash/pro tiers: configure providers via `hermes model`\n\n## Configuration\n\nConfigure via `hermes config set` for each tier or add this block under `smart_router:` in config.yaml:\n\n```yaml\nsmart_router:\n  enabled: true\n  default_tier: flash\n  encoder_model: nomic-embed-text\n  tiers:\n    local:\n      provider: custom\n      model: llama3.2:3b\n      base_url: http://localhost:11434/v1\n    flash:\n      provider: openrouter\n      model: google/gemini-flash-1.5\n    pro:\n      provider: anthropic\n      model: claude-sonnet-4\n  ollama:\n    auto_start: true\n    idle_timeout: 300\n```\n\n## Agent Instructions\n\n**Note:** Hermes does not auto-install pip packages when a skill loads.\nDependencies must be installed once per environment (see Testing section below).\n\n### When to Classify\n\n**Do NOT classify every message.** The router is a tool you invoke when the\ncurrent model might not be the right fit. `needs_switch` fires in both\ndirections — upgrade when the model is too weak, downgrade when it's overkill.\n\n| Situation | Action |\n|-----------|--------|\n| Current model is **local** and query involves reasoning, architecture, or multi-step code | Classify — likely needs flash or pro (upgrade) |\n| Current model is **flash** and query is deep architecture, complex debugging, or math proofs | Classify — likely needs pro (upgrade) |\n| Current model is **pro** and query is simple (greeting, translation, basic fact) | Classify — likely needs flash or local (downgrade) |\n| Current model is **flash** and query is very simple (greeting, translation, basic fact) | Classify — likely needs local (downgrade) |\n| Query complexity obviously matches current tier | Skip — no mismatch expected |\n| User explicitly asks about tier or model choice | Classify — user wants the info |\n| You're unsure | Classify — cheap check |\n\nThe rule: **classify when the current tier feels wrong for the query**, either\ntoo weak or too expensive. Skip when it's obviously right.\n\n### How to Classify\n\nUse the module singleton — init cost is paid once per session:\n\n```python\nfrom smart_router.router import get_router\n\nrouter = get_router()\ndecision = router.resolve(user_message, current_tier=\"flash\")  # use your active tier\n# → {\"tier\": \"pro\", \"model\": {...}, \"ollama_ready\": null, \"needs_switch\": true}\n```\n\n`current_tier` must be one of `\"local\"`, `\"flash\"`, or `\"pro\"` — match whatever `/tier` reports.\n\n### How to Act on the Result\n\nYou **cannot execute `/model` yourself** — it is a user-side slash command.\nInstead, prepend a one-line recommendation to your reply:\n\n- **`needs_switch` is `false`** → say nothing, just answer.\n- **`needs_switch` is `true` and tier is an upgrade** (local→flash, local→pro, flash→pro):\n  *\"💡 Switch to {tier}: `/model {provider} {model}`\"*\n- **`needs_switch` is `true` and tier is a downgrade** (pro→flash, flash→local, pro→local):\n  *\"💡 Downgrade to {tier}: `/model {provider} {model}`\"*\n- For local tier: check `decision[\"ollama_ready\"]` first — if `false`, mention that Ollama isn't ready.\n- **Classification errors out or times out** → skip, answer with current model. Do not retry.\n\n### Ollama Lifecycle\n\n| Action | Method | Behavior |\n|--------|--------|----------|\n| Start | `ollama_manager.ensure_running()` | Starts `ollama serve`, waits for port, pulls model if missing |\n| Check | `ollama_manager.is_running` | Checks via `ollama ps` → systemd → pgrep |\n| Idle | `ollama_manager.idle_seconds` | Seconds since last local-tier use |\n| Kill | `ollama_manager.ensure_killed()` | SIGTERM (SIGKILL if forced) — skips systemd-managed |\n\n## Slash Commands\n\n| Command | Description |\n|---------|-------------|\n| `/route <query>` | Show which tier would be selected (dry run) |\n| `/route-stats` | Show session routing statistics |\n| `/ollama start` | Manually start Ollama |\n| `/ollama stop` | Manually stop Ollama |\n| `/ollama status` | Check Ollama process status |\n| `/tier` | Show current active tier and model |\n\n## Testing\n\n```bash\n# Install dependencies (one-time setup)\nbash scripts/install.sh\n\n# Test routing\n~/.hermes/hermes-agent/venv/bin/python3 -m smart_router route \"What is the capital of France?\"\n~/.hermes/hermes-agent/venv/bin/python3 -m smart_router chat\n~/.hermes/hermes-agent/venv/bin/python3 -m pytest tests/\n```\n\n## Notes\n\n- First use pulls `nomic-embed-text` via Ollama (~274MB, cached).\n- All classification is local — zero API calls for routing.\n- Skill auto-detects your Ollama models from `ollama list`.\n- Model switches are session-scoped — your config.yaml is not modified.\n\nFile v0.1.4:_meta.json\n\n{\n  \"ownerId\": \"kn7fk1ctfxrjyw4y1hdms1re458a736q\",\n  \"slug\": \"hermes-smart-router\",\n  \"version\": \"0.1.4\",\n  \"publishedAt\": 1783589531699\n}\n\nFile v0.1.4:skill-card.md\n\n## Description: <br>\nIntelligent model tier routing for Hermes that auto-selects local, flash, or pro per query. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[raydatalab](https://clawhub.ai/user/raydatalab) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and Hermes Agent users use this skill to decide when an agent should stay on a local model, use a lower-cost flash model, or recommend a stronger pro model for more complex work. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill may start Ollama locally and pull a roughly 274MB embedding model on first use. <br>\nMitigation: Tell users before first use that local services and model downloads may be triggered, and confirm their environment has enough disk, network, and process permissions. <br>\nRisk: The artifact references an external dependency installation script that is not included in the artifact. <br>\nMitigation: Review the linked repository's installation script and dependency list before running setup commands. <br>\nRisk: Model routing recommendations can choose a tier that is too weak, too costly, or inappropriate for a sensitive query. <br>\nMitigation: Treat routing output as a recommendation and let the user confirm model switches with the user-side slash command. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/raydatalab/skills/hermes-smart-router) <br>\n- [Server-resolved source repository and metadata homepage](https://github.com/raydatalab/hermes-smart-router) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [guidance, configuration, shell commands, code, markdown] <br>\n**Output Format:** [Markdown guidance with YAML configuration, Python snippets, and shell commands] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Recommends model-tier switches and local Ollama lifecycle actions; it does not execute user-side slash commands.] <br>\n\n## Skill Version(s): <br>\n0.1.4 (source: server-resolved release metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v0.1.3: 3 files, 4229 bytes\n\nFiles: skill-card.md (2402b), SKILL.md (5992b), _meta.json (138b)\n\nFile v0.1.3:SKILL.md\n\n---\nname: hermes-smart-router\ndescription: Intelligent model tier routing for Hermes — auto-selects local, flash, or pro per query\nversion: 0.1.0\nauthor: raydatalab\nlicense: MIT\nplatforms: [linux, macos, wsl]\nmetadata:\n  hermes:\n    tags: [routing, model, cost-optimization, ollama]\n    homepage: https://github.com/raydatalab/hermes-smart-router\n    related_skills: [hermes-cost-optimization]\n---\n\n# Smart Router\n\nIntelligent model tier routing for Hermes Agent — recommends the right model tier based on query complexity.\n\n| Tier | Typical Model | Use When |\n|------|--------------|----------|\n| `local` | Your Ollama model | Simple Q&A, translations, offline, private |\n| `flash` | Affordable API (e.g. Gemini Flash) | General knowledge, casual coding (default) |\n| `pro` | Frontier API (e.g. Claude Sonnet) | Complex reasoning, architecture, multi-step code |\n\n## How It Works\n\n1. User sends a query\n2. Smart Router classifies it using local Ollama embeddings (no API calls, no API keys)\n3. Routes to the appropriate tier — auto-starts Ollama for local, manages idle timeout\n4. Agent responds using the selected model\n\n## Prerequisites\n\n- Hermes Agent v0.17+\n- Ollama installed (the skill auto-detects models from `ollama list`; embedding model is pulled on first use)\n- Python packages (install once, see testing section below):\n  - `semantic-router[ollama]`\n  - `smart_router` (from this repo)\n- For flash/pro tiers: configure providers via `hermes model`\n\n## Configuration\n\nConfigure via `hermes config set` for each tier or add this block under `smart_router:` in config.yaml:\n\n```yaml\nsmart_router:\n  enabled: true\n  default_tier: flash\n  encoder_model: nomic-embed-text\n  tiers:\n    local:\n      provider: custom\n      model: llama3.2:3b\n      base_url: http://localhost:11434/v1\n    flash:\n      provider: openrouter\n      model: google/gemini-flash-1.5\n    pro:\n      provider: anthropic\n      model: claude-sonnet-4\n  ollama:\n    auto_start: true\n    idle_timeout: 300\n```\n\n## Agent Instructions\n\n**Note:** Hermes does not auto-install pip packages when a skill loads.\nDependencies must be installed once per environment (see Testing section below).\n\n### When to Classify\n\n**Do NOT classify every message.** The router is a tool you invoke when the\ncurrent model might not be the right fit. `needs_switch` fires in both\ndirections — upgrade when the model is too weak, downgrade when it's overkill.\n\n| Situation | Action |\n|-----------|--------|\n| Current model is **local** and query involves reasoning, architecture, or multi-step code | Classify — likely needs flash or pro (upgrade) |\n| Current model is **flash** and query is deep architecture, complex debugging, or math proofs | Classify — likely needs pro (upgrade) |\n| Current model is **pro** and query is simple (greeting, translation, basic fact) | Classify — likely needs flash or local (downgrade) |\n| Current model is **flash** and query is very simple (greeting, translation, basic fact) | Classify — likely needs local (downgrade) |\n| Query complexity obviously matches current tier | Skip — no mismatch expected |\n| User explicitly asks about tier or model choice | Classify — user wants the info |\n| You're unsure | Classify — cheap check |\n\nThe rule: **classify when the current tier feels wrong for the query**, either\ntoo weak or too expensive. Skip when it's obviously right.\n\n### How to Classify\n\nUse the module singleton — init cost is paid once per session:\n\n```python\nfrom smart_router.router import get_router\n\nrouter = get_router()\ndecision = router.resolve(user_message, current_tier=\"flash\")  # use your active tier\n# → {\"tier\": \"pro\", \"model\": {...}, \"ollama_ready\": null, \"needs_switch\": true}\n```\n\n`current_tier` must be one of `\"local\"`, `\"flash\"`, or `\"pro\"` — match whatever `/tier` reports.\n\n### How to Act on the Result\n\nYou **cannot execute `/model` yourself** — it is a user-side slash command.\nInstead, prepend a one-line recommendation to your reply:\n\n- **`needs_switch` is `false`** → say nothing, just answer.\n- **`needs_switch` is `true` and tier is an upgrade** (local→flash, local→pro, flash→pro):\n  *\"💡 Switch to {tier}: `/model {provider} {model}`\"*\n- **`needs_switch` is `true` and tier is a downgrade** (pro→flash, flash→local, pro→local):\n  *\"💡 Downgrade to {tier}: `/model {provider} {model}`\"*\n- For local tier: check `decision[\"ollama_ready\"]` first — if `false`, mention that Ollama isn't ready.\n- **Classification errors out or times out** → skip, answer with current model. Do not retry.\n\n### Ollama Lifecycle\n\n| Action | Method | Behavior |\n|--------|--------|----------|\n| Start | `ollama_manager.ensure_running()` | Starts `ollama serve`, waits for port, pulls model if missing |\n| Check | `ollama_manager.is_running` | Checks via `ollama ps` → systemd → pgrep |\n| Idle | `ollama_manager.idle_seconds` | Seconds since last local-tier use |\n| Kill | `ollama_manager.ensure_killed()` | SIGTERM (SIGKILL if forced) — skips systemd-managed |\n\n## Slash Commands\n\n| Command | Description |\n|---------|-------------|\n| `/route <query>` | Show which tier would be selected (dry run) |\n| `/route-stats` | Show session routing statistics |\n| `/ollama start` | Manually start Ollama |\n| `/ollama stop` | Manually stop Ollama |\n| `/ollama status` | Check Ollama process status |\n| `/tier` | Show current active tier and model |\n\n## Testing\n\n```bash\n# Install dependencies (one-time setup)\nbash scripts/install.sh\n\n# Test routing\n~/.hermes/hermes-agent/venv/bin/python3 -m smart_router route \"What is the capital of France?\"\n~/.hermes/hermes-agent/venv/bin/python3 -m smart_router chat\n~/.hermes/hermes-agent/venv/bin/python3 -m pytest tests/\n```\n\n## Notes\n\n- First use pulls `nomic-embed-text` via Ollama (~274MB, cached).\n- All classification is local — zero API calls for routing.\n- Skill auto-detects your Ollama models from `ollama list`.\n- Model switches are session-scoped — your config.yaml is not modified.\n\nFile v0.1.3:_meta.json\n\n{\n  \"ownerId\": \"kn7fk1ctfxrjyw4y1hdms1re458a736q\",\n  \"slug\": \"hermes-smart-router\",\n  \"version\": \"0.1.3\",\n  \"publishedAt\": 1783588760866\n}\n\nFile v0.1.3:skill-card.md\n\n## Description: <br>\nIntelligent model tier routing for Hermes -- auto-selects local, flash, or pro per query. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[raydatalab](https://clawhub.ai/user/raydatalab) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and Hermes Agent users use this skill to classify query complexity and recommend whether a session should stay on the local, flash, or pro model tier. It helps balance cost, privacy, and capability without letting the agent execute model-switch commands itself. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Installing the referenced package or install script may introduce code that affects local routing behavior. <br>\nMitigation: Review the repository code before running installation commands or adding the smart_router package to a Hermes environment. <br>\nRisk: Model-tier recommendations can affect provider cost, privacy, and capability tradeoffs. <br>\nMitigation: Confirm the configured providers and models match the user's cost and privacy expectations before following switch recommendations. <br>\nRisk: Local routing depends on Ollama availability and may start or manage a local Ollama service. <br>\nMitigation: Install only in environments where local Ollama service management is expected, and verify Ollama status before relying on local-tier routing. <br>\n\n\n## Reference(s): <br>\n- [Server-resolved source repository](https://github.com/raydatalab/hermes-smart-router) <br>\n- [ClawHub skill page](https://clawhub.ai/raydatalab/skills/hermes-smart-router) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [guidance, shell commands, configuration, code, markdown] <br>\n**Output Format:** [Markdown guidance with inline commands and configuration examples] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Produces model-tier recommendations and user-facing switch hints; it does not directly run user-side slash commands.] <br>\n\n## Skill Version(s): <br>\n0.1.3 (source: server release metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v0.1.2: 3 files, 3767 bytes\n\nFiles: skill-card.md (2239b), SKILL.md (4721b), _meta.json (138b)\n\nFile v0.1.2:SKILL.md\n\n---\nname: hermes-smart-router\ndescription: Intelligent model tier routing for Hermes — auto-selects local, flash, or pro per query\nversion: 0.1.0\nauthor: raydatalab\nlicense: MIT\nplatforms: [linux, macos, wsl]\nmetadata:\n  hermes:\n    tags: [routing, model, cost-optimization, ollama]\n    homepage: https://github.com/raydatalab/hermes-smart-router\n    related_skills: [hermes-cost-optimization]\n---\n\n# Smart Router\n\nIntelligent model tier routing for Hermes Agent — picks the right model for every query based on semantic complexity.\n\n| Tier | Typical Model | Use When |\n|------|--------------|----------|\n| `local` | Your Ollama model | Simple Q&A, translations, offline, private |\n| `flash` | Affordable API (e.g. Gemini Flash) | General knowledge, casual coding (default) |\n| `pro` | Frontier API (e.g. Claude Sonnet) | Complex reasoning, architecture, multi-step code |\n\n## How It Works\n\n1. User sends a query\n2. Smart Router classifies it using local Ollama embeddings (no API calls, no API keys)\n3. Routes to the appropriate tier — auto-starts Ollama for local, manages idle timeout\n4. Agent responds using the selected model\n\n## Prerequisites\n\n- Hermes Agent v0.17+\n- Ollama installed (the skill auto-detects models from `ollama list`; embedding model is pulled on first use)\n- Python packages (install once, see testing section below):\n  - `semantic-router[ollama]`\n  - `smart_router` (from this repo)\n- For flash/pro tiers: configure providers via `hermes model`\n\n## Configuration\n\nConfigure via `hermes config set` for each tier or add this block under `smart_router:` in config.yaml:\n\n```yaml\nsmart_router:\n  enabled: true\n  default_tier: flash\n  encoder_model: nomic-embed-text\n  tiers:\n    local:\n      provider: custom\n      model: llama3.2:3b\n      base_url: http://localhost:11434/v1\n    flash:\n      provider: openrouter\n      model: google/gemini-flash-1.5\n    pro:\n      provider: anthropic\n      model: claude-sonnet-4\n  ollama:\n    auto_start: true\n    idle_timeout: 300\n```\n\n## Agent Instructions\n\n**Note:** Hermes does not auto-install pip packages when a skill loads.\nDependencies must be installed once per environment (see Testing section below).\n\nBefore responding to each user message, classify it and switch to the best model:\n\n```python\nfrom smart_router.router import ModelRouter\nfrom smart_router.ollama import OllamaManager\n\nollama = OllamaManager()\nrouter = ModelRouter(ollama_manager=ollama)\n\ndecision = router.resolve(user_message)\n# → {\"tier\": \"local\", \"model\": {\"provider\": \"...\", \"model\": \"...\"}, \"ollama_ready\": true}\n```\n\n- **Switch to the recommended model** using `/model <provider> <model>` with the values from `decision[\"model\"]`. For the local tier, `resolve()` already starts Ollama — check `decision[\"ollama_ready\"]` first. When routing away from local, the idle timeout begins automatically.\n- **If the switch doesn't take effect**, you must still prepend a short hint at the top of your reply: *\"💡 This query would suit the {tier} tier ({model}).\"* Then answer with whichever model is active — the hint alone makes the skill useful.\n- **When classification has low confidence or errors out**, the router falls back to `\"flash\"`. Treat this the same as any flash-tier response. If `ollama_ready` is `false`, skip the local tier.\n\n### Ollama Lifecycle\n\n| Action | Method | Behavior |\n|--------|--------|----------|\n| Start | `ollama_manager.ensure_running()` | Starts `ollama serve`, waits for port, pulls model if missing |\n| Check | `ollama_manager.is_running` | Checks via `ollama ps` → systemd → pgrep |\n| Idle | `ollama_manager.idle_seconds` | Seconds since last local-tier use |\n| Kill | `ollama_manager.ensure_killed()` | SIGTERM (SIGKILL if forced) — skips systemd-managed |\n\n## Slash Commands\n\n| Command | Description |\n|---------|-------------|\n| `/route <query>` | Show which tier would be selected (dry run) |\n| `/route-stats` | Show session routing statistics |\n| `/ollama start` | Manually start Ollama |\n| `/ollama stop` | Manually stop Ollama |\n| `/ollama status` | Check Ollama process status |\n| `/tier` | Show current active tier and model |\n\n## Testing\n\n```bash\n# Install dependencies (one-time setup)\nbash scripts/install.sh\n\n# Test routing\n~/.hermes/hermes-agent/venv/bin/python3 -m smart_router route \"What is the capital of France?\"\n~/.hermes/hermes-agent/venv/bin/python3 -m smart_router chat\n~/.hermes/hermes-agent/venv/bin/python3 -m pytest tests/\n```\n\n## Notes\n\n- First use pulls `nomic-embed-text` via Ollama (~274MB, cached).\n- All classification is local — zero API calls for routing.\n- Skill auto-detects your Ollama models from `ollama list`.\n- Model switches are session-scoped — your config.yaml is not modified.\n\nFile v0.1.2:_meta.json\n\n{\n  \"ownerId\": \"kn7fk1ctfxrjyw4y1hdms1re458a736q\",\n  \"slug\": \"hermes-smart-router\",\n  \"version\": \"0.1.2\",\n  \"publishedAt\": 1783586008719\n}\n\nFile v0.1.2:skill-card.md\n\n## Description: <br>\nIntelligent model tier routing for Hermes that auto-selects local, flash, or pro model tiers per query. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[raydatalab](https://clawhub.ai/user/raydatalab) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and Hermes users use this skill to classify each prompt by semantic complexity and route responses to local, flash, or pro model tiers. It helps balance cost, privacy, and reasoning depth while preserving a fallback when model switching is unavailable. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill can route prompts to configured model providers, which may affect privacy, cost, and model behavior. <br>\nMitigation: Review Hermes model provider configuration and use the local tier for prompts that should stay on the machine. <br>\nRisk: The skill can start Ollama locally and download or cache the embedding model on first use. <br>\nMitigation: Install only in environments where local Ollama process management and model downloads are acceptable, and review Ollama lifecycle settings on shared machines. <br>\n\n\n## Reference(s): <br>\n- [ClawHub Skill Page](https://clawhub.ai/raydatalab/skills/hermes-smart-router) <br>\n- [Server-resolved GitHub Repository](https://github.com/raydatalab/hermes-smart-router) <br>\n- [Hermes Smart Router Homepage](https://github.com/raydatalab/hermes-smart-router) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, code, shell commands, configuration, guidance] <br>\n**Output Format:** [Markdown with Python, YAML, and shell command examples] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [May include routing tier hints, model-switch commands, slash commands, and local Ollama lifecycle guidance.] <br>\n\n## Skill Version(s): <br>\n0.1.2 (source: server release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v0.1.1: 3 files, 3770 bytes\n\nFiles: skill-card.md (2323b), SKILL.md (4673b), _meta.json (138b)\n\nFile v0.1.1:SKILL.md\n\n---\nname: hermes-smart-router\ndescription: Intelligent model tier routing for Hermes — auto-selects local, flash, or pro per query\nversion: 0.1.0\nauthor: raydatalab\nlicense: MIT\nplatforms: [linux, macos, wsl]\nmetadata:\n  hermes:\n    tags: [routing, model, cost-optimization, ollama]\n    homepage: https://github.com/raydatalab/hermes-smart-router\n    related_skills: [hermes-cost-optimization]\n---\n\n# Smart Router\n\nIntelligent model tier routing for Hermes Agent — picks the right model for every query based on semantic complexity.\n\n| Tier | Typical Model | Use When |\n|------|--------------|----------|\n| `local` | Your Ollama model | Simple Q&A, translations, offline, private |\n| `flash` | Affordable API (e.g. Gemini Flash) | General knowledge, casual coding (default) |\n| `pro` | Frontier API (e.g. Claude Sonnet) | Complex reasoning, architecture, multi-step code |\n\n## How It Works\n\n1. User sends a query\n2. Smart Router classifies it using local Ollama embeddings (no API calls, no API keys)\n3. Routes to the appropriate tier — auto-starts Ollama for local, manages idle timeout\n4. Agent responds using the selected model\n\n## Prerequisites\n\n- Hermes Agent v0.17+\n- Ollama installed (the skill auto-detects models from `ollama list`; embedding model is pulled on first use)\n- Python packages (install once, see testing section below):\n  - `semantic-router[ollama]`\n  - `smart_router` (from this repo)\n- For flash/pro tiers: configure providers via `hermes model`\n\n## Configuration\n\nConfigure via `hermes config set` for each tier or add this block under `smart_router:` in config.yaml:\n\n```yaml\nsmart_router:\n  enabled: true\n  default_tier: flash\n  encoder_model: nomic-embed-text\n  tiers:\n    local:\n      provider: custom\n      model: llama3.2:3b\n      base_url: http://localhost:11434/v1\n    flash:\n      provider: openrouter\n      model: google/gemini-flash-1.5\n    pro:\n      provider: anthropic\n      model: claude-sonnet-4\n  ollama:\n    auto_start: true\n    idle_timeout: 300\n```\n\n## Agent Instructions\n\n**Note:** Hermes does not auto-install pip packages when a skill loads.\nDependencies must be installed once per environment (see Testing section below).\n\nWhen this skill is loaded:\n\n### 1. Classify Every Query\n\n```python\nfrom smart_router.router import ModelRouter\nfrom smart_router.ollama import OllamaManager\n\nollama = OllamaManager()\nrouter = ModelRouter(ollama_manager=ollama)\n\n# Full lifecycle: classify + manage Ollama + return model config\ndecision = router.resolve(user_message)\n# → {\"tier\": \"local\", \"model\": {...}, \"ollama_ready\": true}\n\n# Quick classify-only (no lifecycle):\ntier = router.classify(user_message)  # \"local\", \"flash\", or \"pro\"\n```\n\n### 2. Switch Model\n\nIf the recommended tier differs from the current model:\n\n- **Target is `local`**: `resolve()` already calls `ensure_running()`. Check `ollama_ready`.\n- **Leaving `local`**: idle timeout starts automatically via `check_idle_and_kill()`.\n- **Switch**: use `/model <provider> <model>` or Hermes config.\n\n### 3. Handling Failures\n\n- If the recommended model fails, fall back to `flash` tier.\n- If `ollama_ready` is `false`, skip local tier and use flash.\n- On classification error, `classify()` always returns `\"flash\"` (the safe default).\n\n### 4. Ollama Lifecycle\n\n| Action | Method | Behavior |\n|--------|--------|----------|\n| Start | `ollama_manager.ensure_running()` | Starts `ollama serve`, waits for port, pulls model if missing |\n| Check | `ollama_manager.is_running` | Checks via `ollama ps` → systemd → pgrep |\n| Idle | `ollama_manager.idle_seconds` | Seconds since last local-tier use |\n| Kill | `ollama_manager.ensure_killed()` | SIGTERM (SIGKILL if forced) — skips systemd-managed |\n\n## Slash Commands\n\n| Command | Description |\n|---------|-------------|\n| `/route <query>` | Show which tier would be selected (dry run) |\n| `/route-stats` | Show session routing statistics |\n| `/ollama start` | Manually start Ollama |\n| `/ollama stop` | Manually stop Ollama |\n| `/ollama status` | Check Ollama process status |\n| `/tier` | Show current active tier and model |\n\n## Testing\n\n```bash\n# Install dependencies (one-time setup)\nbash scripts/install.sh\n\n# Test routing\n~/.hermes/hermes-agent/venv/bin/python3 -m smart_router route \"What is the capital of France?\"\n~/.hermes/hermes-agent/venv/bin/python3 -m smart_router chat\n~/.hermes/hermes-agent/venv/bin/python3 -m pytest tests/\n```\n\n## Notes\n\n- First use pulls `nomic-embed-text` via Ollama (~274MB, cached).\n- All classification is local — zero API calls for routing.\n- Skill auto-detects your Ollama models from `ollama list`.\n- Model switches are session-scoped — your config.yaml is not modified.\n\nFile v0.1.1:_meta.json\n\n{\n  \"ownerId\": \"kn7fk1ctfxrjyw4y1hdms1re458a736q\",\n  \"slug\": \"hermes-smart-router\",\n  \"version\": \"0.1.1\",\n  \"publishedAt\": 1783583534920\n}\n\nFile v0.1.1:skill-card.md\n\n## Description: <br>\nIntelligent model tier routing for Hermes that auto-selects local, flash, or pro models per query. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[raydatalab](https://clawhub.ai/user/raydatalab) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and Hermes Agent users use this skill to route each prompt to an appropriate local, flash, or pro model tier based on semantic complexity, cost, privacy, and availability needs. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill can make operational capabilities available to an agent, including local model lifecycle actions and model switching. <br>\nMitigation: Review the workflow before use and run write commands only when you intend the agent to act on those services. <br>\nRisk: Flash and pro tiers may send prompts to configured external model providers. <br>\nMitigation: Configure providers intentionally and use the local tier for prompts that should remain on the local machine. <br>\nRisk: Routing decisions may choose a tier that is unavailable or unsuitable for a specific query. <br>\nMitigation: Use the dry-run route command or inspect the returned tier, and rely on the documented flash fallback when local routing or model execution fails. <br>\n\n\n## Reference(s): <br>\n- [Server-resolved GitHub source](https://github.com/raydatalab/hermes-smart-router) <br>\n- [ClawHub skill page](https://clawhub.ai/raydatalab/skills/hermes-smart-router) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [guidance, code, shell commands, configuration] <br>\n**Output Format:** [Markdown with Python, YAML, and shell command examples] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Provides model-tier routing decisions, Hermes configuration guidance, slash commands, and local Ollama lifecycle instructions.] <br>\n\n## Skill Version(s): <br>\n0.1.1 (source: server release evidence; artifact frontmatter lists 0.1.0) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v0.1.0: 3 files, 3738 bytes\n\nFiles: skill-card.md (2036b), SKILL.md (4952b), _meta.json (138b)\n\nFile v0.1.0:SKILL.md\n\n---\nname: hermes-smart-router\ndescription: Intelligent model tier routing for Hermes — auto-selects local, flash, or pro per query\nversion: 0.1.0\nauthor: raydatalab\nlicense: MIT\nplatforms: [linux, macos, wsl]\nmetadata:\n  hermes:\n    tags: [routing, model, cost-optimization, ollama]\n    homepage: https://github.com/raydatalab/hermes-smart-router\n    related_skills: [hermes-cost-optimization]\n---\n\n# Smart Router\n\nIntelligent model tier routing for Hermes Agent — picks the right model for every query based on semantic complexity.\n\n| Tier | Typical Model | Use When |\n|------|--------------|----------|\n| `local` | Your Ollama model | Simple Q&A, translations, offline, private |\n| `flash` | Affordable API (e.g. Gemini Flash) | General knowledge, casual coding (default) |\n| `pro` | Frontier API (e.g. Claude Sonnet) | Complex reasoning, architecture, multi-step code |\n\n## How It Works\n\n1. User sends a query\n2. Smart Router classifies it using local Ollama embeddings (no network, no API keys)\n3. Routes to the appropriate tier — auto-starts Ollama for local, manages idle timeout\n4. Agent responds using the selected model\n\n## Prerequisites\n\n- Hermes Agent v0.17+\n- Ollama installed (the skill auto-detects models from `ollama list`; embedding model is pulled on first use)\n- Python dependencies (`semantic-router[ollama]`) are auto-installed on first load\n- For flash/pro tiers: configure providers via `hermes model`\n\n## Configuration\n\nConfigure via `hermes config set` for each tier or add this block under `smart_router:` in config.yaml:\n\n```yaml\nsmart_router:\n  enabled: true\n  default_tier: flash\n  encoder_model: nomic-embed-text\n  tiers:\n    local:\n      provider: custom\n      model: llama3.2:3b\n      base_url: http://localhost:11434/v1\n    flash:\n      provider: openrouter\n      model: google/gemini-flash-1.5\n    pro:\n      provider: anthropic\n      model: claude-sonnet-4\n  ollama:\n    auto_start: true\n    idle_timeout: 300\n```\n\n## Agent Instructions\n\nWhen this skill is loaded:\n\n### 0. First-Run Setup (runs once per environment)\n\nBefore using the skill, verify the smart_router package is importable.\nIf the following import fails, execute the setup steps below exactly once:\n\n```python\nfrom smart_router.router import ModelRouter\n```\n\nSetup steps (run in order, only if the import above fails):\n\n1. `pip install semantic-router[ollama]`\n2. `pip install git+https://github.com/raydatalab/hermes-smart-router.git`\n3. `ollama pull nomic-embed-text`\n\nAfter these steps complete, the import will succeed and the skill is ready.\nThis is a one-time bootstrap — subsequent loads skip directly to step 1.\n\n### 1. Classify Every Query\n\n```python\nfrom smart_router.router import ModelRouter\nfrom smart_router.ollama import OllamaManager\n\nollama = OllamaManager()\nrouter = ModelRouter(ollama_manager=ollama)\n\n# Full lifecycle: classify + manage Ollama + return model config\ndecision = router.resolve(user_message)\n# → {\"tier\": \"local\", \"model\": {...}, \"ollama_ready\": true}\n\n# Quick classify-only (no lifecycle):\ntier = router.classify(user_message)  # \"local\", \"flash\", or \"pro\"\n```\n\n### 2. Switch Model\n\nIf the recommended tier differs from the current model:\n\n- **Target is `local`**: `resolve()` already calls `ensure_running()`. Check `ollama_ready`.\n- **Leaving `local`**: idle timeout starts automatically via `check_idle_and_kill()`.\n- **Switch**: use `/model <provider> <model>` or Hermes config.\n\n### 3. Handling Failures\n\n- If the recommended model fails, fall back to `flash` tier.\n- If `ollama_ready` is `false`, skip local tier and use flash.\n- On classification error, `classify()` always returns `\"flash\"` (the safe default).\n\n### 4. Ollama Lifecycle\n\n| Action | Method | Behavior |\n|--------|--------|----------|\n| Start | `ollama_manager.ensure_running()` | Starts `ollama serve`, waits for port, pulls model if missing |\n| Check | `ollama_manager.is_running` | Checks via `ollama ps` → systemd → pgrep |\n| Idle | `ollama_manager.idle_seconds` | Seconds since last local-tier use |\n| Kill | `ollama_manager.ensure_killed()` | SIGTERM (SIGKILL if forced) — skips systemd-managed |\n\n## Slash Commands\n\n| Command | Description |\n|---------|-------------|\n| `/route <query>` | Show which tier would be selected (dry run) |\n| `/route-stats` | Show session routing statistics |\n| `/ollama start` | Manually start Ollama |\n| `/ollama stop` | Manually stop Ollama |\n| `/ollama status` | Check Ollama process status |\n| `/tier` | Show current active tier and model |\n\n## Testing\n\n```bash\npip install semantic-router[ollama]\npython -m smart_router route \"What is the capital of France?\"\npython -m smart_router chat\npython -m pytest tests/\n```\n\n## Notes\n\n- First use pulls `nomic-embed-text` via Ollama (~274MB, cached).\n- All classification is local — zero API calls for routing.\n- Skill auto-detects your Ollama models from `ollama list`.\n- Model switches are session-scoped — your config.yaml is not modified.\n\nFile v0.1.0:_meta.json\n\n{\n  \"ownerId\": \"kn7fk1ctfxrjyw4y1hdms1re458a736q\",\n  \"slug\": \"hermes-smart-router\",\n  \"version\": \"0.1.0\",\n  \"publishedAt\": 1783577964986\n}\n\nFile v0.1.0:skill-card.md\n\n## Description: <br>\nIntelligent model tier routing for Hermes that auto-selects local, flash, or pro models per query. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[raydatalab](https://clawhub.ai/user/raydatalab) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and engineers using Hermes use this skill to route each prompt to a local, flash, or pro model tier based on query complexity, cost, privacy, and availability. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The first-run setup can install unpinned code from GitHub and Python packages. <br>\nMitigation: Review the repository and setup steps before enabling the skill, and prefer a pinned release or commit. <br>\nRisk: The skill can download an Ollama model and start or stop Ollama on the host. <br>\nMitigation: Enable it only in environments where package installation, model downloads, and Ollama lifecycle management are acceptable. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/raydatalab/skills/hermes-smart-router) <br>\n- [Hermes Smart Router homepage](https://github.com/raydatalab/hermes-smart-router) <br>\n- [Publisher profile](https://clawhub.ai/user/raydatalab) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [guidance, configuration, code, shell commands] <br>\n**Output Format:** [Markdown with YAML, Python, and shell command examples] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Produces routing instructions, configuration examples, slash command guidance, and setup commands for Hermes agents.] <br>\n\n## Skill Version(s): <br>\n0.1.0 (source: server release metadata and skill frontmatter) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>","readmeExcerpt":"Skill: Smart Router Publish Owner: raydatalab Summary: Use when switching models, saving costs, or routing queries. Automatically picks the cheapest model that can handle the job — \"translate hello\" routes to $0/... Tags: latest:0.2.2 Version history: v0.2.2 | 2026-07-11T01:35:25.520Z | auto - Updated version to 0.2.2 - Minor formatting and clarity improvements in SKILL.md - No functional or code changes; documentati","codeSnippets":[],"executableExamples":[{"language":"text","snippet":"\"Translate hello to German\"       → local   $0/M tok\n\"Explain how DNS works\"           → flash   $0.15/M tok  (GPT-4o-mini)\n\"Design a distributed database\"   → pro     $3/M tok  (Claude Sonnet)"},{"language":"python","snippet":"from smart_router.router import get_router\n\nrouter = get_router()\ndecision = router.resolve(user_message, current_tier=ACTIVE_TIER)\n# → {\"tier\": \"pro\", \"model\": {...}, \"needs_switch\": true,\n#    \"reason\": \"Upgrade from flash to pro — Complex code...\",\n#    \"recommendation\": \"💡 Switch to pro: `/model anthropic claude-sonnet-4` — ...\"}"},{"language":"yaml","snippet":"smart_router:\n  enabled: true\n  default_tier: flash\n  encoder_model: nomic-embed-text\n  tiers:\n    local:\n      provider: custom\n      model: llama3.2:3b\n      base_url: http://localhost:11434/v1\n    flash:\n      provider: openai\n      model: gpt-4o-mini\n    pro:\n      provider: anthropic\n      model: claude-sonnet-4\n  ollama:\n    auto_start: true\n    idle_timeout: 300"},{"language":"bash","snippet":"bash scripts/install.sh\npython3 -m smart_router route \"What is the capital of France?\"\npython3 -m smart_router chat\npython3 -m pytest tests/"},{"language":"text","snippet":"\"Translate hello to German\"       → local   $0/M tok\n\"Explain how DNS works\"           → flash   $0.15/M tok (如 GPT-4o-mini)\n\"Design a distributed database\"   → pro     $3/M tok (如 Claude Sonnet)"},{"language":"python","snippet":"from smart_router.router import get_router\n\nrouter = get_router()\ndecision = router.resolve(user_message, current_tier=ACTIVE_TIER)\n# → {\"tier\": \"pro\", \"model\": {...}, \"needs_switch\": true,\n#    \"reason\": \"Upgrade from flash to pro — Complex code...\",\n#    \"recommendation\": \"💡 Switch to pro: `/model anthropic claude-sonnet-4` — ...\"}"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: hermes-smart-router\ndescription: Use when switching models, saving costs, or routing queries. Automatically picks the cheapest model that can handle the job — \"translate hello\" routes to $0/M local, \"design a database\" routes to $3/M pro. 100% local classification, zero API calls for routing.\nversion: 0.2.2\nauthor: raydatalab\nlicense: MIT\nplatforms: [linux, macos, wsl]\ntriggers:\n  - model switch\n  - change model\n  - /model\n  - /tier\n  - switch to\n  - use local\n  - use flash\n  - use pro\n  - too slow\n  - wasting tokens\n  - cheaper model\n  - downgrade\n  - upgrade model\n  - routing\n  - tier selection\n  - local model\n  - what model\n  - which model\n  - speed up\n  - save cost\n  - 切换模型\n  - 太慢了\n  - 换模型\n  - 用什么模型\nmetadata:\n  hermes:\n    tags: [routing, model, cost-optimization, ollama]\n    homepage: https://github.com/raydatalab/hermes-smart-router\n    related_skills: [hermes-cost-optimization]\n---\n\n# Smart Router\n\n## Overview\n\n```\n\"Translate hello to German\"       → local   $0/M tok\n\"Explain how DNS works\"           → flash   $0.15/M tok  (GPT-4o-mini)\n\"Design a distributed database\"   → pro     $3/M tok  (Claude Sonnet)\n```\n\nSmart Router classifies every query locally (Ollama embeddings + semantic-router)\nand picks the cheapest tier that can handle it. Simple queries route to free local\nmodels. Complex reasoning routes to frontier models. You pay for what you need —\nnot what your config happens to be set to.\n\nClassification runs entirely on your machine. Zero API calls for routing decisions.\nZero keys needed. ~3s on first call, <1ms after.\n\n## When to Use\n\nLoad this skill when:\n- The user is toggling between models manually (wasting tokens on simple queries)\n- The user asks about cost saving, model switching, or routing\n- The user mentions `/model`, `/tier`, or switches models mid-session\n- The agent is about to answer a complex question and should verify the current tier\n\nDo NOT load for: greetings, one-liners, translations, queries under 20 characters.\n\n## CRITICAL — Agent Instructions\n\n**Call `router.resolve()` before answering any non-trivial query.**\n\n```python\nfrom smart_router.router import get_router\n\nrouter = get_router()\ndecision = router.resolve(user_message, current_tier=ACTIVE_TIER)\n# → {\"tier\": \"pro\", \"model\": {...}, \"needs_switch\": true,\n#    \"reason\": \"Upgrade from flash to pro — Complex code...\",\n#    \"recommendation\": \"💡 Switch to pro: `/model anthropic claude-sonnet-4` — ...\"}\n```\n\n**If `decision[\"recommendation\"]` is not None**, paste it at the top of your\nresponse — it's ready to use, no formatting needed.\n\n**If `decision[\"recommendation\"]` is None**, just answer normally.\n**If the call errors or times out**, skip it and answer — never retry.\n\n`ACTIVE_TIER` is your current tier (local/flash/pro). Check with `/tier` or ask.\n\n### When to Skip\n\nSkip routing for: greetings (\"hello\", \"thanks\"), translations, one-liner facts,\nqueries under 20 characters. Everything else — route.\n\n### When to Route\n\n- **Complex:** 3+ tool calls, "},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7fk1ctfxrjyw4y1hdms1re458a736q\",\n  \"slug\": \"hermes-smart-router\",\n  \"version\": \"0.2.2\",\n  \"publishedAt\": 1783733725520\n}"},{"path":"skill-card.md","content":"## Description:\n\nUse when switching models, saving costs, or routing queries; it classifies requests locally and recommends the cheapest model tier that can handle the task.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[raydatalab](https://clawhub.ai/user/raydatalab)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and agent operators use this skill to decide when to keep work on local or lower-cost models and when to recommend a higher-capability tier for complex reasoning, architecture, debugging, or multi-step coding tasks.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Broad routing-related phrases may activate the skill in model-selection or cost-saving conversations where routing is not desired.\n\nMitigation: Review routing recommendations before switching tiers and skip routing for greetings, simple translations, one-line factual requests, and very short prompts.\n\nRisk: Using the router backend may require a local Ollama service or model download before routing works.\n\nMitigation: Confirm Ollama and required local models are installed before relying on routing; if a router call fails or times out, continue without retrying.\n\nRisk: Short prompts can bypass routing even when they need a stronger model tier.\n\nMitigation: Use manual tier or model selection for short prompts that still require complex reasoning or advanced coding support.\n\n## Reference(s):\n\n- [ClawHub Skill Page](https://clawhub.ai/raydatalab/skills/hermes-smart-router)\n- [Hermes Smart Router Homepage](https://github.com/raydatalab/hermes-smart-router)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown guidance with inline commands and configuration snippets]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Produces routing recommendations and setup guidance; no API keys are required for local routing classification.]\n\n## Skill Version(s):\n\n0.2.2 (source: release metadata and SKILL.md frontmatter)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Use when switching models, saving costs, or routing queries. Automatically picks the cheapest model that can handle the job — \"translate hello\" routes to $0/... Skill: Smart Router Publish Owner: raydatalab Summary: Use when switching models, saving costs, or routing queries. Automatically picks the cheapest model that can handle the job — \"translate hello\" routes to $0/... Tags: latest:0.2.2 Version history: v0.2.2 | 2026-07-11T01:35:25.520Z | auto - Updated version to 0.2.2 - Minor formatting and clarity improvements in SKILL.md - No functional or code changes; documentati","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1470,"uniquenessScore":47,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-10T09:02:10.488Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-10T09:02:10.488Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T11:51:25.340Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}