{"id":"0a897ac3-03bd-401b-b231-8c178c7b3d5b","entityType":"agent","slug":"clawhub-zc-kama-gateway-resilience-guard","name":"OpenClaw Gateway Resilience Guard","canonicalUrl":"https://www.xpersona.co/agent/clawhub-zc-kama-gateway-resilience-guard","canonicalPath":"/agent/clawhub-zc-kama-gateway-resilience-guard","generatedAt":"2026-10-11T05:43:22.635Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-11T03:00:00.984Z","emptyReason":null},"description":"OpenClaw Gateway Resilience Guard keeps Gateway, channels, logs, and optional model-provider probes observable after Wi-Fi changes, WSL sleep/resume, session... Skill: OpenClaw Gateway Resilience Guard Owner: zc-kama Summary: OpenClaw Gateway Resilience Guard keeps Gateway, channels, logs, and optional model-provider probes observable after Wi-Fi changes, WSL sleep/resume, session... Tags: bash:1.2.0, channels:1.2.0, dashboard:1.4.4, gateway:1.4.4, guardian:1.2.0, latest:1.4.4, macos:1.4.4, openclaw:1.4.4, resilience:1.4.4, systemd:1.2.0, watchdog:1.4.4, wechat:1.4.4, weixin","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.2K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s1708vs1r95rbeb3za8tjcexfn871nt4:gateway-resilience-guard","sourceUrl":"https://clawhub.ai/zc-kama/gateway-resilience-guard","homepage":"https://clawhub.ai/zc-kama/skills/gateway-resilience-guard","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/zc-kama/gateway-resilience-guard","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/zc-kama/skills/gateway-resilience-guard","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":61,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"OpenClaw Gateway Resilience Guard keeps Gateway, channels, logs, and optional model-provider probes observable after Wi-Fi changes, WSL sleep/resume, session..."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T03:00:00.984Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T03:00:00.984Z","emptyReason":null},"stars":null,"forks":null,"downloads":1180,"packageName":null,"latestVersion":"1.4.4","tractionLabel":"1.2K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T03:00:00.971Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T03:00:00.984Z","lastCrawledAt":"2026-10-11T03:00:00.971Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T03:00:00.971Z","lastVerifiedAt":null,"highlights":[{"version":"1.4.4","createdAt":"2026-05-21T17:06:43.702Z","changelog":"Make dashboard controls robust with native semantic buttons/selects, fixed hover states, centered brand mark, and non-overlapping config rows.","fileCount":20,"zipByteSize":67828},{"version":"1.4.3","createdAt":"2026-05-21T16:46:22.393Z","changelog":"Rebuild dashboard on local Fluent Web Components with denser pages, refined controls, motion, and bundled vendor assets.","fileCount":21,"zipByteSize":153868},{"version":"1.4.2","createdAt":"2026-05-21T16:16:40.353Z","changelog":"Redesign dashboard navigation, alignment, trend chart axis, strategy controls, and visual system.","fileCount":19,"zipByteSize":60605},{"version":"1.4.1","createdAt":"2026-05-21T15:51:15.499Z","changelog":"Polish dashboard layout, language switching, themes, event trend chart, log following, and guarded action unlock.","fileCount":19,"zipByteSize":59097},{"version":"1.4.0","createdAt":"2026-05-20T20:31:34.069Z","changelog":"Add standalone localhost dashboard with charts, strategy presets, guarded actions, and an OpenClaw plugin bridge.","fileCount":19,"zipByteSize":53948},{"version":"1.3.1","createdAt":"2026-05-20T20:05:04.627Z","changelog":"Include Windows PowerShell scripts in the ClawHub package.","fileCount":12,"zipByteSize":36186},{"version":"1.3.0","createdAt":"2026-05-20T20:03:11.219Z","changelog":"Add OpenClaw runtime diagnostics, log-signal classification, and opt-in model-provider probes.","fileCount":9,"zipByteSize":26751},{"version":"1.2.0","createdAt":"2026-05-19T21:30:47.532Z","changelog":"Add native OpenClaw probes and cross-platform installers","fileCount":13,"zipByteSize":24248}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s1708vs1r95rbeb3za8tjcexfn871nt4:gateway-resilience-guard","setupComplexity":"low","setupSteps":["Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zc-kama-gateway-resilience-guard/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zc-kama-gateway-resilience-guard/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zc-kama-gateway-resilience-guard/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-zc-kama-gateway-resilience-guard/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-zc-kama-gateway-resilience-guard/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-zc-kama-gateway-resilience-guard/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T05:43:22.630Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zc-kama-gateway-resilience-guard/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zc-kama-gateway-resilience-guard/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zc-kama-gateway-resilience-guard/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zc-kama-gateway-resilience-guard/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-11T03:00:00.984Z","emptyReason":null},"readme":"Skill: OpenClaw Gateway Resilience Guard\n\nOwner: zc-kama\n\nSummary: OpenClaw Gateway Resilience Guard keeps Gateway, channels, logs, and optional model-provider probes observable after Wi-Fi changes, WSL sleep/resume, session...\n\nTags: bash:1.2.0, channels:1.2.0, dashboard:1.4.4, gateway:1.4.4, guardian:1.2.0, latest:1.4.4, macos:1.4.4, openclaw:1.4.4, resilience:1.4.4, systemd:1.2.0, watchdog:1.4.4, wechat:1.4.4, weixin:1.2.0, windows:1.4.4, wsl:1.4.4\n\nVersion history:\n\nv1.4.4 | 2026-05-21T17:06:43.702Z | user\n\nMake dashboard controls robust with native semantic buttons/selects, fixed hover states, centered brand mark, and non-overlapping config rows.\n\nv1.4.3 | 2026-05-21T16:46:22.393Z | user\n\nRebuild dashboard on local Fluent Web Components with denser pages, refined controls, motion, and bundled vendor assets.\n\nv1.4.2 | 2026-05-21T16:16:40.353Z | user\n\nRedesign dashboard navigation, alignment, trend chart axis, strategy controls, and visual system.\n\nv1.4.1 | 2026-05-21T15:51:15.499Z | user\n\nPolish dashboard layout, language switching, themes, event trend chart, log following, and guarded action unlock.\n\nv1.4.0 | 2026-05-20T20:31:34.069Z | user\n\nAdd standalone localhost dashboard with charts, strategy presets, guarded actions, and an OpenClaw plugin bridge.\n\nv1.3.1 | 2026-05-20T20:05:04.627Z | user\n\nInclude Windows PowerShell scripts in the ClawHub package.\n\nv1.3.0 | 2026-05-20T20:03:11.219Z | user\n\nAdd OpenClaw runtime diagnostics, log-signal classification, and opt-in model-provider probes.\n\nv1.2.0 | 2026-05-19T21:30:47.532Z | user\n\nAdd native OpenClaw probes and cross-platform installers\n\nv1.1.0 | 2026-05-19T20:02:11.176Z | user\n\nRename package and expand public summary\n\nArchive index:\n\nArchive v1.4.4: 20 files, 67828 bytes\n\nFiles: CHANGELOG.md (5186b), dashboard/server.py (17990b), dashboard/static/app.js (28515b), dashboard/static/index.html (12785b), dashboard/static/styles.css (21358b), gateway-watchdog.ps1 (29656b), gateway-watchdog.sh (31639b), install-watchdog.ps1 (5020b), install-watchdog.sh (10659b), openclaw-plugin/index.js (1493b), openclaw-plugin/openclaw.plugin.json (624b), openclaw-plugin/package.json (485b), README-watchdog.md (1588b), README.md (15992b), README.zh-CN.md (15896b), skill-card.md (2497b), SKILL.md (4331b), uninstall-watchdog.ps1 (1496b), uninstall-watchdog.sh (2009b), _meta.json (143b)\n\nFile v1.4.4:SKILL.md\n\n---\nname: gateway-resilience-guard\ndescription: OpenClaw Gateway Resilience Guard keeps Gateway, channels, logs, and optional model-provider probes observable after Wi-Fi changes, WSL sleep/resume, session expiry, provider timeouts, or partial outages. It adds a localhost dashboard with charts, event trends, language/theme switching, strategy presets, guarded action unlock, native OpenClaw health probes, restart safeguards, systemd, LaunchAgent, Task Scheduler, and an optional OpenClaw plugin bridge.\ntags:\n  - openclaw\n  - gateway\n  - watchdog\n  - resilience\n  - wechat\n  - wsl\n  - systemd\n  - macos\n  - windows\nrequirements:\n  tools:\n    - bash\n    - curl\n    - systemctl optional\n    - launchctl optional\n    - powershell optional\n    - openclaw recommended\npermissions:\n  - Writes a user-level systemd service, macOS LaunchAgent, or Windows scheduled task when requested.\n  - Writes config and logs under user-level config/state directories.\n  - Starts a localhost dashboard on 127.0.0.1:18790 when Python is available.\n  - Restarts OpenClaw Gateway through systemctl --user or openclaw gateway restart.\n  - Optional model probe sends real OpenClaw model requests only when explicitly enabled.\n---\n\n# OpenClaw Gateway Resilience Guard\n\nUse this skill when a user wants to keep OpenClaw Gateway and message channels online after network drops, WSL sleep/resume, macOS/Windows wake events, long-lived connection failures, or recurring model-provider timeouts.\n\n## Install\n\nLinux, WSL, or macOS:\n\n```bash\nbash install-watchdog.sh\n```\n\nThe installer works with defaults. It prompts for the main channel probe URL when running interactively, but pressing Enter is enough for the default WeChat probe.\n\nFor unattended install:\n\n```bash\nbash install-watchdog.sh --yes\n```\n\nFor a custom channel:\n\n```bash\nbash install-watchdog.sh --channel-url \"https://your-channel.example.com/health\"\n```\n\nWindows PowerShell:\n\n```powershell\npowershell -ExecutionPolicy Bypass -File .\\install-watchdog.ps1\n```\n\nAfter install, open the standalone dashboard:\n\n```text\nhttp://127.0.0.1:18790/\n```\n\nThe dashboard is served by the watchdog process, not Gateway, so it remains the recovery entry when Gateway is down.\n\nOptional Gateway-side bridge while Gateway is healthy:\n\n```bash\nopenclaw plugins install ./openclaw-plugin\nopenclaw plugins enable resilience-guard\nopenclaw gateway restart\n```\n\n## Operate\n\n```bash\nsystemctl --user status gateway-watchdog\njournalctl --user -u gateway-watchdog -f\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh\n```\n\nOn macOS, inspect `launchctl print gui/$(id -u)/ai.clawhub.gateway-resilience-guard`.\nOn Windows, inspect `Get-ScheduledTask -TaskName \"OpenClaw Gateway Resilience Guard\"`.\n\nIf user systemd is unavailable, the installer starts a direct background fallback and stores its pid under `~/.local/state/openclaw-gateway-watchdog/watchdog.pid`.\n\n## Safety Model\n\nThe watchdog restarts only after layered checks:\n\n1. OpenClaw native health probes: `openclaw gateway status --require-rpc`, `openclaw health --json --verbose`, and `openclaw status --deep` when available.\n2. Runtime diagnostics: `openclaw models status` plus `openclaw logs --plain` signal scanning for provider timeout, proxy/network, rate limit, auth, channel session, gateway degraded, config reload, and task runtime warnings.\n3. Local gateway health URL/TCP and main channel URL fallbacks.\n4. General network URLs to avoid restarting during whole-machine network failure.\n5. Optional model-provider probe via `openclaw agent --json`; default action is evidence logging only.\n6. Dashboard action token, backoff, and hourly restart limits to avoid restart storms.\n\nModel probing is disabled by default because it consumes real provider quota. To diagnose provider timeouts, set `MODEL_PROBE_ENABLED=1`, keep `MODEL_PROBE_ACTION=log`, and inspect `model-probe-history.jsonl` the next day.\nDiagnostic log scanning is enabled by default, but `OPENCLAW_DIAG_ACTION=log` keeps it non-invasive unless the operator explicitly opts into a restart or custom command.\nUse the dashboard strategy buttons for common modes: observe, overnight diagnosis, channel recovery, and conservative circuit breaker.\n\nTell users to review `~/.config/openclaw-gateway-watchdog/watchdog.env` before publishing, sharing logs, or reporting issues.\n\nFile v1.4.4:README.md\n\n# OpenClaw Gateway Resilience Guard\n\nExternal recovery guard for OpenClaw Gateway and long-lived channel plugins such as `openclaw-weixin`.\n\nThis project is for people who run OpenClaw continuously on Linux, WSL, macOS, or Windows and need the gateway to recover from channel disconnects, network sleep/resume, and long-lived session failures without babysitting the terminal.\n\n## Problem\n\nOpenClaw Gateway can still be alive while an individual channel is no longer healthy. This is common after laptop sleep, Wi-Fi changes, WSL network hiccups, or long idle periods.\n\nThe WeChat plugin is especially sensitive because it depends on a long-poll `getUpdates` loop. In the upstream `Tencent/openclaw-weixin` code, the monitor has a limited retry loop and session guard:\n\n- `monitor.ts` defines `MAX_CONSECUTIVE_FAILURES = 3` and `BACKOFF_DELAY_MS = 30_000`.\n- `session-guard.ts` defines `SESSION_PAUSE_DURATION_MS = 60 * 60 * 1000` and `SESSION_EXPIRED_ERRCODE = -14`.\n- Issue [Tencent/openclaw-weixin#141](https://github.com/Tencent/openclaw-weixin/issues/141) reports that after a config hot reload the monitor can end without starting again; the workaround is `openclaw gateway restart`.\n- Issue [Tencent/openclaw-weixin#155](https://github.com/Tencent/openclaw-weixin/issues/155) reports that `errcode=-14` can enter a 60-minute session pause loop and block outbound messages.\n\nThis watchdog does not replace the official plugin. It is an external safety net: when the gateway or channel stops behaving like a live system, it restarts the gateway with guardrails.\n\n## Design\n\nThe script uses a layered health model before it restarts anything:\n\n| Layer | Probe | Purpose |\n| --- | --- | --- |\n| Gateway | `openclaw gateway status --json --require-rpc`, local health URL, local TCP port, service/process fallback | Detect whether OpenClaw Gateway is down or locally unreachable. |\n| Channel | `openclaw health --json --verbose`, `openclaw status --deep`, optional `openclaw channels status --probe`, then URL fallback | Prefer OpenClaw's own per-channel health model, then fall back to a configured URL when native probes are unavailable. |\n| Runtime diagnostics | `openclaw models status --json`, `openclaw logs --plain`, warning classification | Separate provider, proxy/network, auth/rate-limit, gateway, channel-session, config reload, and task-runtime evidence before choosing an action. |\n| Model API, optional | `openclaw agent --json` with the configured model provider | Detect whether the configured model path is timing out while Gateway and channels still look healthy. Disabled by default because it makes real model calls. |\n| Network | multiple independent URLs, default Baidu/QQ/Weixin | Avoid restarting the gateway during whole-machine or ISP network failure. |\n\nOnly gateway failures restart immediately. Channel failures go through confirmation, network split-brain protection, exponential backoff, and restart-rate limits.\nModel probe failures default to evidence logging only; users can opt in to restart or a custom command after consecutive failures.\n\n## Recovery Policy\n\n- Gateway down: restart immediately.\n- Gateway running but channel probe fails: confirm with general network probes.\n- General network also fails: do nothing except wait; restarting will not fix an offline machine.\n- General network works but channel stays down: wait with exponential backoff, re-check, then restart.\n- OpenClaw warning logs: classify and record evidence first; default action is log-only.\n- Five consecutive successful channel probes reset the failure state.\n- Restart storm protection limits gateway restarts per hour.\n- Night hours can use a slower probe interval to reduce noise.\n\n## Files\n\n| File | Purpose |\n| --- | --- |\n| `gateway-watchdog.sh` | Main daemon loop: probes, backoff, restart decisions, log rotation, single-instance lock. |\n| `gateway-watchdog.ps1` | Windows-native daemon loop for Task Scheduler. |\n| `dashboard/` | Standalone local Web UI and API served by the watchdog, independent of Gateway. |\n| `openclaw-plugin/` | Optional native OpenClaw plugin bridge that redirects `/resilience-guard` to the standalone dashboard. |\n| `install-watchdog.sh` | Linux/WSL/macOS installer: copies files, writes config, creates systemd user service or macOS LaunchAgent. |\n| `install-watchdog.ps1` | Windows installer: creates config and a Task Scheduler job. |\n| `uninstall-watchdog.sh` | Stops and removes the service and installed scripts. |\n| `uninstall-watchdog.ps1` | Windows uninstaller. |\n| `SKILL.md` | ClawHub/OpenClaw skill metadata and operator instructions. |\n| `README.zh-CN.md` | Chinese documentation. |\n\n## Install\n\nInstall from ClawHub:\n\n```bash\nclawhub install gateway-resilience-guard\n```\n\nOr use this repository directly.\n\nLinux, WSL, or macOS:\n\n```bash\nbash install-watchdog.sh\n```\n\nFor unattended install:\n\n```bash\nbash install-watchdog.sh --yes\n```\n\nFor a custom channel probe:\n\n```bash\nbash install-watchdog.sh --channel-url \"https://your-channel.example.com/health\"\n```\n\nWindows PowerShell:\n\n```powershell\npowershell -ExecutionPolicy Bypass -File .\\install-watchdog.ps1\n```\n\n## Dashboard\n\nThe installer enables a standalone local dashboard:\n\n```text\nhttp://127.0.0.1:18790/\n```\n\nThis dashboard is served by the watchdog, not by OpenClaw Gateway. If Gateway is down, the dashboard can still open and show the last known evidence.\n\nIt includes:\n\n- Gateway, channel, network, OpenClaw log, and model-provider status.\n- Category charts for provider timeout, proxy/network, rate-limit, auth, channel session, Gateway degraded, config reload, and task runtime warnings.\n- Sidebar view switching for Overview, Trends, Strategy, Logs, and Config so monitoring, actions, and detail evidence are separated.\n- Local design-system controls: native semantic selects, buttons, badges, and guarded controls styled consistently without relying on CDN or component-script upgrades.\n- Denser page composition: trend digest and latest signals, strategy matrix and manual actions, config runtime map.\n- Event trend chart with separate lanes for API failures, log warnings, successful model probes, and healthy diagnostics.\n- Chinese/English language switching and Light, Dark, Ocean, and Forest themes.\n- Status-file freshness checks so stale data is obvious.\n- Quick strategy buttons: observe, overnight diagnosis, channel recovery, and conservative circuit breaker.\n- Guarded actions with an unlock flow: run diagnostics, restart Gateway, apply presets, and export a diagnostic JSON bundle.\n\nDashboard actions are bound to localhost and protected with the generated `DASHBOARD_TOKEN`. The token is injected only into the same-origin dashboard page. A random token is written during install; the value is not shown in logs.\n\nOptional OpenClaw plugin bridge:\n\n```bash\nopenclaw plugins install ./openclaw-plugin\nopenclaw plugins enable resilience-guard\nopenclaw gateway restart\n```\n\nThen open the Gateway route while Gateway is healthy:\n\n```text\nhttp://127.0.0.1:18789/resilience-guard\n```\n\nThat route redirects to the external dashboard. It is a convenience entry only; the external dashboard remains the recovery entry when Gateway is unavailable.\n\nThe installer writes:\n\n- scripts to `~/.local/share/openclaw-gateway-watchdog`;\n- config to `~/.config/openclaw-gateway-watchdog/watchdog.env`;\n- logs to `~/.local/state/openclaw-gateway-watchdog/watchdog.log`;\n- a Linux/WSL user service to `~/.config/systemd/user/gateway-watchdog.service`;\n- a macOS LaunchAgent to `~/Library/LaunchAgents/ai.clawhub.gateway-resilience-guard.plist`;\n- a Windows scheduled task named `OpenClaw Gateway Resilience Guard`.\n\nIf user systemd is unavailable, the installer starts a direct background fallback process and stores its pid in the state directory.\n\n## Manage\n\n```bash\nsystemctl --user status gateway-watchdog\njournalctl --user -u gateway-watchdog -f\nsystemctl --user restart gateway-watchdog\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh\n```\n\nmacOS:\n\n```bash\nlaunchctl print gui/$(id -u)/ai.clawhub.gateway-resilience-guard\ntail -f ~/.local/state/openclaw-gateway-watchdog/watchdog.log\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh\n```\n\nWindows:\n\n```powershell\nGet-ScheduledTask -TaskName \"OpenClaw Gateway Resilience Guard\"\nGet-Content \"$env:LOCALAPPDATA\\openclaw-gateway-watchdog\\watchdog.log\" -Wait\npowershell -ExecutionPolicy Bypass -File \"$env:LOCALAPPDATA\\openclaw-gateway-watchdog\\uninstall-watchdog.ps1\"\n```\n\nRemove config and logs too:\n\n```bash\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh --purge\n```\n\n## Configuration\n\nMost users can keep the generated defaults. Advanced settings live in:\n\n```text\n~/.config/openclaw-gateway-watchdog/watchdog.env\n```\n\nCommon keys:\n\n```bash\nGATEWAY_SERVICE=\"openclaw-gateway\"\nGATEWAY_HEALTH_URL=\"http://127.0.0.1:18789/healthz\"\nGATEWAY_HOST=\"127.0.0.1\"\nGATEWAY_PORT=\"18789\"\nCHANNEL_URL=\"https://ilinkai.weixin.qq.com\"\nNETWORK_URLS=\"https://www.baidu.com https://www.qq.com https://api.weixin.qq.com\"\nRESTART_COMMAND=\"systemctl --user restart openclaw-gateway\"\nOPENCLAW_NATIVE_PROBES=\"auto\"\nOPENCLAW_HEALTH_TIMEOUT_MS=\"12000\"\nOPENCLAW_GATEWAY_STRICT=\"0\"\nOPENCLAW_CHANNELS_PROBE=\"1\"\nOPENCLAW_DIAG_ENABLED=\"1\"\nOPENCLAW_DIAG_INTERVAL=\"300\"\nOPENCLAW_LOG_SCAN_ENABLED=\"1\"\nOPENCLAW_LOG_LIMIT=\"200\"\nOPENCLAW_LOG_SIGNAL_LIMIT=\"40\"\nOPENCLAW_LOG_TIMEOUT_MS=\"15000\"\nOPENCLAW_LOG_WARN_PATTERNS=\"fetch failed|fetch timeout|LLM idle timeout|model silent|...\"\nOPENCLAW_DIAG_ACTION=\"log\"\nOPENCLAW_DIAG_FAILURES_BEFORE_ACTION=\"2\"\nOPENCLAW_DIAG_COMMAND=\"\"\nDASHBOARD_ENABLED=\"1\"\nDASHBOARD_HOST=\"127.0.0.1\"\nDASHBOARD_PORT=\"18790\"\nDASHBOARD_ACTIONS_ENABLED=\"1\"\nDASHBOARD_TOKEN=\"generated-at-install\"\nDASHBOARD_DIR=\"~/.local/share/openclaw-gateway-watchdog/dashboard\"\nMODEL_PROBE_ENABLED=\"0\"\nMODEL_EDGE_PROBE_ENABLED=\"1\"\nMODEL_PROBE_INTERVAL=\"1800\"\nMODEL_PROBE_TIMEOUT=\"120\"\nMODEL_PROBE_FAILURES_BEFORE_ACTION=\"2\"\nMODEL_PROBE_ACTION=\"log\"\nMODEL_PROBE_COMMAND=\"\"\nMODEL_PROBE_MODEL=\"\"\nMODEL_PROBE_THINKING=\"off\"\nMODEL_PROBE_SESSION_ID=\"watchdog-model-probe\"\nMODEL_PROBE_MESSAGE=\"Reply with exactly OK.\"\nBASE_INTERVAL=\"60\"\nNIGHT_INTERVAL=\"300\"\nMAX_INTERVAL=\"1800\"\nCHANNEL_FAILURES_BEFORE_RESTART=\"2\"\nSUCCESS_COUNT_TO_RESET=\"5\"\nMAX_RESTARTS_PER_HOUR=\"6\"\n```\n\nUse `RESTART_COMMAND` if your OpenClaw install is not managed by a user-level systemd unit. Example:\n\n```bash\nRESTART_COMMAND=\"openclaw gateway restart\"\n```\n\nOn Windows, the generated config is JSON:\n\n```text\n%APPDATA%\\openclaw-gateway-watchdog\\watchdog.json\n```\n\nSet `OpenClawNativeProbes` to `false` if your OpenClaw CLI is too old for `openclaw health` or `openclaw status --deep`.\n\n### OpenClaw diagnostics and log signals\n\nThe watchdog does more than ping one URL. Every `OPENCLAW_DIAG_INTERVAL` seconds it collects an OpenClaw diagnostic snapshot:\n\n```text\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-gateway-status.json\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-health.json\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-model-status.json\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-status-deep.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-logs.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-log-signals.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-log-signal-categories.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-diagnostics.jsonl\n```\n\n`last-openclaw-log-signals.txt` is filtered from `openclaw logs --plain`. It classifies common failure families such as `provider_timeout`, `proxy_or_network`, `provider_rate_limit`, `provider_auth`, `abort_stuck`, `memory_dream_timeout`, `channel_session`, `gateway_degraded`, `config_reload`, and `task_runtime`.\n\nDefault action is `OPENCLAW_DIAG_ACTION=\"log\"` because warnings are evidence, not always proof that a restart is correct. If you explicitly set `OPENCLAW_DIAG_ACTION=\"restart\"` or `command`, the action only runs after consecutive diagnostic warnings and only when the general network probes still pass.\n\nPractical interpretation:\n\n| Evidence | Likely scope | Default strategy |\n| --- | --- | --- |\n| Gateway status/health is down | Gateway process or RPC path | Restart Gateway immediately through the normal gateway policy. |\n| Channel probe fails, network probes pass | Channel/session path | Backoff, re-check, then restart Gateway if still failed. |\n| OpenClaw logs show provider timeout, model probe also fails | Provider/API path | Log evidence; optional custom action. A Gateway restart may not fix provider outage. |\n| OpenClaw logs show provider timeout, model probe succeeds | OpenClaw runtime, task, session, or specific request path | Keep evidence, inspect logs; avoid blaming the provider alone. |\n| Logs show proxy/DNS/TLS errors | Local proxy, DNS, TLS, or ISP route | Log evidence and avoid restart storms; fix network/proxy route first. |\n| Logs show session expiry or monitor stopped | Channel plugin/session | Gateway restart is often useful after confirmation. |\n\n### Optional model probe\n\nSet `MODEL_PROBE_ENABLED=\"1\"` when you need to prove whether failures are in the model-provider path instead of Gateway or channel health.\n\nWhen enabled, the watchdog first reads OpenClaw's configured model provider and probes its provider `baseUrl` without credentials. Then it runs the end-to-end model probe:\n\n```bash\nopenclaw agent --session-id \"$MODEL_PROBE_SESSION_ID\" \\\n  --thinking \"$MODEL_PROBE_THINKING\" \\\n  --timeout \"$MODEL_PROBE_TIMEOUT\" \\\n  --json \\\n  --message \"$MODEL_PROBE_MESSAGE\"\n```\n\nIf `MODEL_PROBE_MODEL` is empty, OpenClaw's configured default model is used. Results are written to the main log and to:\n\n```text\n~/.local/state/openclaw-gateway-watchdog/model-probe-history.jsonl\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-model-probe.json\n~/.local/state/openclaw-gateway-watchdog/last-model-api-edge-probe.txt\n```\n\nRecommended diagnostic settings for overnight provider issues:\n\n```bash\nMODEL_PROBE_ENABLED=\"1\"\nMODEL_EDGE_PROBE_ENABLED=\"1\"\nMODEL_PROBE_INTERVAL=\"600\"\nMODEL_PROBE_TIMEOUT=\"120\"\nMODEL_PROBE_ACTION=\"log\"\nMODEL_PROBE_THINKING=\"off\"\nMODEL_PROBE_MESSAGE=\"Reply with exactly OK.\"\n```\n\n`MODEL_PROBE_ACTION` can be:\n\n- `log`: record evidence only. This is the default.\n- `restart`: restart Gateway after `MODEL_PROBE_FAILURES_BEFORE_ACTION` consecutive model failures, but only if general network probes still pass.\n- `command`: run `MODEL_PROBE_COMMAND` after consecutive failures.\n\nThis feature sends real model requests and may consume provider quota or money. It does not print API keys, but it does store provider/model names, timing, exit status, and the first non-empty error line.\n`MODEL_EDGE_PROBE_ENABLED` does not use credentials and does not call `/chat/completions`; it only checks whether the provider API edge such as `https://api.deepseek.com` is reachable quickly.\n\n## Safety Notes\n\nThis project intentionally avoids destructive behavior. It does not edit OpenClaw configuration, tokens, sessions, or plugin files. By default it only probes URLs and restarts the gateway through the configured command. The optional model probe makes real model calls only after you explicitly enable it.\n\nBefore sharing logs, review them for local paths, service names, and channel URLs.\n\n## License\n\nMIT-0. This matches ClawHub's skill publishing requirement and allows reuse without attribution requirements.\n\n## Publish To ClawHub\n\nPublished package:\n\n- ClawHub: <https://clawhub.ai/zc-kama/gateway-resilience-guard>\n- Slug: `gateway-resilience-guard`\n\nThis repository includes `SKILL.md`, so it can also be republished as an OpenClaw skill bundle:\n\n```bash\nclawhub publish . \\\n  --slug gateway-resilience-guard \\\n  --name \"OpenClaw Gateway Resilience Guard\" \\\n  --version 1.4.4 \\\n  --changelog \"Make dashboard controls robust with native semantic buttons/selects, fixed hover states, centered brand mark, and non-overlapping config rows\"\n```\n\nClawHub requires CLI authentication. Run `clawhub login` first.\n\nFile v1.4.4:_meta.json\n\n{\n  \"ownerId\": \"kn70e6hn9pv8yrpzjn5kqczygd8708vw\",\n  \"slug\": \"gateway-resilience-guard\",\n  \"version\": \"1.4.4\",\n  \"publishedAt\": 1779383203702\n}\n\nFile v1.4.4:CHANGELOG.md\n\n# Changelog\n\n## 1.4.4\n\n- Replace fragile Web Component controls with semantic native buttons and selects styled by the local dashboard design system.\n- Fix the top action bar, sidebar navigation, refresh/export controls, and strategy buttons so they always render as clickable UI even if component scripts are blocked or stale.\n- Center the brand mark text and restore pointer cursor, hover, focus, and active states across navigation and action controls.\n- Fix config summary row layout so long environment keys no longer overlap values.\n\n## 1.4.3\n\n- Rebuild the dashboard visual system on local Fluent UI Web Components instead of plain native controls.\n- Replace the old dropdowns and buttons with componentized controls, dark glass panels, animated ambient background, and refined hover states.\n- Rebalance every dashboard page so Overview, Trends, Strategy, Logs, and Config each have complete work areas rather than sparse split-up cards.\n- Add richer summaries for trend digest, latest signals, strategy matrix, action center, and runtime map.\n- Serve bundled local component assets under `/vendor/` so the recovery UI does not depend on CDN availability.\n\n## 1.4.2\n\n- Rework the dashboard sidebar into real view switching for Overview, Trends, Strategy, Logs, and Config instead of one long page.\n- Tighten the dashboard grid, panel sizing, button alignment, and sidebar polish for a cleaner operations-console layout.\n- Fix the trend chart axis label collision by removing the redundant x-axis title and reserving more space for tick labels.\n- Keep charts stable across refreshes by drawing only the active view and ignoring hidden canvases.\n- Move strategy explanations into centered button content and remove the sidebar localhost note.\n\n## 1.4.1\n\n- Redesign the dashboard with a sidebar layout, stronger visual grouping, and selectable Light, Dark, Ocean, and Forest themes.\n- Add Chinese/English language switching for dashboard labels, buttons, hints, legends, and notifications.\n- Fix canvas redraw sizing so repeated refreshes do not stretch dashboard panels.\n- Replace \"overnight timeline\" with a general event trend chart that includes lanes, legends, and time-axis ticks.\n- Auto-follow the newest watchdog log lines while keeping a toggle for manual scrolling.\n- Add an unlock/lock action flow and strategy hover hints so guarded controls are discoverable.\n\n## 1.4.0\n\n- Add a standalone local dashboard at `http://127.0.0.1:18790` that remains available when OpenClaw Gateway is down.\n- Add dashboard API summaries, log-signal charts, overnight timelines, status file freshness, safe config summaries, diagnostic export, and quick strategy presets.\n- Add guarded dashboard actions for run diagnostics, restart Gateway, and apply presets using a generated local action token.\n- Add an OpenClaw native plugin bridge that registers `/resilience-guard` and redirects to the external dashboard when Gateway is healthy.\n- Install dashboard files on Linux/WSL, macOS, and Windows.\n\n## 1.3.1\n\n- Publish a ClawHub package that includes the Windows PowerShell installer, watchdog, and uninstaller files.\n\n## 1.3.0\n\n- Add an opt-in model-provider probe that uses `openclaw agent --json` against the configured OpenClaw model path.\n- Add a no-credential provider edge probe that checks the configured provider `baseUrl` before the end-to-end model call.\n- Add OpenClaw runtime diagnostics that snapshot gateway status, health, model status, deep status, and recent OpenClaw logs.\n- Classify log signals for provider timeout, proxy/network, rate limit, auth, channel session, gateway degraded, config reload, task runtime, and related warning families.\n- Record model probe evidence in the main log and `model-probe-history.jsonl` without logging API keys.\n- Add configurable model failure actions: log-only, gateway restart, or a custom command after consecutive failures.\n- Document safe overnight diagnostics for separating Gateway/channel failures from model-provider timeouts.\n\n## 1.2.0\n\n- Add OpenClaw-native health probing via `openclaw gateway status --json --require-rpc`, `openclaw health --json --verbose`, and `openclaw status --deep`.\n- Add Windows Task Scheduler support with native PowerShell install, watchdog, and uninstall scripts.\n- Add macOS LaunchAgent support to the Bash installer and uninstaller.\n- Document cross-platform installation and native probe configuration.\n\n## 1.1.0\n\n- Rename the ClawHub package to `gateway-resilience-guard`.\n- Expand the public summary to describe the layered probes, restart guardrails, and advantages over simple restart loops or one-URL monitors.\n\n## 1.0.1\n\n- Add the published ClawHub URL and install command to the README files.\n\n## 1.0.0\n\n- First public zero-config release.\n- Install from the current folder instead of a hard-coded workspace path.\n- Generate user config and systemd service automatically.\n- Add layered gateway/channel/network probes.\n- Add exponential backoff, hourly restart limits, lock protection, and log rotation.\n- Reset failure state after five consecutive successful channel probes.\n- Add ClawHub-ready `SKILL.md`.\n- Initial ClawHub publication used a temporary slug before the 1.1.0 rename.\n\nFile v1.4.4:README-watchdog.md\n\n# OpenClaw Gateway 看门狗\n\n这是项目的中文快捷入口。完整中文文档见 [README.zh-CN.md](README.zh-CN.md)，英文文档见 [README.md](README.md)。\n\n最简单安装：\n\n```bash\nbash install-watchdog.sh\n```\n\nWindows：\n\n```powershell\npowershell -ExecutionPolicy Bypass -File .\\install-watchdog.ps1\n```\n\n默认配置即可启动；需要指定自己的通道地址时：\n\n```bash\nbash install-watchdog.sh --channel-url \"https://你的通道地址/health\"\n```\n\n安装后还有一个独立图形化控制台：\n\n```text\nhttp://127.0.0.1:18790/\n```\n\n它由 watchdog 自己提供，不依赖 OpenClaw Gateway。Gateway 挂掉时，这个页面仍然可以看状态、图表、策略、日志和诊断导出。Dashboard 支持中英文切换、主题切换、事件趋势图、异常分类图和受保护操作解锁。\n\n安装后它会默认采集 OpenClaw 诊断快照和日志信号，包括 gateway status、health、models status、status --deep 和 `openclaw logs --plain` 的 WARN/ERROR 摘要。关键证据在：\n\n```text\n~/.local/state/openclaw-gateway-watchdog/watchdog.log\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-diagnostics.jsonl\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-log-signals.txt\n```\n\n可选模型探针默认关闭。需要排查模型 API 是否在某个时段超时时，可以在配置里显式开启 `MODEL_PROBE_ENABLED=\"1\"`；它会发送极小的 `openclaw agent --json` 请求，并把结果写入 watchdog 日志和 `model-probe-history.jsonl`。默认动作仍是只记录证据，不会擅自改 OpenClaw 配置。\n\nFile v1.4.4:README.zh-CN.md\n\n# OpenClaw Gateway Resilience Guard\n\nOpenClaw Gateway 外部恢复守护脚本，面向 `openclaw-weixin` 这类需要长期在线的通道。\n\n这个项目解决的是一个很具体的运维问题：OpenClaw Gateway 进程还活着，但通道、长轮询、WSL/系统网络或 session 状态已经坏掉，导致消息收不到、发不出，最后只能手动 `openclaw gateway restart`。\n\n## 问题背景\n\nOpenClaw Gateway 和通道插件是长连接系统。电脑睡眠、切换 Wi-Fi、WSL 网络重建、iLink 长时间空闲、配置热加载，都可能让“进程存活”和“通道可用”变成两件事。\n\n以官方 `Tencent/openclaw-weixin` 为例，源码和 issue 里能看到几个已知边界：\n\n- `monitor.ts` 里 `MAX_CONSECUTIVE_FAILURES = 3`，失败后 `BACKOFF_DELAY_MS = 30_000`，也就是插件内部主要是 3 次失败后的 30 秒退避。\n- `session-guard.ts` 里 `SESSION_PAUSE_DURATION_MS = 60 * 60 * 1000`，`SESSION_EXPIRED_ERRCODE = -14`，session 过期会进入 60 分钟暂停窗口。\n- [Tencent/openclaw-weixin#141](https://github.com/Tencent/openclaw-weixin/issues/141) 记录了配置热加载后 Monitor 结束但不再启动，临时处理方式是手动重启 gateway。\n- [Tencent/openclaw-weixin#155](https://github.com/Tencent/openclaw-weixin/issues/155) 记录了 `errcode=-14` 后进入 60 分钟循环暂停、出站消息被阻塞的问题。\n\n所以这个项目不是替换官方插件，而是在外面加一层独立 watchdog：当通道或 gateway 进入“看起来还活着，实际上已经不能工作”的状态时，用更保守的探测和熔断策略自动恢复。\n\n## 工作原理\n\n看门狗使用分层健康检查，从浅到深判断是否真的需要重启：\n\n| 层级 | 检查对象 | 作用 |\n| --- | --- | --- |\n| Gateway 本机状态 | `openclaw gateway status --json --require-rpc`、本机 health URL、本机 TCP 端口、服务/进程兜底 | 判断 OpenClaw Gateway 是否已经挂掉或本机不可达。 |\n| 通道状态 | `openclaw health --json --verbose`、`openclaw status --deep`、可选 `openclaw channels status --probe`，最后才是 URL 兜底 | 优先使用 OpenClaw 自己的全通道健康模型；CLI 不支持时再退回 URL 探测。 |\n| 运行时诊断 | `openclaw models status --json`、`openclaw logs --plain`、WARN/ERROR 分类 | 在动作前区分 provider、代理/网络、鉴权/限流、gateway、通道 session、配置热加载和任务运行时证据。 |\n| 模型 API，可选 | 使用 `openclaw agent --json` 走 OpenClaw 当前配置的模型 provider | 判断 Gateway 和通道都健康时，真正卡住的是不是模型 API 链路。默认关闭，因为它会真实消耗模型调用。 |\n| 外部网络状态 | 百度、QQ、微信 API 等多个独立 URL | 排除全局断网，避免电脑没网时误重启 gateway。 |\n\n核心策略是：gateway 真挂了就立即重启；通道不通时先确认不是全局断网；网络正常但通道持续失败，才进入退避等待和重启流程。\n模型探针默认只记录证据日志；如果你显式配置，也可以在连续失败后重启 gateway 或执行自定义命令。\n\n## 恢复策略\n\n- Gateway 本机健康检查失败：立即重启。\n- 通道 URL 失败：累计失败次数，进入故障流程。\n- 外部网络全部失败：认为是全局断网，只等待，不重启。\n- 外部网络正常但通道仍失败：指数退避后再次确认，再重启 gateway。\n- OpenClaw 日志出现 WARN/ERROR：先分类和记录证据，默认只写日志。\n- 连续 5 次通道探测成功后，清空失败状态。\n- 每小时最多重启固定次数，防止网络抖动时形成重启风暴。\n- 深夜可以降低检查频率，减少无意义日志。\n\n## 文件说明\n\n| 文件 | 作用 |\n| --- | --- |\n| `gateway-watchdog.sh` | 主守护脚本，负责探测、退避、熔断、日志、单实例锁和重启决策。 |\n| `gateway-watchdog.ps1` | Windows 原生守护脚本，用于 Task Scheduler。 |\n| `dashboard/` | 独立本地 Web UI 和 API，由 watchdog 自己提供，不依赖 Gateway。 |\n| `openclaw-plugin/` | 可选 OpenClaw 原生插件桥接入口，把 `/resilience-guard` 跳转到独立 dashboard。 |\n| `install-watchdog.sh` | Linux/WSL/macOS 安装脚本，自动复制文件、生成配置、创建 systemd 用户服务或 macOS LaunchAgent。 |\n| `install-watchdog.ps1` | Windows 安装脚本，创建配置和计划任务。 |\n| `uninstall-watchdog.sh` | 卸载脚本，停止服务并删除安装目录。 |\n| `uninstall-watchdog.ps1` | Windows 卸载脚本。 |\n| `SKILL.md` | ClawHub/OpenClaw 技能元数据和使用说明。 |\n| `README.md` | 英文文档。 |\n\n## 安装\n\n通过 ClawHub 安装：\n\n```bash\nclawhub install gateway-resilience-guard\n```\n\n也可以直接使用本仓库。\n\nLinux、WSL 或 macOS：\n\n```bash\nbash install-watchdog.sh\n```\n\n无人值守安装：\n\n```bash\nbash install-watchdog.sh --yes\n```\n\n指定自己的通道探测地址：\n\n```bash\nbash install-watchdog.sh --channel-url \"https://你的通道地址/health\"\n```\n\nWindows PowerShell：\n\n```powershell\npowershell -ExecutionPolicy Bypass -File .\\install-watchdog.ps1\n```\n\n## 图形化 Dashboard\n\n安装后会启用一个独立本地 dashboard：\n\n```text\nhttp://127.0.0.1:18790/\n```\n\n这个页面由 watchdog 自己提供，不依赖 OpenClaw Gateway。所以 Gateway 挂掉时，它仍然可以打开，看到最后一次诊断证据。\n\n它包含：\n\n- Gateway、通道、外部网络、OpenClaw 日志、模型 provider 的分层状态。\n- provider timeout、代理/网络、限流、鉴权、通道 session、Gateway degraded、配置热加载、任务运行时异常的分类图表。\n- 侧边栏分页：总览、趋势、策略、日志、配置分开呈现，避免把监控、操作和明细堆在一个长页面里。\n- 本地设计系统控件：下拉框、按钮、状态标签使用原生语义控件加统一样式，不依赖 CDN 或组件脚本升级。\n- 重新分配每页信息密度：趋势页有摘要和最新信号，策略页有策略矩阵和手动操作，配置页有运行地图。\n- 事件趋势图，把 API 失败、日志 WARN、模型探针成功、健康诊断分成不同泳道，并保留清晰的时间刻度。\n- 中英文语言切换，以及 Light、Dark、Ocean、Forest 四套主题。\n- 状态文件新鲜度，避免把过期快照误认为当前状态。\n- 快速策略按钮：观察模式、夜间诊断、通道恢复、保守熔断。\n- 带解锁流程的受保护操作：立即诊断、重启 Gateway、应用策略、导出诊断 JSON。\n\nDashboard 操作只绑定 localhost，并使用安装时生成的 `DASHBOARD_TOKEN` 保护。token 只注入同源页面，不会写入日志。\n\n可选 OpenClaw 插件桥接入口：\n\n```bash\nopenclaw plugins install ./openclaw-plugin\nopenclaw plugins enable resilience-guard\nopenclaw gateway restart\n```\n\nGateway 正常时可以打开：\n\n```text\nhttp://127.0.0.1:18789/resilience-guard\n```\n\n这个路由会跳转到外部 dashboard。它只是方便入口；真正救急的入口仍然是 `http://127.0.0.1:18790/`。\n\n安装后会生成：\n\n- 脚本目录：`~/.local/share/openclaw-gateway-watchdog`\n- 配置文件：`~/.config/openclaw-gateway-watchdog/watchdog.env`\n- 日志文件：`~/.local/state/openclaw-gateway-watchdog/watchdog.log`\n- Linux/WSL systemd 用户服务：`~/.config/systemd/user/gateway-watchdog.service`\n- macOS LaunchAgent：`~/Library/LaunchAgents/ai.clawhub.gateway-resilience-guard.plist`\n- Windows 计划任务：`OpenClaw Gateway Resilience Guard`\n\n如果当前环境没有 user systemd，安装脚本会退回到后台进程模式，并把 pid 写到状态目录。\n\n## 管理命令\n\n```bash\nsystemctl --user status gateway-watchdog\njournalctl --user -u gateway-watchdog -f\nsystemctl --user restart gateway-watchdog\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh\n```\n\nmacOS：\n\n```bash\nlaunchctl print gui/$(id -u)/ai.clawhub.gateway-resilience-guard\ntail -f ~/.local/state/openclaw-gateway-watchdog/watchdog.log\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh\n```\n\nWindows：\n\n```powershell\nGet-ScheduledTask -TaskName \"OpenClaw Gateway Resilience Guard\"\nGet-Content \"$env:LOCALAPPDATA\\openclaw-gateway-watchdog\\watchdog.log\" -Wait\npowershell -ExecutionPolicy Bypass -File \"$env:LOCALAPPDATA\\openclaw-gateway-watchdog\\uninstall-watchdog.ps1\"\n```\n\n连配置和日志一起删除：\n\n```bash\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh --purge\n```\n\n## 配置\n\n普通用户一般不用改。高级配置在：\n\n```text\n~/.config/openclaw-gateway-watchdog/watchdog.env\n```\n\n常用项：\n\n```bash\nGATEWAY_SERVICE=\"openclaw-gateway\"\nGATEWAY_HEALTH_URL=\"http://127.0.0.1:18789/healthz\"\nGATEWAY_HOST=\"127.0.0.1\"\nGATEWAY_PORT=\"18789\"\nCHANNEL_URL=\"https://ilinkai.weixin.qq.com\"\nNETWORK_URLS=\"https://www.baidu.com https://www.qq.com https://api.weixin.qq.com\"\nRESTART_COMMAND=\"systemctl --user restart openclaw-gateway\"\nOPENCLAW_NATIVE_PROBES=\"auto\"\nOPENCLAW_HEALTH_TIMEOUT_MS=\"12000\"\nOPENCLAW_GATEWAY_STRICT=\"0\"\nOPENCLAW_CHANNELS_PROBE=\"1\"\nOPENCLAW_DIAG_ENABLED=\"1\"\nOPENCLAW_DIAG_INTERVAL=\"300\"\nOPENCLAW_LOG_SCAN_ENABLED=\"1\"\nOPENCLAW_LOG_LIMIT=\"200\"\nOPENCLAW_LOG_SIGNAL_LIMIT=\"40\"\nOPENCLAW_LOG_TIMEOUT_MS=\"15000\"\nOPENCLAW_LOG_WARN_PATTERNS=\"fetch failed|fetch timeout|LLM idle timeout|model silent|...\"\nOPENCLAW_DIAG_ACTION=\"log\"\nOPENCLAW_DIAG_FAILURES_BEFORE_ACTION=\"2\"\nOPENCLAW_DIAG_COMMAND=\"\"\nDASHBOARD_ENABLED=\"1\"\nDASHBOARD_HOST=\"127.0.0.1\"\nDASHBOARD_PORT=\"18790\"\nDASHBOARD_ACTIONS_ENABLED=\"1\"\nDASHBOARD_TOKEN=\"安装时生成\"\nDASHBOARD_DIR=\"~/.local/share/openclaw-gateway-watchdog/dashboard\"\nMODEL_PROBE_ENABLED=\"0\"\nMODEL_EDGE_PROBE_ENABLED=\"1\"\nMODEL_PROBE_INTERVAL=\"1800\"\nMODEL_PROBE_TIMEOUT=\"120\"\nMODEL_PROBE_FAILURES_BEFORE_ACTION=\"2\"\nMODEL_PROBE_ACTION=\"log\"\nMODEL_PROBE_COMMAND=\"\"\nMODEL_PROBE_MODEL=\"\"\nMODEL_PROBE_THINKING=\"off\"\nMODEL_PROBE_SESSION_ID=\"watchdog-model-probe\"\nMODEL_PROBE_MESSAGE=\"Reply with exactly OK.\"\nBASE_INTERVAL=\"60\"\nNIGHT_INTERVAL=\"300\"\nMAX_INTERVAL=\"1800\"\nCHANNEL_FAILURES_BEFORE_RESTART=\"2\"\nSUCCESS_COUNT_TO_RESET=\"5\"\nMAX_RESTARTS_PER_HOUR=\"6\"\n```\n\n如果你的 OpenClaw 不是 systemd 用户服务管理，可以显式指定重启命令：\n\n```bash\nRESTART_COMMAND=\"openclaw gateway restart\"\n```\n\nWindows 的配置是 JSON：\n\n```text\n%APPDATA%\\openclaw-gateway-watchdog\\watchdog.json\n```\n\n如果你的 OpenClaw CLI 版本太旧，不支持 `openclaw health` 或 `openclaw status --deep`，可以把 `OpenClawNativeProbes` 设为 `false`。\n\n### OpenClaw 诊断和日志信号\n\n这个 watchdog 不只是 ping 一个 URL。它会每隔 `OPENCLAW_DIAG_INTERVAL` 秒采集一组 OpenClaw 运行快照：\n\n```text\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-gateway-status.json\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-health.json\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-model-status.json\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-status-deep.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-logs.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-log-signals.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-log-signal-categories.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-diagnostics.jsonl\n```\n\n`last-openclaw-log-signals.txt` 来自 `openclaw logs --plain` 的过滤结果，会把常见异常归类成 `provider_timeout`、`proxy_or_network`、`provider_rate_limit`、`provider_auth`、`abort_stuck`、`memory_dream_timeout`、`channel_session`、`gateway_degraded`、`config_reload`、`task_runtime` 等类别。\n\n默认策略是 `OPENCLAW_DIAG_ACTION=\"log\"`，因为 WARN 是证据，不一定等于“应该马上重启”。如果你显式改成 `restart` 或 `command`，也必须连续多次出现诊断异常，并且外部网络探测正常，才会执行动作。\n\n排查时可以这样判断：\n\n| 证据 | 更可能的问题范围 | 默认策略 |\n| --- | --- | --- |\n| Gateway status/health 挂了 | Gateway 进程或 RPC 链路 | 走原本的 Gateway 策略，立即重启。 |\n| 通道探测失败，但外部网络正常 | 通道或 session 链路 | 退避、复查，仍失败再重启 Gateway。 |\n| OpenClaw 日志显示 provider timeout，模型探针也失败 | 模型 provider/API 链路 | 先记录证据；可选自定义动作。单纯重启 Gateway 未必有用。 |\n| OpenClaw 日志显示 provider timeout，但模型探针成功 | OpenClaw 运行时、任务、session 或特定请求路径 | 继续保留证据，不把锅直接甩给 provider。 |\n| 日志显示 proxy/DNS/TLS 错误 | 本机代理、DNS、TLS 或运营商路由 | 记录证据，避免重启风暴，优先修代理/网络路由。 |\n| 日志显示 session expired 或 monitor stopped | 通道插件/session | 确认后重启 Gateway 往往有价值。 |\n\n### 可选模型探针\n\n如果你想判断问题到底出在 Gateway/通道，还是模型 provider 链路，可以显式开启：\n\n```bash\nMODEL_PROBE_ENABLED=\"1\"\n```\n\n开启后，看门狗会先读取 OpenClaw 当前模型 provider 的 `baseUrl`，做一次不带凭据、不消耗 token 的入口连通性探测。然后再执行端到端模型探针：\n\n```bash\nopenclaw agent --session-id \"$MODEL_PROBE_SESSION_ID\" \\\n  --thinking \"$MODEL_PROBE_THINKING\" \\\n  --timeout \"$MODEL_PROBE_TIMEOUT\" \\\n  --json \\\n  --message \"$MODEL_PROBE_MESSAGE\"\n```\n\n如果 `MODEL_PROBE_MODEL` 为空，就使用 OpenClaw 当前配置的默认模型。结果会写入主日志，以及：\n\n```text\n~/.local/state/openclaw-gateway-watchdog/model-probe-history.jsonl\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-model-probe.json\n~/.local/state/openclaw-gateway-watchdog/last-model-api-edge-probe.txt\n```\n\n排查凌晨模型 provider 超时，可以先用这组设置：\n\n```bash\nMODEL_PROBE_ENABLED=\"1\"\nMODEL_EDGE_PROBE_ENABLED=\"1\"\nMODEL_PROBE_INTERVAL=\"600\"\nMODEL_PROBE_TIMEOUT=\"120\"\nMODEL_PROBE_ACTION=\"log\"\nMODEL_PROBE_THINKING=\"off\"\nMODEL_PROBE_MESSAGE=\"Reply with exactly OK.\"\n```\n\n`MODEL_PROBE_ACTION` 支持：\n\n- `log`：只记录证据，默认策略。\n- `restart`：连续失败达到 `MODEL_PROBE_FAILURES_BEFORE_ACTION` 后重启 gateway，但会先确认外部网络不是全局断网。\n- `command`：连续失败后执行 `MODEL_PROBE_COMMAND`。\n\n这个功能会真实调用模型，可能消耗额度或费用。日志不会打印 API key，但会记录 provider/model 名称、耗时、退出状态和第一行错误摘要。\n`MODEL_EDGE_PROBE_ENABLED` 不使用凭据，也不调用 `/chat/completions`；它只检查 provider API 入口，比如 `https://api.deepseek.com`，是否能快速完成 DNS/TLS/HTTP 连接。\n\n## 安全边界\n\n这个项目不修改 OpenClaw 配置、不改微信插件源码、不处理消息内容。默认情况下它只做三件事：\n\n1. 探测本机 gateway 和外部 URL。\n2. 写自己的日志和状态文件。\n3. 在满足保护条件后执行配置好的 gateway 重启命令。\n\n可选模型探针只有在你显式开启后才会发起真实模型请求。\n\n分享日志前，请检查里面是否包含本机路径、服务名或私有通道地址。\n\n## 开源许可\n\nMIT-0。这个许可证符合 ClawHub skill 发布要求，也方便别人直接复用、改造和分发。\n\n## 发布到 ClawHub\n\n已发布包：\n\n- ClawHub：<https://clawhub.ai/zc-kama/gateway-resilience-guard>\n- Slug：`gateway-resilience-guard`\n\n本仓库带 `SKILL.md`，也可以重新发布为 OpenClaw skill：\n\n```bash\nclawhub publish . \\\n  --slug gateway-resilience-guard \\\n  --name \"OpenClaw Gateway Resilience Guard\" \\\n  --version 1.4.4 \\\n  --changelog \"Make dashboard controls robust with native semantic buttons/selects, fixed hover states, centered brand mark, and non-overlapping config rows\"\n```\n\n发布前需要先执行 `clawhub login` 完成 CLI 登录。\n\nFile v1.4.4:skill-card.md\n\n## Description:\n\nOpenClaw Gateway Resilience Guard keeps Gateway, channels, logs, and optional model-provider probes observable after Wi-Fi changes, WSL sleep/resume, session expiry, provider timeouts, or partial outages.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[zc-kama](https://clawhub.ai/user/zc-kama)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and operators use this skill to install and run a persistent local watchdog for OpenClaw Gateway, channel health, diagnostics, and recovery workflows on Linux, WSL, macOS, or Windows.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The watchdog installs persistent user-level services and can restart OpenClaw Gateway.\n\nMitigation: Install only when persistent local monitoring is intended, review the generated service or scheduled task, and keep restart limits enabled.\n\nRisk: Dashboard token and command settings create local command-execution exposure if actions or custom commands are enabled without review.\n\nMitigation: Keep the dashboard bound to 127.0.0.1, avoid custom command settings, disable dashboard actions where possible, and protect config and state directories.\n\nRisk: Diagnostics and exported logs can expose local paths, operational details, channel URLs, or provider metadata.\n\nMitigation: Review diagnostic exports and logs before sharing them outside the local operator environment.\n\n## Reference(s):\n\n- [OpenClaw Gateway Resilience Guard on ClawHub](https://clawhub.ai/zc-kama/skills/gateway-resilience-guard)\n- [Tencent/openclaw-weixin issue 141](https://github.com/Tencent/openclaw-weixin/issues/141)\n- [Tencent/openclaw-weixin issue 155](https://github.com/Tencent/openclaw-weixin/issues/155)\n\n## Skill Output:\n\n**Output Type(s):** [guidance, shell commands, configuration, code]\n\n**Output Format:** [Markdown with shell, PowerShell, and configuration snippets]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Produces operator guidance for installing, configuring, monitoring, and uninstalling the watchdog and optional OpenClaw plugin bridge.]\n\n## Skill Version(s):\n\n1.4.4 (source: server evidence release metadata, CHANGELOG, and package.json)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.4.4:openclaw-plugin/openclaw.plugin.json\n\n{\n  \"id\": \"resilience-guard\",\n  \"name\": \"Resilience Guard\",\n  \"description\": \"Adds a Gateway Control UI entry that opens the external OpenClaw watchdog dashboard.\",\n  \"version\": \"1.4.4\",\n  \"configSchema\": {\n    \"type\": \"object\",\n    \"additionalProperties\": false,\n    \"properties\": {\n      \"dashboardUrl\": {\n        \"type\": \"string\",\n        \"default\": \"http://127.0.0.1:18790/\"\n      }\n    }\n  },\n  \"uiHints\": {\n    \"dashboardUrl\": {\n      \"label\": \"Dashboard URL\",\n      \"help\": \"External watchdog dashboard URL. It remains available even when Gateway is down.\",\n      \"placeholder\": \"http://127.0.0.1:18790/\"\n    }\n  }\n}\n\nFile v1.4.4:openclaw-plugin/package.json\n\n{\n  \"name\": \"@zc-kama/openclaw-resilience-guard\",\n  \"version\": \"1.4.4\",\n  \"type\": \"module\",\n  \"description\": \"OpenClaw plugin bridge for the external Gateway Resilience Guard dashboard.\",\n  \"license\": \"MIT-0\",\n  \"openclaw\": {\n    \"extensions\": [\n      \"./index.js\"\n    ],\n    \"compat\": {\n      \"pluginApi\": \">=2026.3.24-beta.2\",\n      \"minGatewayVersion\": \"2026.3.24-beta.2\"\n    },\n    \"build\": {\n      \"openclawVersion\": \"2026.5.18\",\n      \"pluginSdkVersion\": \"2026.5.18\"\n    }\n  }\n}\n\nArchive v1.4.3: 21 files, 153868 bytes\n\nFiles: CHANGELOG.md (4659b), dashboard/server.py (18356b), dashboard/static/app.js (28443b), dashboard/static/index.html (13189b), dashboard/static/styles.css (18960b), dashboard/static/vendor/fluent-register.js (255b), dashboard/static/vendor/fluent-web-components.min.js (375019b), gateway-watchdog.ps1 (29656b), gateway-watchdog.sh (31639b), install-watchdog.ps1 (5020b), install-watchdog.sh (10659b), openclaw-plugin/index.js (1493b), openclaw-plugin/openclaw.plugin.json (624b), openclaw-plugin/package.json (485b), README-watchdog.md (1588b), README.md (15927b), README.zh-CN.md (15851b), SKILL.md (4331b), uninstall-watchdog.ps1 (1496b), uninstall-watchdog.sh (2009b), _meta.json (143b)\n\nFile v1.4.3:SKILL.md\n\n---\nname: gateway-resilience-guard\ndescription: OpenClaw Gateway Resilience Guard keeps Gateway, channels, logs, and optional model-provider probes observable after Wi-Fi changes, WSL sleep/resume, session expiry, provider timeouts, or partial outages. It adds a localhost dashboard with charts, event trends, language/theme switching, strategy presets, guarded action unlock, native OpenClaw health probes, restart safeguards, systemd, LaunchAgent, Task Scheduler, and an optional OpenClaw plugin bridge.\ntags:\n  - openclaw\n  - gateway\n  - watchdog\n  - resilience\n  - wechat\n  - wsl\n  - systemd\n  - macos\n  - windows\nrequirements:\n  tools:\n    - bash\n    - curl\n    - systemctl optional\n    - launchctl optional\n    - powershell optional\n    - openclaw recommended\npermissions:\n  - Writes a user-level systemd service, macOS LaunchAgent, or Windows scheduled task when requested.\n  - Writes config and logs under user-level config/state directories.\n  - Starts a localhost dashboard on 127.0.0.1:18790 when Python is available.\n  - Restarts OpenClaw Gateway through systemctl --user or openclaw gateway restart.\n  - Optional model probe sends real OpenClaw model requests only when explicitly enabled.\n---\n\n# OpenClaw Gateway Resilience Guard\n\nUse this skill when a user wants to keep OpenClaw Gateway and message channels online after network drops, WSL sleep/resume, macOS/Windows wake events, long-lived connection failures, or recurring model-provider timeouts.\n\n## Install\n\nLinux, WSL, or macOS:\n\n```bash\nbash install-watchdog.sh\n```\n\nThe installer works with defaults. It prompts for the main channel probe URL when running interactively, but pressing Enter is enough for the default WeChat probe.\n\nFor unattended install:\n\n```bash\nbash install-watchdog.sh --yes\n```\n\nFor a custom channel:\n\n```bash\nbash install-watchdog.sh --channel-url \"https://your-channel.example.com/health\"\n```\n\nWindows PowerShell:\n\n```powershell\npowershell -ExecutionPolicy Bypass -File .\\install-watchdog.ps1\n```\n\nAfter install, open the standalone dashboard:\n\n```text\nhttp://127.0.0.1:18790/\n```\n\nThe dashboard is served by the watchdog process, not Gateway, so it remains the recovery entry when Gateway is down.\n\nOptional Gateway-side bridge while Gateway is healthy:\n\n```bash\nopenclaw plugins install ./openclaw-plugin\nopenclaw plugins enable resilience-guard\nopenclaw gateway restart\n```\n\n## Operate\n\n```bash\nsystemctl --user status gateway-watchdog\njournalctl --user -u gateway-watchdog -f\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh\n```\n\nOn macOS, inspect `launchctl print gui/$(id -u)/ai.clawhub.gateway-resilience-guard`.\nOn Windows, inspect `Get-ScheduledTask -TaskName \"OpenClaw Gateway Resilience Guard\"`.\n\nIf user systemd is unavailable, the installer starts a direct background fallback and stores its pid under `~/.local/state/openclaw-gateway-watchdog/watchdog.pid`.\n\n## Safety Model\n\nThe watchdog restarts only after layered checks:\n\n1. OpenClaw native health probes: `openclaw gateway status --require-rpc`, `openclaw health --json --verbose`, and `openclaw status --deep` when available.\n2. Runtime diagnostics: `openclaw models status` plus `openclaw logs --plain` signal scanning for provider timeout, proxy/network, rate limit, auth, channel session, gateway degraded, config reload, and task runtime warnings.\n3. Local gateway health URL/TCP and main channel URL fallbacks.\n4. General network URLs to avoid restarting during whole-machine network failure.\n5. Optional model-provider probe via `openclaw agent --json`; default action is evidence logging only.\n6. Dashboard action token, backoff, and hourly restart limits to avoid restart storms.\n\nModel probing is disabled by default because it consumes real provider quota. To diagnose provider timeouts, set `MODEL_PROBE_ENABLED=1`, keep `MODEL_PROBE_ACTION=log`, and inspect `model-probe-history.jsonl` the next day.\nDiagnostic log scanning is enabled by default, but `OPENCLAW_DIAG_ACTION=log` keeps it non-invasive unless the operator explicitly opts into a restart or custom command.\nUse the dashboard strategy buttons for common modes: observe, overnight diagnosis, channel recovery, and conservative circuit breaker.\n\nTell users to review `~/.config/openclaw-gateway-watchdog/watchdog.env` before publishing, sharing logs, or reporting issues.\n\nFile v1.4.3:README.md\n\n# OpenClaw Gateway Resilience Guard\n\nExternal recovery guard for OpenClaw Gateway and long-lived channel plugins such as `openclaw-weixin`.\n\nThis project is for people who run OpenClaw continuously on Linux, WSL, macOS, or Windows and need the gateway to recover from channel disconnects, network sleep/resume, and long-lived session failures without babysitting the terminal.\n\n## Problem\n\nOpenClaw Gateway can still be alive while an individual channel is no longer healthy. This is common after laptop sleep, Wi-Fi changes, WSL network hiccups, or long idle periods.\n\nThe WeChat plugin is especially sensitive because it depends on a long-poll `getUpdates` loop. In the upstream `Tencent/openclaw-weixin` code, the monitor has a limited retry loop and session guard:\n\n- `monitor.ts` defines `MAX_CONSECUTIVE_FAILURES = 3` and `BACKOFF_DELAY_MS = 30_000`.\n- `session-guard.ts` defines `SESSION_PAUSE_DURATION_MS = 60 * 60 * 1000` and `SESSION_EXPIRED_ERRCODE = -14`.\n- Issue [Tencent/openclaw-weixin#141](https://github.com/Tencent/openclaw-weixin/issues/141) reports that after a config hot reload the monitor can end without starting again; the workaround is `openclaw gateway restart`.\n- Issue [Tencent/openclaw-weixin#155](https://github.com/Tencent/openclaw-weixin/issues/155) reports that `errcode=-14` can enter a 60-minute session pause loop and block outbound messages.\n\nThis watchdog does not replace the official plugin. It is an external safety net: when the gateway or channel stops behaving like a live system, it restarts the gateway with guardrails.\n\n## Design\n\nThe script uses a layered health model before it restarts anything:\n\n| Layer | Probe | Purpose |\n| --- | --- | --- |\n| Gateway | `openclaw gateway status --json --require-rpc`, local health URL, local TCP port, service/process fallback | Detect whether OpenClaw Gateway is down or locally unreachable. |\n| Channel | `openclaw health --json --verbose`, `openclaw status --deep`, optional `openclaw channels status --probe`, then URL fallback | Prefer OpenClaw's own per-channel health model, then fall back to a configured URL when native probes are unavailable. |\n| Runtime diagnostics | `openclaw models status --json`, `openclaw logs --plain`, warning classification | Separate provider, proxy/network, auth/rate-limit, gateway, channel-session, config reload, and task-runtime evidence before choosing an action. |\n| Model API, optional | `openclaw agent --json` with the configured model provider | Detect whether the configured model path is timing out while Gateway and channels still look healthy. Disabled by default because it makes real model calls. |\n| Network | multiple independent URLs, default Baidu/QQ/Weixin | Avoid restarting the gateway during whole-machine or ISP network failure. |\n\nOnly gateway failures restart immediately. Channel failures go through confirmation, network split-brain protection, exponential backoff, and restart-rate limits.\nModel probe failures default to evidence logging only; users can opt in to restart or a custom command after consecutive failures.\n\n## Recovery Policy\n\n- Gateway down: restart immediately.\n- Gateway running but channel probe fails: confirm with general network probes.\n- General network also fails: do nothing except wait; restarting will not fix an offline machine.\n- General network works but channel stays down: wait with exponential backoff, re-check, then restart.\n- OpenClaw warning logs: classify and record evidence first; default action is log-only.\n- Five consecutive successful channel probes reset the failure state.\n- Restart storm protection limits gateway restarts per hour.\n- Night hours can use a slower probe interval to reduce noise.\n\n## Files\n\n| File | Purpose |\n| --- | --- |\n| `gateway-watchdog.sh` | Main daemon loop: probes, backoff, restart decisions, log rotation, single-instance lock. |\n| `gateway-watchdog.ps1` | Windows-native daemon loop for Task Scheduler. |\n| `dashboard/` | Standalone local Web UI and API served by the watchdog, independent of Gateway. |\n| `openclaw-plugin/` | Optional native OpenClaw plugin bridge that redirects `/resilience-guard` to the standalone dashboard. |\n| `install-watchdog.sh` | Linux/WSL/macOS installer: copies files, writes config, creates systemd user service or macOS LaunchAgent. |\n| `install-watchdog.ps1` | Windows installer: creates config and a Task Scheduler job. |\n| `uninstall-watchdog.sh` | Stops and removes the service and installed scripts. |\n| `uninstall-watchdog.ps1` | Windows uninstaller. |\n| `SKILL.md` | ClawHub/OpenClaw skill metadata and operator instructions. |\n| `README.zh-CN.md` | Chinese documentation. |\n\n## Install\n\nInstall from ClawHub:\n\n```bash\nclawhub install gateway-resilience-guard\n```\n\nOr use this repository directly.\n\nLinux, WSL, or macOS:\n\n```bash\nbash install-watchdog.sh\n```\n\nFor unattended install:\n\n```bash\nbash install-watchdog.sh --yes\n```\n\nFor a custom channel probe:\n\n```bash\nbash install-watchdog.sh --channel-url \"https://your-channel.example.com/health\"\n```\n\nWindows PowerShell:\n\n```powershell\npowershell -ExecutionPolicy Bypass -File .\\install-watchdog.ps1\n```\n\n## Dashboard\n\nThe installer enables a standalone local dashboard:\n\n```text\nhttp://127.0.0.1:18790/\n```\n\nThis dashboard is served by the watchdog, not by OpenClaw Gateway. If Gateway is down, the dashboard can still open and show the last known evidence.\n\nIt includes:\n\n- Gateway, channel, network, OpenClaw log, and model-provider status.\n- Category charts for provider timeout, proxy/network, rate-limit, auth, channel session, Gateway degraded, config reload, and task runtime warnings.\n- Sidebar view switching for Overview, Trends, Strategy, Logs, and Config so monitoring, actions, and detail evidence are separated.\n- Locally bundled Fluent Web Components for refined selects, buttons, badges, and guarded controls without relying on a CDN.\n- Denser page composition: trend digest and latest signals, strategy matrix and manual actions, config runtime map.\n- Event trend chart with separate lanes for API failures, log warnings, successful model probes, and healthy diagnostics.\n- Chinese/English language switching and Light, Dark, Ocean, and Forest themes.\n- Status-file freshness checks so stale data is obvious.\n- Quick strategy buttons: observe, overnight diagnosis, channel recovery, and conservative circuit breaker.\n- Guarded actions with an unlock flow: run diagnostics, restart Gateway, apply presets, and export a diagnostic JSON bundle.\n\nDashboard actions are bound to localhost and protected with the generated `DASHBOARD_TOKEN`. The token is injected only into the same-origin dashboard page. A random token is written during install; the value is not shown in logs.\n\nOptional OpenClaw plugin bridge:\n\n```bash\nopenclaw plugins install ./openclaw-plugin\nopenclaw plugins enable resilience-guard\nopenclaw gateway restart\n```\n\nThen open the Gateway route while Gateway is healthy:\n\n```text\nhttp://127.0.0.1:18789/resilience-guard\n```\n\nThat route redirects to the external dashboard. It is a convenience entry only; the external dashboard remains the recovery entry when Gateway is unavailable.\n\nThe installer writes:\n\n- scripts to `~/.local/share/openclaw-gateway-watchdog`;\n- config to `~/.config/openclaw-gateway-watchdog/watchdog.env`;\n- logs to `~/.local/state/openclaw-gateway-watchdog/watchdog.log`;\n- a Linux/WSL user service to `~/.config/systemd/user/gateway-watchdog.service`;\n- a macOS LaunchAgent to `~/Library/LaunchAgents/ai.clawhub.gateway-resilience-guard.plist`;\n- a Windows scheduled task named `OpenClaw Gateway Resilience Guard`.\n\nIf user systemd is unavailable, the installer starts a direct background fallback process and stores its pid in the state directory.\n\n## Manage\n\n```bash\nsystemctl --user status gateway-watchdog\njournalctl --user -u gateway-watchdog -f\nsystemctl --user restart gateway-watchdog\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh\n```\n\nmacOS:\n\n```bash\nlaunchctl print gui/$(id -u)/ai.clawhub.gateway-resilience-guard\ntail -f ~/.local/state/openclaw-gateway-watchdog/watchdog.log\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh\n```\n\nWindows:\n\n```powershell\nGet-ScheduledTask -TaskName \"OpenClaw Gateway Resilience Guard\"\nGet-Content \"$env:LOCALAPPDATA\\openclaw-gateway-watchdog\\watchdog.log\" -Wait\npowershell -ExecutionPolicy Bypass -File \"$env:LOCALAPPDATA\\openclaw-gateway-watchdog\\uninstall-watchdog.ps1\"\n```\n\nRemove config and logs too:\n\n```bash\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh --purge\n```\n\n## Configuration\n\nMost users can keep the generated defaults. Advanced settings live in:\n\n```text\n~/.config/openclaw-gateway-watchdog/watchdog.env\n```\n\nCommon keys:\n\n```bash\nGATEWAY_SERVICE=\"openclaw-gateway\"\nGATEWAY_HEALTH_URL=\"http://127.0.0.1:18789/healthz\"\nGATEWAY_HOST=\"127.0.0.1\"\nGATEWAY_PORT=\"18789\"\nCHANNEL_URL=\"https://ilinkai.weixin.qq.com\"\nNETWORK_URLS=\"https://www.baidu.com https://www.qq.com https://api.weixin.qq.com\"\nRESTART_COMMAND=\"systemctl --user restart openclaw-gateway\"\nOPENCLAW_NATIVE_PROBES=\"auto\"\nOPENCLAW_HEALTH_TIMEOUT_MS=\"12000\"\nOPENCLAW_GATEWAY_STRICT=\"0\"\nOPENCLAW_CHANNELS_PROBE=\"1\"\nOPENCLAW_DIAG_ENABLED=\"1\"\nOPENCLAW_DIAG_INTERVAL=\"300\"\nOPENCLAW_LOG_SCAN_ENABLED=\"1\"\nOPENCLAW_LOG_LIMIT=\"200\"\nOPENCLAW_LOG_SIGNAL_LIMIT=\"40\"\nOPENCLAW_LOG_TIMEOUT_MS=\"15000\"\nOPENCLAW_LOG_WARN_PATTERNS=\"fetch failed|fetch timeout|LLM idle timeout|model silent|...\"\nOPENCLAW_DIAG_ACTION=\"log\"\nOPENCLAW_DIAG_FAILURES_BEFORE_ACTION=\"2\"\nOPENCLAW_DIAG_COMMAND=\"\"\nDASHBOARD_ENABLED=\"1\"\nDASHBOARD_HOST=\"127.0.0.1\"\nDASHBOARD_PORT=\"18790\"\nDASHBOARD_ACTIONS_ENABLED=\"1\"\nDASHBOARD_TOKEN=\"generated-at-install\"\nDASHBOARD_DIR=\"~/.local/share/openclaw-gateway-watchdog/dashboard\"\nMODEL_PROBE_ENABLED=\"0\"\nMODEL_EDGE_PROBE_ENABLED=\"1\"\nMODEL_PROBE_INTERVAL=\"1800\"\nMODEL_PROBE_TIMEOUT=\"120\"\nMODEL_PROBE_FAILURES_BEFORE_ACTION=\"2\"\nMODEL_PROBE_ACTION=\"log\"\nMODEL_PROBE_COMMAND=\"\"\nMODEL_PROBE_MODEL=\"\"\nMODEL_PROBE_THINKING=\"off\"\nMODEL_PROBE_SESSION_ID=\"watchdog-model-probe\"\nMODEL_PROBE_MESSAGE=\"Reply with exactly OK.\"\nBASE_INTERVAL=\"60\"\nNIGHT_INTERVAL=\"300\"\nMAX_INTERVAL=\"1800\"\nCHANNEL_FAILURES_BEFORE_RESTART=\"2\"\nSUCCESS_COUNT_TO_RESET=\"5\"\nMAX_RESTARTS_PER_HOUR=\"6\"\n```\n\nUse `RESTART_COMMAND` if your OpenClaw install is not managed by a user-level systemd unit. Example:\n\n```bash\nRESTART_COMMAND=\"openclaw gateway restart\"\n```\n\nOn Windows, the generated config is JSON:\n\n```text\n%APPDATA%\\openclaw-gateway-watchdog\\watchdog.json\n```\n\nSet `OpenClawNativeProbes` to `false` if your OpenClaw CLI is too old for `openclaw health` or `openclaw status --deep`.\n\n### OpenClaw diagnostics and log signals\n\nThe watchdog does more than ping one URL. Every `OPENCLAW_DIAG_INTERVAL` seconds it collects an OpenClaw diagnostic snapshot:\n\n```text\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-gateway-status.json\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-health.json\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-model-status.json\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-status-deep.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-logs.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-log-signals.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-log-signal-categories.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-diagnostics.jsonl\n```\n\n`last-openclaw-log-signals.txt` is filtered from `openclaw logs --plain`. It classifies common failure families such as `provider_timeout`, `proxy_or_network`, `provider_rate_limit`, `provider_auth`, `abort_stuck`, `memory_dream_timeout`, `channel_session`, `gateway_degraded`, `config_reload`, and `task_runtime`.\n\nDefault action is `OPENCLAW_DIAG_ACTION=\"log\"` because warnings are evidence, not always proof that a restart is correct. If you explicitly set `OPENCLAW_DIAG_ACTION=\"restart\"` or `command`, the action only runs after consecutive diagnostic warnings and only when the general network probes still pass.\n\nPractical interpretation:\n\n| Evidence | Likely scope | Default strategy |\n| --- | --- | --- |\n| Gateway status/health is down | Gateway process or RPC path | Restart Gateway immediately through the normal gateway policy. |\n| Channel probe fails, network probes pass | Channel/session path | Backoff, re-check, then restart Gateway if still failed. |\n| OpenClaw logs show provider timeout, model probe also fails | Provider/API path | Log evidence; optional custom action. A Gateway restart may not fix provider outage. |\n| OpenClaw logs show provider timeout, model probe succeeds | OpenClaw runtime, task, session, or specific request path | Keep evidence, inspect logs; avoid blaming the provider alone. |\n| Logs show proxy/DNS/TLS errors | Local proxy, DNS, TLS, or ISP route | Log evidence and avoid restart storms; fix network/proxy route first. |\n| Logs show session expiry or monitor stopped | Channel plugin/session | Gateway restart is often useful after confirmation. |\n\n### Optional model probe\n\nSet `MODEL_PROBE_ENABLED=\"1\"` when you need to prove whether failures are in the model-provider path instead of Gateway or channel health.\n\nWhen enabled, the watchdog first reads OpenClaw's configured model provider and probes its provider `baseUrl` without credentials. Then it runs the end-to-end model probe:\n\n```bash\nopenclaw agent --session-id \"$MODEL_PROBE_SESSION_ID\" \\\n  --thinking \"$MODEL_PROBE_THINKING\" \\\n  --timeout \"$MODEL_PROBE_TIMEOUT\" \\\n  --json \\\n  --message \"$MODEL_PROBE_MESSAGE\"\n```\n\nIf `MODEL_PROBE_MODEL` is empty, OpenClaw's configured default model is used. Results are written to the main log and to:\n\n```text\n~/.local/state/openclaw-gateway-watchdog/model-probe-history.jsonl\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-model-probe.json\n~/.local/state/openclaw-gateway-watchdog/last-model-api-edge-probe.txt\n```\n\nRecommended diagnostic settings for overnight provider issues:\n\n```bash\nMODEL_PROBE_ENABLED=\"1\"\nMODEL_EDGE_PROBE_ENABLED=\"1\"\nMODEL_PROBE_INTERVAL=\"600\"\nMODEL_PROBE_TIMEOUT=\"120\"\nMODEL_PROBE_ACTION=\"log\"\nMODEL_PROBE_THINKING=\"off\"\nMODEL_PROBE_MESSAGE=\"Reply with exactly OK.\"\n```\n\n`MODEL_PROBE_ACTION` can be:\n\n- `log`: record evidence only. This is the default.\n- `restart`: restart Gateway after `MODEL_PROBE_FAILURES_BEFORE_ACTION` consecutive model failures, but only if general network probes still pass.\n- `command`: run `MODEL_PROBE_COMMAND` after consecutive failures.\n\nThis feature sends real model requests and may consume provider quota or money. It does not print API keys, but it does store provider/model names, timing, exit status, and the first non-empty error line.\n`MODEL_EDGE_PROBE_ENABLED` does not use credentials and does not call `/chat/completions`; it only checks whether the provider API edge such as `https://api.deepseek.com` is reachable quickly.\n\n## Safety Notes\n\nThis project intentionally avoids destructive behavior. It does not edit OpenClaw configuration, tokens, sessions, or plugin files. By default it only probes URLs and restarts the gateway through the configured command. The optional model probe makes real model calls only after you explicitly enable it.\n\nBefore sharing logs, review them for local paths, service names, and channel URLs.\n\n## License\n\nMIT-0. This matches ClawHub's skill publishing requirement and allows reuse without attribution requirements.\n\n## Publish To ClawHub\n\nPublished package:\n\n- ClawHub: <https://clawhub.ai/zc-kama/gateway-resilience-guard>\n- Slug: `gateway-resilience-guard`\n\nThis repository includes `SKILL.md`, so it can also be republished as an OpenClaw skill bundle:\n\n```bash\nclawhub publish . \\\n  --slug gateway-resilience-guard \\\n  --name \"OpenClaw Gateway Resilience Guard\" \\\n  --version 1.4.3 \\\n  --changelog \"Rebuild dashboard on local Fluent Web Components with denser pages, refined controls, motion, and bundled vendor assets\"\n```\n\nClawHub requires CLI authentication. Run `clawhub login` first.\n\nFile v1.4.3:_meta.json\n\n{\n  \"ownerId\": \"kn70e6hn9pv8yrpzjn5kqczygd8708vw\",\n  \"slug\": \"gateway-resilience-guard\",\n  \"version\": \"1.4.3\",\n  \"publishedAt\": 1779381982393\n}\n\nFile v1.4.3:CHANGELOG.md\n\n# Changelog\n\n## 1.4.3\n\n- Rebuild the dashboard visual system on local Fluent UI Web Components instead of plain native controls.\n- Replace the old dropdowns and buttons with componentized controls, dark glass panels, animated ambient background, and refined hover states.\n- Rebalance every dashboard page so Overview, Trends, Strategy, Logs, and Config each have complete work areas rather than sparse split-up cards.\n- Add richer summaries for trend digest, latest signals, strategy matrix, action center, and runtime map.\n- Serve bundled local component assets under `/vendor/` so the recovery UI does not depend on CDN availability.\n\n## 1.4.2\n\n- Rework the dashboard sidebar into real view switching for Overview, Trends, Strategy, Logs, and Config instead of one long page.\n- Tighten the dashboard grid, panel sizing, button alignment, and sidebar polish for a cleaner operations-console layout.\n- Fix the trend chart axis label collision by removing the redundant x-axis title and reserving more space for tick labels.\n- Keep charts stable across refreshes by drawing only the active view and ignoring hidden canvases.\n- Move strategy explanations into centered button content and remove the sidebar localhost note.\n\n## 1.4.1\n\n- Redesign the dashboard with a sidebar layout, stronger visual grouping, and selectable Light, Dark, Ocean, and Forest themes.\n- Add Chinese/English language switching for dashboard labels, buttons, hints, legends, and notifications.\n- Fix canvas redraw sizing so repeated refreshes do not stretch dashboard panels.\n- Replace \"overnight timeline\" with a general event trend chart that includes lanes, legends, and time-axis ticks.\n- Auto-follow the newest watchdog log lines while keeping a toggle for manual scrolling.\n- Add an unlock/lock action flow and strategy hover hints so guarded controls are discoverable.\n\n## 1.4.0\n\n- Add a standalone local dashboard at `http://127.0.0.1:18790` that remains available when OpenClaw Gateway is down.\n- Add dashboard API summaries, log-signal charts, overnight timelines, status file freshness, safe config summaries, diagnostic export, and quick strategy presets.\n- Add guarded dashboard actions for run diagnostics, restart Gateway, and apply presets using a generated local action token.\n- Add an OpenClaw native plugin bridge that registers `/resilience-guard` and redirects to the external dashboard when Gateway is healthy.\n- Install dashboard files on Linux/WSL, macOS, and Windows.\n\n## 1.3.1\n\n- Publish a ClawHub package that includes the Windows PowerShell installer, watchdog, and uninstaller files.\n\n## 1.3.0\n\n- Add an opt-in model-provider probe that uses `openclaw agent --json` against the configured OpenClaw model path.\n- Add a no-credential provider edge probe that checks the configured provider `baseUrl` before the end-to-end model call.\n- Add OpenClaw runtime diagnostics that snapshot gateway status, health, model status, deep status, and recent OpenClaw logs.\n- Classify log signals for provider timeout, proxy/network, rate limit, auth, channel session, gateway degraded, config reload, task runtime, and related warning families.\n- Record model probe evidence in the main log and `model-probe-history.jsonl` without logging API keys.\n- Add configurable model failure actions: log-only, gateway restart, or a custom command after consecutive failures.\n- Document safe overnight diagnostics for separating Gateway/channel failures from model-provider timeouts.\n\n## 1.2.0\n\n- Add OpenClaw-native health probing via `openclaw gateway status --json --require-rpc`, `openclaw health --json --verbose`, and `openclaw status --deep`.\n- Add Windows Task Scheduler support with native PowerShell install, watchdog, and uninstall scripts.\n- Add macOS LaunchAgent support to the Bash installer and uninstaller.\n- Document cross-platform installation and native probe configuration.\n\n## 1.1.0\n\n- Rename the ClawHub package to `gateway-resilience-guard`.\n- Expand the public summary to describe the layered probes, restart guardrails, and advantages over simple restart loops or one-URL monitors.\n\n## 1.0.1\n\n- Add the published ClawHub URL and install command to the README files.\n\n## 1.0.0\n\n- First public zero-config release.\n- Install from the current folder instead of a hard-coded workspace path.\n- Generate user config and systemd service automatically.\n- Add layered gateway/channel/network probes.\n- Add exponential backoff, hourly restart limits, lock protection, and log rotation.\n- Reset failure state after five consecutive successful channel probes.\n- Add ClawHub-ready `SKILL.md`.\n- Initial ClawHub publication used a temporary slug before the 1.1.0 rename.\n\nFile v1.4.3:README-watchdog.md\n\n# OpenClaw Gateway 看门狗\n\n这是项目的中文快捷入口。完整中文文档见 [README.zh-CN.md](README.zh-CN.md)，英文文档见 [README.md](README.md)。\n\n最简单安装：\n\n```bash\nbash install-watchdog.sh\n```\n\nWindows：\n\n```powershell\npowershell -ExecutionPolicy Bypass -File .\\install-watchdog.ps1\n```\n\n默认配置即可启动；需要指定自己的通道地址时：\n\n```bash\nbash install-watchdog.sh --channel-url \"https://你的通道地址/health\"\n```\n\n安装后还有一个独立图形化控制台：\n\n```text\nhttp://127.0.0.1:18790/\n```\n\n它由 watchdog 自己提供，不依赖 OpenClaw Gateway。Gateway 挂掉时，这个页面仍然可以看状态、图表、策略、日志和诊断导出。Dashboard 支持中英文切换、主题切换、事件趋势图、异常分类图和受保护操作解锁。\n\n安装后它会默认采集 OpenClaw 诊断快照和日志信号，包括 gateway status、health、models status、status --deep 和 `openclaw logs --plain` 的 WARN/ERROR 摘要。关键证据在：\n\n```text\n~/.local/state/openclaw-gateway-watchdog/watchdog.log\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-diagnostics.jsonl\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-log-signals.txt\n```\n\n可选模型探针默认关闭。需要排查模型 API 是否在某个时段超时时，可以在配置里显式开启 `MODEL_PROBE_ENABLED=\"1\"`；它会发送极小的 `openclaw agent --json` 请求，并把结果写入 watchdog 日志和 `model-probe-history.jsonl`。默认动作仍是只记录证据，不会擅自改 OpenClaw 配置。\n\nFile v1.4.3:README.zh-CN.md\n\n# OpenClaw Gateway Resilience Guard\n\nOpenClaw Gateway 外部恢复守护脚本，面向 `openclaw-weixin` 这类需要长期在线的通道。\n\n这个项目解决的是一个很具体的运维问题：OpenClaw Gateway 进程还活着，但通道、长轮询、WSL/系统网络或 session 状态已经坏掉，导致消息收不到、发不出，最后只能手动 `openclaw gateway restart`。\n\n## 问题背景\n\nOpenClaw Gateway 和通道插件是长连接系统。电脑睡眠、切换 Wi-Fi、WSL 网络重建、iLink 长时间空闲、配置热加载，都可能让“进程存活”和“通道可用”变成两件事。\n\n以官方 `Tencent/openclaw-weixin` 为例，源码和 issue 里能看到几个已知边界：\n\n- `monitor.ts` 里 `MAX_CONSECUTIVE_FAILURES = 3`，失败后 `BACKOFF_DELAY_MS = 30_000`，也就是插件内部主要是 3 次失败后的 30 秒退避。\n- `session-guard.ts` 里 `SESSION_PAUSE_DURATION_MS = 60 * 60 * 1000`，`SESSION_EXPIRED_ERRCODE = -14`，session 过期会进入 60 分钟暂停窗口。\n- [Tencent/openclaw-weixin#141](https://github.com/Tencent/openclaw-weixin/issues/141) 记录了配置热加载后 Monitor 结束但不再启动，临时处理方式是手动重启 gateway。\n- [Tencent/openclaw-weixin#155](https://github.com/Tencent/openclaw-weixin/issues/155) 记录了 `errcode=-14` 后进入 60 分钟循环暂停、出站消息被阻塞的问题。\n\n所以这个项目不是替换官方插件，而是在外面加一层独立 watchdog：当通道或 gateway 进入“看起来还活着，实际上已经不能工作”的状态时，用更保守的探测和熔断策略自动恢复。\n\n## 工作原理\n\n看门狗使用分层健康检查，从浅到深判断是否真的需要重启：\n\n| 层级 | 检查对象 | 作用 |\n| --- | --- | --- |\n| Gateway 本机状态 | `openclaw gateway status --json --require-rpc`、本机 health URL、本机 TCP 端口、服务/进程兜底 | 判断 OpenClaw Gateway 是否已经挂掉或本机不可达。 |\n| 通道状态 | `openclaw health --json --verbose`、`openclaw status --deep`、可选 `openclaw channels status --probe`，最后才是 URL 兜底 | 优先使用 OpenClaw 自己的全通道健康模型；CLI 不支持时再退回 URL 探测。 |\n| 运行时诊断 | `openclaw models status --json`、`openclaw logs --plain`、WARN/ERROR 分类 | 在动作前区分 provider、代理/网络、鉴权/限流、gateway、通道 session、配置热加载和任务运行时证据。 |\n| 模型 API，可选 | 使用 `openclaw agent --json` 走 OpenClaw 当前配置的模型 provider | 判断 Gateway 和通道都健康时，真正卡住的是不是模型 API 链路。默认关闭，因为它会真实消耗模型调用。 |\n| 外部网络状态 | 百度、QQ、微信 API 等多个独立 URL | 排除全局断网，避免电脑没网时误重启 gateway。 |\n\n核心策略是：gateway 真挂了就立即重启；通道不通时先确认不是全局断网；网络正常但通道持续失败，才进入退避等待和重启流程。\n模型探针默认只记录证据日志；如果你显式配置，也可以在连续失败后重启 gateway 或执行自定义命令。\n\n## 恢复策略\n\n- Gateway 本机健康检查失败：立即重启。\n- 通道 URL 失败：累计失败次数，进入故障流程。\n- 外部网络全部失败：认为是全局断网，只等待，不重启。\n- 外部网络正常但通道仍失败：指数退避后再次确认，再重启 gateway。\n- OpenClaw 日志出现 WARN/ERROR：先分类和记录证据，默认只写日志。\n- 连续 5 次通道探测成功后，清空失败状态。\n- 每小时最多重启固定次数，防止网络抖动时形成重启风暴。\n- 深夜可以降低检查频率，减少无意义日志。\n\n## 文件说明\n\n| 文件 | 作用 |\n| --- | --- |\n| `gateway-watchdog.sh` | 主守护脚本，负责探测、退避、熔断、日志、单实例锁和重启决策。 |\n| `gateway-watchdog.ps1` | Windows 原生守护脚本，用于 Task Scheduler。 |\n| `dashboard/` | 独立本地 Web UI 和 API，由 watchdog 自己提供，不依赖 Gateway。 |\n| `openclaw-plugin/` | 可选 OpenClaw 原生插件桥接入口，把 `/resilience-guard` 跳转到独立 dashboard。 |\n| `install-watchdog.sh` | Linux/WSL/macOS 安装脚本，自动复制文件、生成配置、创建 systemd 用户服务或 macOS LaunchAgent。 |\n| `install-watchdog.ps1` | Windows 安装脚本，创建配置和计划任务。 |\n| `uninstall-watchdog.sh` | 卸载脚本，停止服务并删除安装目录。 |\n| `uninstall-watchdog.ps1` | Windows 卸载脚本。 |\n| `SKILL.md` | ClawHub/OpenClaw 技能元数据和使用说明。 |\n| `README.md` | 英文文档。 |\n\n## 安装\n\n通过 ClawHub 安装：\n\n```bash\nclawhub install gateway-resilience-guard\n```\n\n也可以直接使用本仓库。\n\nLinux、WSL 或 macOS：\n\n```bash\nbash install-watchdog.sh\n```\n\n无人值守安装：\n\n```bash\nbash install-watchdog.sh --yes\n```\n\n指定自己的通道探测地址：\n\n```bash\nbash install-watchdog.sh --channel-url \"https://你的通道地址/health\"\n```\n\nWindows PowerShell：\n\n```powershell\npowershell -ExecutionPolicy Bypass -File .\\install-watchdog.ps1\n```\n\n## 图形化 Dashboard\n\n安装后会启用一个独立本地 dashboard：\n\n```text\nhttp://127.0.0.1:18790/\n```\n\n这个页面由 watchdog 自己提供，不依赖 OpenClaw Gateway。所以 Gateway 挂掉时，它仍然可以打开，看到最后一次诊断证据。\n\n它包含：\n\n- Gateway、通道、外部网络、OpenClaw 日志、模型 provider 的分层状态。\n- provider timeout、代理/网络、限流、鉴权、通道 session、Gateway degraded、配置热加载、任务运行时异常的分类图表。\n- 侧边栏分页：总览、趋势、策略、日志、配置分开呈现，避免把监控、操作和明细堆在一个长页面里。\n- 本地打包的 Fluent Web Components 控件：下拉框、按钮、状态标签不依赖 CDN，断网时仍可用。\n- 重新分配每页信息密度：趋势页有摘要和最新信号，策略页有策略矩阵和手动操作，配置页有运行地图。\n- 事件趋势图，把 API 失败、日志 WARN、模型探针成功、健康诊断分成不同泳道，并保留清晰的时间刻度。\n- 中英文语言切换，以及 Light、Dark、Ocean、Forest 四套主题。\n- 状态文件新鲜度，避免把过期快照误认为当前状态。\n- 快速策略按钮：观察模式、夜间诊断、通道恢复、保守熔断。\n- 带解锁流程的受保护操作：立即诊断、重启 Gateway、应用策略、导出诊断 JSON。\n\nDashboard 操作只绑定 localhost，并使用安装时生成的 `DASHBOARD_TOKEN` 保护。token 只注入同源页面，不会写入日志。\n\n可选 OpenClaw 插件桥接入口：\n\n```bash\nopenclaw plugins install ./openclaw-plugin\nopenclaw plugins enable resilience-guard\nopenclaw gateway restart\n```\n\nGateway 正常时可以打开：\n\n```text\nhttp://127.0.0.1:18789/resilience-guard\n```\n\n这个路由会跳转到外部 dashboard。它只是方便入口；真正救急的入口仍然是 `http://127.0.0.1:18790/`。\n\n安装后会生成：\n\n- 脚本目录：`~/.local/share/openclaw-gateway-watchdog`\n- 配置文件：`~/.config/openclaw-gateway-watchdog/watchdog.env`\n- 日志文件：`~/.local/state/openclaw-gateway-watchdog/watchdog.log`\n- Linux/WSL systemd 用户服务：`~/.config/systemd/user/gateway-watchdog.service`\n- macOS LaunchAgent：`~/Library/LaunchAgents/ai.clawhub.gateway-resilience-guard.plist`\n- Windows 计划任务：`OpenClaw Gateway Resilience Guard`\n\n如果当前环境没有 user systemd，安装脚本会退回到后台进程模式，并把 pid 写到状态目录。\n\n## 管理命令\n\n```bash\nsystemctl --user status gateway-watchdog\njournalctl --user -u gateway-watchdog -f\nsystemctl --user restart gateway-watchdog\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh\n```\n\nmacOS：\n\n```bash\nlaunchctl print gui/$(id -u)/ai.clawhub.gateway-resilience-guard\ntail -f ~/.local/state/openclaw-gateway-watchdog/watchdog.log\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh\n```\n\nWindows：\n\n```powershell\nGet-ScheduledTask -TaskName \"OpenClaw Gateway Resilience Guard\"\nGet-Content \"$env:LOCALAPPDATA\\openclaw-gateway-watchdog\\watchdog.log\" -Wait\npowershell -ExecutionPolicy Bypass -File \"$env:LOCALAPPDATA\\openclaw-gateway-watchdog\\uninstall-watchdog.ps1\"\n```\n\n连配置和日志一起删除：\n\n```bash\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh --purge\n```\n\n## 配置\n\n普通用户一般不用改。高级配置在：\n\n```text\n~/.config/openclaw-gateway-watchdog/watchdog.env\n```\n\n常用项：\n\n```bash\nGATEWAY_SERVICE=\"openclaw-gateway\"\nGATEWAY_HEALTH_URL=\"http://127.0.0.1:18789/healthz\"\nGATEWAY_HOST=\"127.0.0.1\"\nGATEWAY_PORT=\"18789\"\nCHANNEL_URL=\"https://ilinkai.weixin.qq.com\"\nNETWORK_URLS=\"https://www.baidu.com https://www.qq.com https://api.weixin.qq.com\"\nRESTART_COMMAND=\"systemctl --user restart openclaw-gateway\"\nOPENCLAW_NATIVE_PROBES=\"auto\"\nOPENCLAW_HEALTH_TIMEOUT_MS=\"12000\"\nOPENCLAW_GATEWAY_STRICT=\"0\"\nOPENCLAW_CHANNELS_PROBE=\"1\"\nOPENCLAW_DIAG_ENABLED=\"1\"\nOPENCLAW_DIAG_INTERVAL=\"300\"\nOPENCLAW_LOG_SCAN_ENABLED=\"1\"\nOPENCLAW_LOG_LIMIT=\"200\"\nOPENCLAW_LOG_SIGNAL_LIMIT=\"40\"\nOPENCLAW_LOG_TIMEOUT_MS=\"15000\"\nOPENCLAW_LOG_WARN_PATTERNS=\"fetch failed|fetch timeout|LLM idle timeout|model silent|...\"\nOPENCLAW_DIAG_ACTION=\"log\"\nOPENCLAW_DIAG_FAILURES_BEFORE_ACTION=\"2\"\nOPENCLAW_DIAG_COMMAND=\"\"\nDASHBOARD_ENABLED=\"1\"\nDASHBOARD_HOST=\"127.0.0.1\"\nDASHBOARD_PORT=\"18790\"\nDASHBOARD_ACTIONS_ENABLED=\"1\"\nDASHBOARD_TOKEN=\"安装时生成\"\nDASHBOARD_DIR=\"~/.local/share/openclaw-gateway-watchdog/dashboard\"\nMODEL_PROBE_ENABLED=\"0\"\nMODEL_EDGE_PROBE_ENABLED=\"1\"\nMODEL_PROBE_INTERVAL=\"1800\"\nMODEL_PROBE_TIMEOUT=\"120\"\nMODEL_PROBE_FAILURES_BEFORE_ACTION=\"2\"\nMODEL_PROBE_ACTION=\"log\"\nMODEL_PROBE_COMMAND=\"\"\nMODEL_PROBE_MODEL=\"\"\nMODEL_PROBE_THINKING=\"off\"\nMODEL_PROBE_SESSION_ID=\"watchdog-model-probe\"\nMODEL_PROBE_MESSAGE=\"Reply with exactly OK.\"\nBASE_INTERVAL=\"60\"\nNIGHT_INTERVAL=\"300\"\nMAX_INTERVAL=\"1800\"\nCHANNEL_FAILURES_BEFORE_RESTART=\"2\"\nSUCCESS_COUNT_TO_RESET=\"5\"\nMAX_RESTARTS_PER_HOUR=\"6\"\n```\n\n如果你的 OpenClaw 不是 systemd 用户服务管理，可以显式指定重启命令：\n\n```bash\nRESTART_COMMAND=\"openclaw gateway restart\"\n```\n\nWindows 的配置是 JSON：\n\n```text\n%APPDATA%\\openclaw-gateway-watchdog\\watchdog.json\n```\n\n如果你的 OpenClaw CLI 版本太旧，不支持 `openclaw health` 或 `openclaw status --deep`，可以把 `OpenClawNativeProbes` 设为 `false`。\n\n### OpenClaw 诊断和日志信号\n\n这个 watchdog 不只是 ping 一个 URL。它会每隔 `OPENCLAW_DIAG_INTERVAL` 秒采集一组 OpenClaw 运行快照：\n\n```text\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-gateway-status.json\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-health.json\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-model-status.json\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-status-deep.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-logs.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-log-signals.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-log-signal-categories.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-diagnostics.jsonl\n```\n\n`last-openclaw-log-signals.txt` 来自 `openclaw logs --plain` 的过滤结果，会把常见异常归类成 `provider_timeout`、`proxy_or_network`、`provider_rate_limit`、`provider_auth`、`abort_stuck`、`memory_dream_timeout`、`channel_session`、`gateway_degraded`、`config_reload`、`task_runtime` 等类别。\n\n默认策略是 `OPENCLAW_DIAG_ACTION=\"log\"`，因为 WARN 是证据，不一定等于“应该马上重启”。如果你显式改成 `restart` 或 `command`，也必须连续多次出现诊断异常，并且外部网络探测正常，才会执行动作。\n\n排查时可以这样判断：\n\n| 证据 | 更可能的问题范围 | 默认策略 |\n| --- | --- | --- |\n| Gateway status/health 挂了 | Gateway 进程或 RPC 链路 | 走原本的 Gateway 策略，立即重启。 |\n| 通道探测失败，但外部网络正常 | 通道或 session 链路 | 退避、复查，仍失败再重启 Gateway。 |\n| OpenClaw 日志显示 provider timeout，模型探针也失败 | 模型 provider/API 链路 | 先记录证据；可选自定义动作。单纯重启 Gateway 未必有用。 |\n| OpenClaw 日志显示 provider timeout，但模型探针成功 | OpenClaw 运行时、任务、session 或特定请求路径 | 继续保留证据，不把锅直接甩给 provider。 |\n| 日志显示 proxy/DNS/TLS 错误 | 本机代理、DNS、TLS 或运营商路由 | 记录证据，避免重启风暴，优先修代理/网络路由。 |\n| 日志显示 session expired 或 monitor stopped | 通道插件/session | 确认后重启 Gateway 往往有价值。 |\n\n### 可选模型探针\n\n如果你想判断问题到底出在 Gateway/通道，还是模型 provider 链路，可以显式开启：\n\n```bash\nMODEL_PROBE_ENABLED=\"1\"\n```\n\n开启后，看门狗会先读取 OpenClaw 当前模型 provider 的 `baseUrl`，做一次不带凭据、不消耗 token 的入口连通性探测。然后再执行端到端模型探针：\n\n```bash\nopenclaw agent --session-id \"$MODEL_PROBE_SESSION_ID\" \\\n  --thinking \"$MODEL_PROBE_THINKING\" \\\n  --timeout \"$MODEL_PROBE_TIMEOUT\" \\\n  --json \\\n  --message \"$MODEL_PROBE_MESSAGE\"\n```\n\n如果 `MODEL_PROBE_MODEL` 为空，就使用 OpenClaw 当前配置的默认模型。结果会写入主日志，以及：\n\n```text\n~/.local/state/openclaw-gateway-watchdog/model-probe-history.jsonl\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-model-probe.json\n~/.local/state/openclaw-gateway-watchdog/last-model-api-edge-probe.txt\n```\n\n排查凌晨模型 provider 超时，可以先用这组设置：\n\n```bash\nMODEL_PROBE_ENABLED=\"1\"\nMODEL_EDGE_PROBE_ENABLED=\"1\"\nMODEL_PROBE_INTERVAL=\"600\"\nMODEL_PROBE_TIMEOUT=\"120\"\nMODEL_PROBE_ACTION=\"log\"\nMODEL_PROBE_THINKING=\"off\"\nMODEL_PROBE_MESSAGE=\"Reply with exactly OK.\"\n```\n\n`MODEL_PROBE_ACTION` 支持：\n\n- `log`：只记录证据，默认策略。\n- `restart`：连续失败达到 `MODEL_PROBE_FAILURES_BEFORE_ACTION` 后重启 gateway，但会先确认外部网络不是全局断网。\n- `command`：连续失败后执行 `MODEL_PROBE_COMMAND`。\n\n这个功能会真实调用模型，可能消耗额度或费用。日志不会打印 API key，但会记录 provider/model 名称、耗时、退出状态和第一行错误摘要。\n`MODEL_EDGE_PROBE_ENABLED` 不使用凭据，也不调用 `/chat/completions`；它只检查 provider API 入口，比如 `https://api.deepseek.com`，是否能快速完成 DNS/TLS/HTTP 连接。\n\n## 安全边界\n\n这个项目不修改 OpenClaw 配置、不改微信插件源码、不处理消息内容。默认情况下它只做三件事：\n\n1. 探测本机 gateway 和外部 URL。\n2. 写自己的日志和状态文件。\n3. 在满足保护条件后执行配置好的 gateway 重启命令。\n\n可选模型探针只有在你显式开启后才会发起真实模型请求。\n\n分享日志前，请检查里面是否包含本机路径、服务名或私有通道地址。\n\n## 开源许可\n\nMIT-0。这个许可证符合 ClawHub skill 发布要求，也方便别人直接复用、改造和分发。\n\n## 发布到 ClawHub\n\n已发布包：\n\n- ClawHub：<https://clawhub.ai/zc-kama/gateway-resilience-guard>\n- Slug：`gateway-resilience-guard`\n\n本仓库带 `SKILL.md`，也可以重新发布为 OpenClaw skill：\n\n```bash\nclawhub publish . \\\n  --slug gateway-resilience-guard \\\n  --name \"OpenClaw Gateway Resilience Guard\" \\\n  --version 1.4.3 \\\n  --changelog \"Rebuild dashboard on local Fluent Web Components with denser pages, refined controls, motion, and bundled vendor assets\"\n```\n\n发布前需要先执行 `clawhub login` 完成 CLI 登录。\n\nFile v1.4.3:openclaw-plugin/openclaw.plugin.json\n\n{\n  \"id\": \"resilience-guard\",\n  \"name\": \"Resilience Guard\",\n  \"description\": \"Adds a Gateway Control UI entry that opens the external OpenClaw watchdog dashboard.\",\n  \"version\": \"1.4.3\",\n  \"configSchema\": {\n    \"type\": \"object\",\n    \"additionalProperties\": false,\n    \"properties\": {\n      \"dashboardUrl\": {\n        \"type\": \"string\",\n        \"default\": \"http://127.0.0.1:18790/\"\n      }\n    }\n  },\n  \"uiHints\": {\n    \"dashboardUrl\": {\n      \"label\": \"Dashboard URL\",\n      \"help\": \"External watchdog dashboard URL. It remains available even when Gateway is down.\",\n      \"placeholder\": \"http://127.0.0.1:18790/\"\n    }\n  }\n}\n\nFile v1.4.3:openclaw-plugin/package.json\n\n{\n  \"name\": \"@zc-kama/openclaw-resilience-guard\",\n  \"version\": \"1.4.3\",\n  \"type\": \"module\",\n  \"description\": \"OpenClaw plugin bridge for the external Gateway Resilience Guard dashboard.\",\n  \"license\": \"MIT-0\",\n  \"openclaw\": {\n    \"extensions\": [\n      \"./index.js\"\n    ],\n    \"compat\": {\n      \"pluginApi\": \">=2026.3.24-beta.2\",\n      \"minGatewayVersion\": \"2026.3.24-beta.2\"\n    },\n    \"build\": {\n      \"openclawVersion\": \"2026.5.18\",\n      \"pluginSdkVersion\": \"2026.5.18\"\n    }\n  }\n}\n\nArchive v1.4.2: 19 files, 60605 bytes\n\nFiles: CHANGELOG.md (4035b), dashboard/server.py (17990b), dashboard/static/app.js (20917b), dashboard/static/index.html (7827b), dashboard/static/styles.css (9866b), gateway-watchdog.ps1 (29656b), gateway-watchdog.sh (31639b), install-watchdog.ps1 (5020b), install-watchdog.sh (10659b), openclaw-plugin/index.js (1493b), openclaw-plugin/openclaw.plugin.json (624b), openclaw-plugin/package.json (485b), README-watchdog.md (1588b), README.md (15663b), README.zh-CN.md (15567b), SKILL.md (4331b), uninstall-watchdog.ps1 (1496b), uninstall-watchdog.sh (2009b), _meta.json (143b)\n\nFile v1.4.2:SKILL.md\n\n---\nname: gateway-resilience-guard\ndescription: OpenClaw Gateway Resilience Guard keeps Gateway, channels, logs, and optional model-provider probes observable after Wi-Fi changes, WSL sleep/resume, session expiry, provider timeouts, or partial outages. It adds a localhost dashboard with charts, event trends, language/theme switching, strategy presets, guarded action unlock, native OpenClaw health probes, restart safeguards, systemd, LaunchAgent, Task Scheduler, and an optional OpenClaw plugin bridge.\ntags:\n  - openclaw\n  - gateway\n  - watchdog\n  - resilience\n  - wechat\n  - wsl\n  - systemd\n  - macos\n  - windows\nrequirements:\n  tools:\n    - bash\n    - curl\n    - systemctl optional\n    - launchctl optional\n    - powershell optional\n    - openclaw recommended\npermissions:\n  - Writes a user-level systemd service, macOS LaunchAgent, or Windows scheduled task when requested.\n  - Writes config and logs under user-level config/state directories.\n  - Starts a localhost dashboard on 127.0.0.1:18790 when Python is available.\n  - Restarts OpenClaw Gateway through systemctl --user or openclaw gateway restart.\n  - Optional model probe sends real OpenClaw model requests only when explicitly enabled.\n---\n\n# OpenClaw Gateway Resilience Guard\n\nUse this skill when a user wants to keep OpenClaw Gateway and message channels online after network drops, WSL sleep/resume, macOS/Windows wake events, long-lived connection failures, or recurring model-provider timeouts.\n\n## Install\n\nLinux, WSL, or macOS:\n\n```bash\nbash install-watchdog.sh\n```\n\nThe installer works with defaults. It prompts for the main channel probe URL when running interactively, but pressing Enter is enough for the default WeChat probe.\n\nFor unattended install:\n\n```bash\nbash install-watchdog.sh --yes\n```\n\nFor a custom channel:\n\n```bash\nbash install-watchdog.sh --channel-url \"https://your-channel.example.com/health\"\n```\n\nWindows PowerShell:\n\n```powershell\npowershell -ExecutionPolicy Bypass -File .\\install-watchdog.ps1\n```\n\nAfter install, open the standalone dashboard:\n\n```text\nhttp://127.0.0.1:18790/\n```\n\nThe dashboard is served by the watchdog process, not Gateway, so it remains the recovery entry when Gateway is down.\n\nOptional Gateway-side bridge while Gateway is healthy:\n\n```bash\nopenclaw plugins install ./openclaw-plugin\nopenclaw plugins enable resilience-guard\nopenclaw gateway restart\n```\n\n## Operate\n\n```bash\nsystemctl --user status gateway-watchdog\njournalctl --user -u gateway-watchdog -f\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh\n```\n\nOn macOS, inspect `launchctl print gui/$(id -u)/ai.clawhub.gateway-resilience-guard`.\nOn Windows, inspect `Get-ScheduledTask -TaskName \"OpenClaw Gateway Resilience Guard\"`.\n\nIf user systemd is unavailable, the installer starts a direct background fallback and stores its pid under `~/.local/state/openclaw-gateway-watchdog/watchdog.pid`.\n\n## Safety Model\n\nThe watchdog restarts only after layered checks:\n\n1. OpenClaw native health probes: `openclaw gateway status --require-rpc`, `openclaw health --json --verbose`, and `openclaw status --deep` when available.\n2. Runtime diagnostics: `openclaw models status` plus `openclaw logs --plain` signal scanning for provider timeout, proxy/network, rate limit, auth, channel session, gateway degraded, config reload, and task runtime warnings.\n3. Local gateway health URL/TCP and main channel URL fallbacks.\n4. General network URLs to avoid restarting during whole-machine network failure.\n5. Optional model-provider probe via `openclaw agent --json`; default action is evidence logging only.\n6. Dashboard action token, backoff, and hourly restart limits to avoid restart storms.\n\nModel probing is disabled by default because it consumes real provider quota. To diagnose provider timeouts, set `MODEL_PROBE_ENABLED=1`, keep `MODEL_PROBE_ACTION=log`, and inspect `model-probe-history.jsonl` the next day.\nDiagnostic log scanning is enabled by default, but `OPENCLAW_DIAG_ACTION=log` keeps it non-invasive unless the operator explicitly opts into a restart or custom command.\nUse the dashboard strategy buttons for common modes: observe, overnight diagnosis, channel recovery, and conservative circuit breaker.\n\nTell users to review `~/.config/openclaw-gateway-watchdog/watchdog.env` before publishing, sharing logs, or reporting issues.\n\nFile v1.4.2:README.md\n\n# OpenClaw Gateway Resilience Guard\n\nExternal recovery guard for OpenClaw Gateway and long-lived channel plugins such as `openclaw-weixin`.\n\nThis project is for people who run OpenClaw continuously on Linux, WSL, macOS, or Windows and need the gateway to recover from channel disconnects, network sleep/resume, and long-lived session failures without babysitting the terminal.\n\n## Problem\n\nOpenClaw Gateway can still be alive while an individual channel is no longer healthy. This is common after laptop sleep, Wi-Fi changes, WSL network hiccups, or long idle periods.\n\nThe WeChat plugin is especially sensitive because it depends on a long-poll `getUpdates` loop. In the upstream `Tencent/openclaw-weixin` code, the monitor has a limited retry loop and session guard:\n\n- `monitor.ts` defines `MAX_CONSECUTIVE_FAILURES = 3` and `BACKOFF_DELAY_MS = 30_000`.\n- `session-guard.ts` defines `SESSION_PAUSE_DURATION_MS = 60 * 60 * 1000` and `SESSION_EXPIRED_ERRCODE = -14`.\n- Issue [Tencent/openclaw-weixin#141](https://github.com/Tencent/openclaw-weixin/issues/141) reports that after a config hot reload the monitor can end without starting again; the workaround is `openclaw gateway restart`.\n- Issue [Tencent/openclaw-weixin#155](https://github.com/Tencent/openclaw-weixin/issues/155) reports that `errcode=-14` can enter a 60-minute session pause loop and block outbound messages.\n\nThis watchdog does not replace the official plugin. It is an external safety net: when the gateway or channel stops behaving like a live system, it restarts the gateway with guardrails.\n\n## Design\n\nThe script uses a layered health model before it restarts anything:\n\n| Layer | Probe | Purpose |\n| --- | --- | --- |\n| Gateway | `openclaw gateway status --json --require-rpc`, local health URL, local TCP port, service/process fallback | Detect whether OpenClaw Gateway is down or locally unreachable. |\n| Channel | `openclaw health --json --verbose`, `openclaw status --deep`, optional `openclaw channels status --probe`, then URL fallback | Prefer OpenClaw's own per-channel health model, then fall back to a configured URL when native probes are unavailable. |\n| Runtime diagnostics | `openclaw models status --json`, `openclaw logs --plain`, warning classification | Separate provider, proxy/network, auth/rate-limit, gateway, channel-session, config reload, and task-runtime evidence before choosing an action. |\n| Model API, optional | `openclaw agent --json` with the configured model provider | Detect whether the configured model path is timing out while Gateway and channels still look healthy. Disabled by default because it makes real model calls. |\n| Network | multiple independent URLs, default Baidu/QQ/Weixin | Avoid restarting the gateway during whole-machine or ISP network failure. |\n\nOnly gateway failures restart immediately. Channel failures go through confirmation, network split-brain protection, exponential backoff, and restart-rate limits.\nModel probe failures default to evidence logging only; users can opt in to restart or a custom command after consecutive failures.\n\n## Recovery Policy\n\n- Gateway down: restart immediately.\n- Gateway running but channel probe fails: confirm with general network probes.\n- General network also fails: do nothing except wait; restarting will not fix an offline machine.\n- General network works but channel stays down: wait with exponential backoff, re-check, then restart.\n- OpenClaw warning logs: classify and record evidence first; default action is log-only.\n- Five consecutive successful channel probes reset the failure state.\n- Restart storm protection limits gateway restarts per hour.\n- Night hours can use a slower probe interval to reduce noise.\n\n## Files\n\n| File | Purpose |\n| --- | --- |\n| `gateway-watchdog.sh` | Main daemon loop: probes, backoff, restart decisions, log rotation, single-instance lock. |\n| `gateway-watchdog.ps1` | Windows-native daemon loop for Task Scheduler. |\n| `dashboard/` | Standalone local Web UI and API served by the watchdog, independent of Gateway. |\n| `openclaw-plugin/` | Optional native OpenClaw plugin bridge that redirects `/resilience-guard` to the standalone dashboard. |\n| `install-watchdog.sh` | Linux/WSL/macOS installer: copies files, writes config, creates systemd user service or macOS LaunchAgent. |\n| `install-watchdog.ps1` | Windows installer: creates config and a Task Scheduler job. |\n| `uninstall-watchdog.sh` | Stops and removes the service and installed scripts. |\n| `uninstall-watchdog.ps1` | Windows uninstaller. |\n| `SKILL.md` | ClawHub/OpenClaw skill metadata and operator instructions. |\n| `README.zh-CN.md` | Chinese documentation. |\n\n## Install\n\nInstall from ClawHub:\n\n```bash\nclawhub install gateway-resilience-guard\n```\n\nOr use this repository directly.\n\nLinux, WSL, or macOS:\n\n```bash\nbash install-watchdog.sh\n```\n\nFor unattended install:\n\n```bash\nbash install-watchdog.sh --yes\n```\n\nFor a custom channel probe:\n\n```bash\nbash install-watchdog.sh --channel-url \"https://your-channel.example.com/health\"\n```\n\nWindows PowerShell:\n\n```powershell\npowershell -ExecutionPolicy Bypass -File .\\install-watchdog.ps1\n```\n\n## Dashboard\n\nThe installer enables a standalone local dashboard:\n\n```text\nhttp://127.0.0.1:18790/\n```\n\nThis dashboard is served by the watchdog, not by OpenClaw Gateway. If Gateway is down, the dashboard can still open and show the last known evidence.\n\nIt includes:\n\n- Gateway, channel, network, OpenClaw log, and model-provider status.\n- Category charts for provider timeout, proxy/network, rate-limit, auth, channel session, Gateway degraded, config reload, and task runtime warnings.\n- Sidebar view switching for Overview, Trends, Strategy, Logs, and Config so monitoring, actions, and detail evidence are separated.\n- Event trend chart with separate lanes for API failures, log warnings, successful model probes, and healthy diagnostics.\n- Chinese/English language switching and Light, Dark, Ocean, and Forest themes.\n- Status-file freshness checks so stale data is obvious.\n- Quick strategy buttons: observe, overnight diagnosis, channel recovery, and conservative circuit breaker.\n- Guarded actions with an unlock flow: run diagnostics, restart Gateway, apply presets, and export a diagnostic JSON bundle.\n\nDashboard actions are bound to localhost and protected with the generated `DASHBOARD_TOKEN`. The token is injected only into the same-origin dashboard page. A random token is written during install; the value is not shown in logs.\n\nOptional OpenClaw plugin bridge:\n\n```bash\nopenclaw plugins install ./openclaw-plugin\nopenclaw plugins enable resilience-guard\nopenclaw gateway restart\n```\n\nThen open the Gateway route while Gateway is healthy:\n\n```text\nhttp://127.0.0.1:18789/resilience-guard\n```\n\nThat route redirects to the external dashboard. It is a convenience entry only; the external dashboard remains the recovery entry when Gateway is unavailable.\n\nThe installer writes:\n\n- scripts to `~/.local/share/openclaw-gateway-watchdog`;\n- config to `~/.config/openclaw-gateway-watchdog/watchdog.env`;\n- logs to `~/.local/state/openclaw-gateway-watchdog/watchdog.log`;\n- a Linux/WSL user service to `~/.config/systemd/user/gateway-watchdog.service`;\n- a macOS LaunchAgent to `~/Library/LaunchAgents/ai.clawhub.gateway-resilience-guard.plist`;\n- a Windows scheduled task named `OpenClaw Gateway Resilience Guard`.\n\nIf user systemd is unavailable, the installer starts a direct background fallback process and stores its pid in the state directory.\n\n## Manage\n\n```bash\nsystemctl --user status gateway-watchdog\njournalctl --user -u gateway-watchdog -f\nsystemctl --user restart gateway-watchdog\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh\n```\n\nmacOS:\n\n```bash\nlaunchctl print gui/$(id -u)/ai.clawhub.gateway-resilience-guard\ntail -f ~/.local/state/openclaw-gateway-watchdog/watchdog.log\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh\n```\n\nWindows:\n\n```powershell\nGet-ScheduledTask -TaskName \"OpenClaw Gateway Resilience Guard\"\nGet-Content \"$env:LOCALAPPDATA\\openclaw-gateway-watchdog\\watchdog.log\" -Wait\npowershell -ExecutionPolicy Bypass -File \"$env:LOCALAPPDATA\\openclaw-gateway-watchdog\\uninstall-watchdog.ps1\"\n```\n\nRemove config and logs too:\n\n```bash\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh --purge\n```\n\n## Configuration\n\nMost users can keep the generated defaults. Advanced settings live in:\n\n```text\n~/.config/openclaw-gateway-watchdog/watchdog.env\n```\n\nCommon keys:\n\n```bash\nGATEWAY_SERVICE=\"openclaw-gateway\"\nGATEWAY_HEALTH_URL=\"http://127.0.0.1:18789/healthz\"\nGATEWAY_HOST=\"127.0.0.1\"\nGATEWAY_PORT=\"18789\"\nCHANNEL_URL=\"https://ilinkai.weixin.qq.com\"\nNETWORK_URLS=\"https://www.baidu.com https://www.qq.com https://api.weixin.qq.com\"\nRESTART_COMMAND=\"systemctl --user restart openclaw-gateway\"\nOPENCLAW_NATIVE_PROBES=\"auto\"\nOPENCLAW_HEALTH_TIMEOUT_MS=\"12000\"\nOPENCLAW_GATEWAY_STRICT=\"0\"\nOPENCLAW_CHANNELS_PROBE=\"1\"\nOPENCLAW_DIAG_ENABLED=\"1\"\nOPENCLAW_DIAG_INTERVAL=\"300\"\nOPENCLAW_LOG_SCAN_ENABLED=\"1\"\nOPENCLAW_LOG_LIMIT=\"200\"\nOPENCLAW_LOG_SIGNAL_LIMIT=\"40\"\nOPENCLAW_LOG_TIMEOUT_MS=\"15000\"\nOPENCLAW_LOG_WARN_PATTERNS=\"fetch failed|fetch timeout|LLM idle timeout|model silent|...\"\nOPENCLAW_DIAG_ACTION=\"log\"\nOPENCLAW_DIAG_FAILURES_BEFORE_ACTION=\"2\"\nOPENCLAW_DIAG_COMMAND=\"\"\nDASHBOARD_ENABLED=\"1\"\nDASHBOARD_HOST=\"127.0.0.1\"\nDASHBOARD_PORT=\"18790\"\nDASHBOARD_ACTIONS_ENABLED=\"1\"\nDASHBOARD_TOKEN=\"generated-at-install\"\nDASHBOARD_DIR=\"~/.local/share/openclaw-gateway-watchdog/dashboard\"\nMODEL_PROBE_ENABLED=\"0\"\nMODEL_EDGE_PROBE_ENABLED=\"1\"\nMODEL_PROBE_INTERVAL=\"1800\"\nMODEL_PROBE_TIMEOUT=\"120\"\nMODEL_PROBE_FAILURES_BEFORE_ACTION=\"2\"\nMODEL_PROBE_ACTION=\"log\"\nMODEL_PROBE_COMMAND=\"\"\nMODEL_PROBE_MODEL=\"\"\nMODEL_PROBE_THINKING=\"off\"\nMODEL_PROBE_SESSION_ID=\"watchdog-model-probe\"\nMODEL_PROBE_MESSAGE=\"Reply with exactly OK.\"\nBASE_INTERVAL=\"60\"\nNIGHT_INTERVAL=\"300\"\nMAX_INTERVAL=\"1800\"\nCHANNEL_FAILURES_BEFORE_RESTART=\"2\"\nSUCCESS_COUNT_TO_RESET=\"5\"\nMAX_RESTARTS_PER_HOUR=\"6\"\n```\n\nUse `RESTART_COMMAND` if your OpenClaw install is not managed by a user-level systemd unit. Example:\n\n```bash\nRESTART_COMMAND=\"openclaw gateway restart\"\n```\n\nOn Windows, the generated config is JSON:\n\n```text\n%APPDATA%\\openclaw-gateway-watchdog\\watchdog.json\n```\n\nSet `OpenClawNativeProbes` to `false` if your OpenClaw CLI is too old for `openclaw health` or `openclaw status --deep`.\n\n### OpenClaw diagnostics and log signals\n\nThe watchdog does more than ping one URL. Every `OPENCLAW_DIAG_INTERVAL` seconds it collects an OpenClaw diagnostic snapshot:\n\n```text\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-gateway-status.json\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-health.json\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-model-status.json\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-status-deep.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-logs.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-log-signals.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-log-signal-categories.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-diagnostics.jsonl\n```\n\n`last-openclaw-log-signals.txt` is filtered from `openclaw logs --plain`. It classifies common failure families such as `provider_timeout`, `proxy_or_network`, `provider_rate_limit`, `provider_auth`, `abort_stuck`, `memory_dream_timeout`, `channel_session`, `gateway_degraded`, `config_reload`, and `task_runtime`.\n\nDefault action is `OPENCLAW_DIAG_ACTION=\"log\"` because warnings are evidence, not always proof that a restart is correct. If you explicitly set `OPENCLAW_DIAG_ACTION=\"restart\"` or `command`, the action only runs after consecutive diagnostic warnings and only when the general network probes still pass.\n\nPractical interpretation:\n\n| Evidence | Likely scope | Default strategy |\n| --- | --- | --- |\n| Gateway status/health is down | Gateway process or RPC path | Restart Gateway immediately through the normal gateway policy. |\n| Channel probe fails, network probes pass | Channel/session path | Backoff, re-check, then restart Gateway if still failed. |\n| OpenClaw logs show provider timeout, model probe also fails | Provider/API path | Log evidence; optional custom action. A Gateway restart may not fix provider outage. |\n| OpenClaw logs show provider timeout, model probe succeeds | OpenClaw runtime, task, session, or specific request path | Keep evidence, inspect logs; avoid blaming the provider alone. |\n| Logs show proxy/DNS/TLS errors | Local proxy, DNS, TLS, or ISP route | Log evidence and avoid restart storms; fix network/proxy route first. |\n| Logs show session expiry or monitor stopped | Channel plugin/session | Gateway restart is often useful after confirmation. |\n\n### Optional model probe\n\nSet `MODEL_PROBE_ENABLED=\"1\"` when you need to prove whether failures are in the model-provider path instead of Gateway or channel health.\n\nWhen enabled, the watchdog first reads OpenClaw's configured model provider and probes its provider `baseUrl` without credentials. Then it runs the end-to-end model probe:\n\n```bash\nopenclaw agent --session-id \"$MODEL_PROBE_SESSION_ID\" \\\n  --thinking \"$MODEL_PROBE_THINKING\" \\\n  --timeout \"$MODEL_PROBE_TIMEOUT\" \\\n  --json \\\n  --message \"$MODEL_PROBE_MESSAGE\"\n```\n\nIf `MODEL_PROBE_MODEL` is empty, OpenClaw's configured default model is used. Results are written to the main log and to:\n\n```text\n~/.local/state/openclaw-gateway-watchdog/model-probe-history.jsonl\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-model-probe.json\n~/.local/state/openclaw-gateway-watchdog/last-model-api-edge-probe.txt\n```\n\nRecommended diagnostic settings for overnight provider issues:\n\n```bash\nMODEL_PROBE_ENABLED=\"1\"\nMODEL_EDGE_PROBE_ENABLED=\"1\"\nMODEL_PROBE_INTERVAL=\"600\"\nMODEL_PROBE_TIMEOUT=\"120\"\nMODEL_PROBE_ACTION=\"log\"\nMODEL_PROBE_THINKING=\"off\"\nMODEL_PROBE_MESSAGE=\"Reply with exactly OK.\"\n```\n\n`MODEL_PROBE_ACTION` can be:\n\n- `log`: record evidence only. This is the default.\n- `restart`: restart Gateway after `MODEL_PROBE_FAILURES_BEFORE_ACTION` consecutive model failures, but only if general network probes still pass.\n- `command`: run `MODEL_PROBE_COMMAND` after consecutive failures.\n\nThis feature sends real model requests and may consume provider quota or money. It does not print API keys, but it does store provider/model names, timing, exit status, and the first non-empty error line.\n`MODEL_EDGE_PROBE_ENABLED` does not use credentials and does not call `/chat/completions`; it only checks whether the provider API edge such as `https://api.deepseek.com` is reachable quickly.\n\n## Safety Notes\n\nThis project intentionally avoids destructive behavior. It does not edit OpenClaw configuration, tokens, sessions, or plugin files. By default it only probes URLs and restarts the gateway through the configured command. The optional model probe makes real model calls only after you explicitly enable it.\n\nBefore sharing logs, review them for local paths, service names, and channel URLs.\n\n## License\n\nMIT-0. This matches ClawHub's skill publishing requirement and allows reuse without attribution requirements.\n\n## Publish To ClawHub\n\nPublished package:\n\n- ClawHub: <https://clawhub.ai/zc-kama/gateway-resilience-guard>\n- Slug: `gateway-resilience-guard`\n\nThis repository includes `SKILL.md`, so it can also be republished as an OpenClaw skill bundle:\n\n```bash\nclawhub publish . \\\n  --slug gateway-resilience-guard \\\n  --name \"OpenClaw Gateway Resilience Guard\" \\\n  --version 1.4.2 \\\n  --changelog \"Redesign dashboard navigation, alignment, trend chart axis, strategy controls, and visual system\"\n```\n\nClawHub requires CLI authentication. Run `clawhub login` first.\n\nFile v1.4.2:_meta.json\n\n{\n  \"ownerId\": \"kn70e6hn9pv8yrpzjn5kqczygd8708vw\",\n  \"slug\": \"gateway-resilience-guard\",\n  \"version\": \"1.4.2\",\n  \"publishedAt\": 1779380200353\n}\n\nFile v1.4.2:CHANGELOG.md\n\n# Changelog\n\n## 1.4.2\n\n- Rework the dashboard sidebar into real view switching for Overview, Trends, Strategy, Logs, and Config instead of one long page.\n- Tighten the dashboard grid, panel sizing, button alignment, and sidebar polish for a cleaner operations-console layout.\n- Fix the trend chart axis label collision by removing the redundant x-axis title and reserving more space for tick labels.\n- Keep charts stable across refreshes by drawing only the active view and ignoring hidden canvases.\n- Move strategy explanations into centered button content and remove the sidebar localhost note.\n\n## 1.4.1\n\n- Redesign the dashboard with a sidebar layout, stronger visual grouping, and selectable Light, Dark, Ocean, and Forest themes.\n- Add Chinese/English language switching for dashboard labels, buttons, hints, legends, and notifications.\n- Fix canvas redraw sizing so repeated refreshes do not stretch dashboard panels.\n- Replace \"overnight timeline\" with a general event trend chart that includes lanes, legends, and time-axis ticks.\n- Auto-follow the newest watchdog log lines while keeping a toggle for manual scrolling.\n- Add an unlock/lock action flow and strategy hover hints so guarded controls are discoverable.\n\n## 1.4.0\n\n- Add a standalone local dashboard at `http://127.0.0.1:18790` that remains available when OpenClaw Gateway is down.\n- Add dashboard API summaries, log-signal charts, overnight timelines, status file freshness, safe config summaries, diagnostic export, and quick strategy presets.\n- Add guarded dashboard actions for run diagnostics, restart Gateway, and apply presets using a generated local action token.\n- Add an OpenClaw native plugin bridge that registers `/resilience-guard` and redirects to the external dashboard when Gateway is healthy.\n- Install dashboard files on Linux/WSL, macOS, and Windows.\n\n## 1.3.1\n\n- Publish a ClawHub package that includes the Windows PowerShell installer, watchdog, and uninstaller files.\n\n## 1.3.0\n\n- Add an opt-in model-provider probe that uses `openclaw agent --json` against the configured OpenClaw model path.\n- Add a no-credential provider edge probe that checks the configured provider `baseUrl` before the end-to-end model call.\n- Add OpenClaw runtime diagnostics that snapshot gateway status, health, model status, deep status, and recent OpenClaw logs.\n- Classify log signals for provider timeout, proxy/network, rate limit, auth, channel session, gateway degraded, config reload, task runtime, and related warning families.\n- Record model probe evidence in the main log and `model-probe-history.jsonl` without logging API keys.\n- Add configurable model failure actions: log-only, gateway restart, or a custom command after consecutive failures.\n- Document safe overnight diagnostics for separating Gateway/channel failures from model-provider timeouts.\n\n## 1.2.0\n\n- Add OpenClaw-native health probing via `openclaw gateway status --json --require-rpc`, `openclaw health --json --verbose`, and `openclaw status --deep`.\n- Add Windows Task Scheduler support with native PowerShell install, watchdog, and uninstall scripts.\n- Add macOS LaunchAgent support to the Bash installer and uninstaller.\n- Document cross-platform installation and native probe configuration.\n\n## 1.1.0\n\n- Rename the ClawHub package to `gateway-resilience-guard`.\n- Expand the public summary to describe the layered probes, restart guardrails, and advantages over simple restart loops or one-URL monitors.\n\n## 1.0.1\n\n- Add the published ClawHub URL and install command to the README files.\n\n## 1.0.0\n\n- First public zero-config release.\n- Install from the current folder instead of a hard-coded workspace path.\n- Generate user config and systemd service automatically.\n- Add layered gateway/channel/network probes.\n- Add exponential backoff, hourly restart limits, lock protection, and log rotation.\n- Reset failure state after five consecutive successful channel probes.\n- Add ClawHub-ready `SKILL.md`.\n- Initial ClawHub publication used a temporary slug before the 1.1.0 rename.\n\nFile v1.4.2:README-watchdog.md\n\n# OpenClaw Gateway 看门狗\n\n这是项目的中文快捷入口。完整中文文档见 [README.zh-CN.md](README.zh-CN.md)，英文文档见 [README.md](README.md)。\n\n最简单安装：\n\n```bash\nbash install-watchdog.sh\n```\n\nWindows：\n\n```powershell\npowershell -ExecutionPolicy Bypass -File .\\install-watchdog.ps1\n```\n\n默认配置即可启动；需要指定自己的通道地址时：\n\n```bash\nbash install-watchdog.sh --channel-url \"https://你的通道地址/health\"\n```\n\n安装后还有一个独立图形化控制台：\n\n```text\nhttp://127.0.0.1:18790/\n```\n\n它由 watchdog 自己提供，不依赖 OpenClaw Gateway。Gateway 挂掉时，这个页面仍然可以看状态、图表、策略、日志和诊断导出。Dashboard 支持中英文切换、主题切换、事件趋势图、异常分类图和受保护操作解锁。\n\n安装后它会默认采集 OpenClaw 诊断快照和日志信号，包括 gateway status、health、models status、status --deep 和 `openclaw logs --plain` 的 WARN/ERROR 摘要。关键证据在：\n\n```text\n~/.local/state/openclaw-gateway-watchdog/watchdog.log\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-diagnostics.jsonl\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-log-signals.txt\n```\n\n可选模型探针默认关闭。需要排查模型 API 是否在某个时段超时时，可以在配置里显式开启 `MODEL_PROBE_ENABLED=\"1\"`；它会发送极小的 `openclaw agent --json` 请求，并把结果写入 watchdog 日志和 `model-probe-history.jsonl`。默认动作仍是只记录证据，不会擅自改 OpenClaw 配置。\n\nFile v1.4.2:README.zh-CN.md\n\n# OpenClaw Gateway Resilience Guard\n\nOpenClaw Gateway 外部恢复守护脚本，面向 `openclaw-weixin` 这类需要长期在线的通道。\n\n这个项目解决的是一个很具体的运维问题：OpenClaw Gateway 进程还活着，但通道、长轮询、WSL/系统网络或 session 状态已经坏掉，导致消息收不到、发不出，最后只能手动 `openclaw gateway restart`。\n\n## 问题背景\n\nOpenClaw Gateway 和通道插件是长连接系统。电脑睡眠、切换 Wi-Fi、WSL 网络重建、iLink 长时间空闲、配置热加载，都可能让“进程存活”和“通道可用”变成两件事。\n\n以官方 `Tencent/openclaw-weixin` 为例，源码和 issue 里能看到几个已知边界：\n\n- `monitor.ts` 里 `MAX_CONSECUTIVE_FAILURES = 3`，失败后 `BACKOFF_DELAY_MS = 30_000`，也就是插件内部主要是 3 次失败后的 30 秒退避。\n- `session-guard.ts` 里 `SESSION_PAUSE_DURATION_MS = 60 * 60 * 1000`，`SESSION_EXPIRED_ERRCODE = -14`，session 过期会进入 60 分钟暂停窗口。\n- [Tencent/openclaw-weixin#141](https://github.com/Tencent/openclaw-weixin/issues/141) 记录了配置热加载后 Monitor 结束但不再启动，临时处理方式是手动重启 gateway。\n- [Tencent/openclaw-weixin#155](https://github.com/Tencent/openclaw-weixin/issues/155) 记录了 `errcode=-14` 后进入 60 分钟循环暂停、出站消息被阻塞的问题。\n\n所以这个项目不是替换官方插件，而是在外面加一层独立 watchdog：当通道或 gateway 进入“看起来还活着，实际上已经不能工作”的状态时，用更保守的探测和熔断策略自动恢复。\n\n## 工作原理\n\n看门狗使用分层健康检查，从浅到深判断是否真的需要重启：\n\n| 层级 | 检查对象 | 作用 |\n| --- | --- | --- |\n| Gateway 本机状态 | `openclaw gateway status --json --require-rpc`、本机 health URL、本机 TCP 端口、服务/进程兜底 | 判断 OpenClaw Gateway 是否已经挂掉或本机不可达。 |\n| 通道状态 | `openclaw health --json --verbose`、`openclaw status --deep`、可选 `openclaw channels status --probe`，最后才是 URL 兜底 | 优先使用 OpenClaw 自己的全通道健康模型；CLI 不支持时再退回 URL 探测。 |\n| 运行时诊断 | `openclaw models status --json`、`openclaw logs --plain`、WARN/ERROR 分类 | 在动作前区分 provider、代理/网络、鉴权/限流、gateway、通道 session、配置热加载和任务运行时证据。 |\n| 模型 API，可选 | 使用 `openclaw agent --json` 走 OpenClaw 当前配置的模型 provider | 判断 Gateway 和通道都健康时，真正卡住的是不是模型 API 链路。默认关闭，因为它会真实消耗模型调用。 |\n| 外部网络状态 | 百度、QQ、微信 API 等多个独立 URL | 排除全局断网，避免电脑没网时误重启 gateway。 |\n\n核心策略是：gateway 真挂了就立即重启；通道不通时先确认不是全局断网；网络正常但通道持续失败，才进入退避等待和重启流程。\n模型探针默认只记录证据日志；如果你显式配置，也可以在连续失败后重启 gateway 或执行自定义命令。\n\n## 恢复策略\n\n- Gateway 本机健康检查失败：立即重启。\n- 通道 URL 失败：累计失败次数，进入故障流程。\n- 外部网络全部失败：认为是全局断网，只等待，不重启。\n- 外部网络正常但通道仍失败：指数退避后再次确认，再重启 gateway。\n- OpenClaw 日志出现 WARN/ERROR：先分类和记录证据，默认只写日志。\n- 连续 5 次通道探测成功后，清空失败状态。\n- 每小时最多重启固定次数，防止网络抖动时形成重启风暴。\n- 深夜可以降低检查频率，减少无意义日志。\n\n## 文件说明\n\n| 文件 | 作用 |\n| --- | --- |\n| `gateway-watchdog.sh` | 主守护脚本，负责探测、退避、熔断、日志、单实例锁和重启决策。 |\n| `gateway-watchdog.ps1` | Windows 原生守护脚本，用于 Task Scheduler。 |\n| `dashboard/` | 独立本地 Web UI 和 API，由 watchdog 自己提供，不依赖 Gateway。 |\n| `openclaw-plugin/` | 可选 OpenClaw 原生插件桥接入口，把 `/resilience-guard` 跳转到独立 dashboard。 |\n| `install-watchdog.sh` | Linux/WSL/macOS 安装脚本，自动复制文件、生成配置、创建 systemd 用户服务或 macOS LaunchAgent。 |\n| `install-watchdog.ps1` | Windows 安装脚本，创建配置和计划任务。 |\n| `uninstall-watchdog.sh` | 卸载脚本，停止服务并删除安装目录。 |\n| `uninstall-watchdog.ps1` | Windows 卸载脚本。 |\n| `SKILL.md` | ClawHub/OpenClaw 技能元数据和使用说明。 |\n| `README.md` | 英文文档。 |\n\n## 安装\n\n通过 ClawHub 安装：\n\n```bash\nclawhub install gateway-resilience-guard\n```\n\n也可以直接使用本仓库。\n\nLinux、WSL 或 macOS：\n\n```bash\nbash install-watchdog.sh\n```\n\n无人值守安装：\n\n```bash\nbash install-watchdog.sh --yes\n```\n\n指定自己的通道探测地址：\n\n```bash\nbash install-watchdog.sh --channel-url \"https://你的通道地址/health\"\n```\n\nWindows PowerShell：\n\n```powershell\npowershell -ExecutionPolicy Bypass -File .\\install-watchdog.ps1\n```\n\n## 图形化 Dashboard\n\n安装后会启用一个独立本地 dashboard：\n\n```text\nhttp://127.0.0.1:18790/\n```\n\n这个页面由 watchdog 自己提供，不依赖 OpenClaw Gateway。所以 Gateway 挂掉时，它仍然可以打开，看到最后一次诊断证据。\n\n它包含：\n\n- Gateway、通道、外部网络、OpenClaw 日志、模型 provider 的分层状态。\n- provider timeout、代理/网络、限流、鉴权、通道 session、Gateway degraded、配置热加载、任务运行时异常的分类图表。\n- 侧边栏分页：总览、趋势、策略、日志、配置分开呈现，避免把监控、操作和明细堆在一个长页面里。\n- 事件趋势图，把 API 失败、日志 WARN、模型探针成功、健康诊断分成不同泳道，并保留清晰的时间刻度。\n- 中英文语言切换，以及 Light、Dark、Ocean、Forest 四套主题。\n- 状态文件新鲜度，避免把过期快照误认为当前状态。\n- 快速策略按钮：观察模式、夜间诊断、通道恢复、保守熔断。\n- 带解锁流程的受保护操作：立即诊断、重启 Gateway、应用策略、导出诊断 JSON。\n\nDashboard 操作只绑定 localhost，并使用安装时生成的 `DASHBOARD_TOKEN` 保护。token 只注入同源页面，不会写入日志。\n\n可选 OpenClaw 插件桥接入口：\n\n```bash\nopenclaw plugins install ./openclaw-plugin\nopenclaw plugins enable resilience-guard\nopenclaw gateway restart\n```\n\nGateway 正常时可以打开：\n\n```text\nhttp://127.0.0.1:18789/resilience-guard\n```\n\n这个路由会跳转到外部 dashboard。它只是方便入口；真正救急的入口仍然是 `http://127.0.0.1:18790/`。\n\n安装后会生成：\n\n- 脚本目录：`~/.local/share/openclaw-gateway-watchdog`\n- 配置文件：`~/.config/openclaw-gateway-watchdog/watchdog.env`\n- 日志文件：`~/.local/state/openclaw-gateway-watchdog/watchdog.log`\n- Linux/WSL systemd 用户服务：`~/.config/systemd/user/gateway-watchdog.service`\n- macOS LaunchAgent：`~/Library/LaunchAgents/ai.clawhub.gateway-resilience-guard.plist`\n- Windows 计划任务：`OpenClaw Gateway Resilience Guard`\n\n如果当前环境没有 user systemd，安装脚本会退回到后台进程模式，并把 pid 写到状态目录。\n\n## 管理命令\n\n```bash\nsystemctl --user status gateway-watchdog\njournalctl --user -u gateway-watchdog -f\nsystemctl --user restart gateway-watchdog\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh\n```\n\nmacOS：\n\n```bash\nlaunchctl print gui/$(id -u)/ai.clawhub.gateway-resilience-guard\ntail -f ~/.local/state/openclaw-gateway-watchdog/watchdog.log\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh\n```\n\nWindows：\n\n```powershell\nGet-ScheduledTask -TaskName \"OpenClaw Gateway Resilience Guard\"\nGet-Content \"$env:LOCALAPPDATA\\openclaw-gateway-watchdog\\watchdog.log\" -Wait\npowershell -ExecutionPolicy Bypass -File \"$env:LOCALAPPDATA\\openclaw-gateway-watchdog\\uninstall-watchdog.ps1\"\n```\n\n连配置和日志一起删除：\n\n```bash\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh --purge\n```\n\n## 配置\n\n普通用户一般不用改。高级配置在：\n\n```text\n~/.config/openclaw-gateway-watchdog/watchdog.env\n```\n\n常用项：\n\n```bash\nGATEWAY_SERVICE=\"openclaw-gateway\"\nGATEWAY_HEALTH_URL=\"http://127.0.0.1:18789/healthz\"\nGATEWAY_HOST=\"127.0.0.1\"\nGATEWAY_PORT=\"18789\"\nCHANNEL_URL=\"https://ilinkai.weixin.qq.com\"\nNETWORK_URLS=\"https://www.baidu.com https://www.qq.com https://api.weixin.qq.com\"\nRESTART_COMMAND=\"systemctl --user restart openclaw-gateway\"\nOPENCLAW_NATIVE_PROBES=\"auto\"\nOPENCLAW_HEALTH_TIMEOUT_MS=\"12000\"\nOPENCLAW_GATEWAY_STRICT=\"0\"\nOPENCLAW_CHANNELS_PROBE=\"1\"\nOPENCLAW_DIAG_ENABLED=\"1\"\nOPENCLAW_DIAG_INTERVAL=\"300\"\nOPENCLAW_LOG_SCAN_ENABLED=\"1\"\nOPENCLAW_LOG_LIMIT=\"200\"\nOPENCLAW_LOG_SIGNAL_LIMIT=\"40\"\nOPENCLAW_LOG_TIMEOUT_MS=\"15000\"\nOPENCLAW_LOG_WARN_PATTERNS=\"fetch failed|fetch timeout|LLM idle timeout|model silent|...\"\nOPENCLAW_DIAG_ACTION=\"log\"\nOPENCLAW_DIAG_FAILURES_BEFORE_ACTION=\"2\"\nOPENCLAW_DIAG_COMMAND=\"\"\nDASHBOARD_ENABLED=\"1\"\nDASHBOARD_HOST=\"127.0.0.1\"\nDASHBOARD_PORT=\"18790\"\nDASHBOARD_ACTIONS_ENABLED=\"1\"\nDASHBOARD_TOKEN=\"安装时生成\"\nDASHBOARD_DIR=\"~/.local/share/openclaw-gateway-watchdog/dashboard\"\nMODEL_PROBE_ENABLED=\"0\"\nMODEL_EDGE_PROBE_ENABLED=\"1\"\nMODEL_PROBE_INTERVAL=\"1800\"\nMODEL_PROBE_TIMEOUT=\"120\"\nMODEL_PROBE_FAILURES_BEFORE_ACTION=\"2\"\nMODEL_PROBE_ACTION=\"log\"\nMODEL_PROBE_COMMAND=\"\"\nMODEL_PROBE_MODEL=\"\"\nMODEL_PROBE_THINKING=\"off\"\nMODEL_PROBE_SESSION_ID=\"watchdog-model-probe\"\nMODEL_PROBE_MESSAGE=\"Reply with exactly OK.\"\nBASE_INTERVAL=\"60\"\nNIGHT_INTERVAL=\"300\"\nMAX_INTERVAL=\"1800\"\nCHANNEL_FAILURES_BEFORE_RESTART=\"2\"\nSUCCESS_COUNT_TO_RESET=\"5\"\nMAX_RESTARTS_PER_HOUR=\"6\"\n```\n\n如果你的 OpenClaw 不是 systemd 用户服务管理，可以显式指定重启命令：\n\n```bash\nRESTART_COMMAND=\"openclaw gateway restart\"\n```\n\nWindows 的配置是 JSON：\n\n```text\n%APPDATA%\\openclaw-gateway-watchdog\\watchdog.json\n```\n\n如果你的 OpenClaw CLI 版本太旧，不支持 `openclaw health` 或 `openclaw status --deep`，可以把 `OpenClawNativeProbes` 设为 `false`。\n\n### OpenClaw 诊断和日志信号\n\n这个 watchdog 不只是 ping 一个 URL。它会每隔 `OPENCLAW_DIAG_INTERVAL` 秒采集一组 OpenClaw 运行快照：\n\n```text\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-gateway-status.json\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-health.json\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-model-status.json\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-status-deep.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-logs.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-log-signals.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-log-signal-categories.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-diagnostics.jsonl\n```\n\n`last-openclaw-log-signals.txt` 来自 `openclaw logs --plain` 的过滤结果，会把常见异常归类成 `provider_timeout`、`proxy_or_network`、`provider_rate_limit`、`provider_auth`、`abort_stuck`、`memory_dream_timeout`、`channel_session`、`gateway_degraded`、`config_reload`、`task_runtime` 等类别。\n\n默认策略是 `OPENCLAW_DIAG_ACTION=\"log\"`，因为 WARN 是证据，不一定等于“应该马上重启”。如果你显式改成 `restart` 或 `command`，也必须连续多次出现诊断异常，并且外部网络探测正常，才会执行动作。\n\n排查时可以这样判断：\n\n| 证据 | 更可能的问题范围 | 默认策略 |\n| --- | --- | --- |\n| Gateway status/health 挂了 | Gateway 进程或 RPC 链路 | 走原本的 Gateway 策略，立即重启。 |\n| 通道探测失败，但外部网络正常 | 通道或 session 链路 | 退避、复查，仍失败再重启 Gateway。 |\n| OpenClaw 日志显示 provider timeout，模型探针也失败 | 模型 provider/API 链路 | 先记录证据；可选自定义动作。单纯重启 Gateway 未必有用。 |\n| OpenClaw 日志显示 provider timeout，但模型探针成功 | OpenClaw 运行时、任务、session 或特定请求路径 | 继续保留证据，不把锅直接甩给 provider。 |\n| 日志显示 proxy/DNS/TLS 错误 | 本机代理、DNS、TLS 或运营商路由 | 记录证据，避免重启风暴，优先修代理/网络路由。 |\n| 日志显示 session expired 或 monitor stopped | 通道插件/session | 确认后重启 Gateway 往往有价值。 |\n\n### 可选模型探针\n\n如果你想判断问题到底出在 Gateway/通道，还是模型 provider 链路，可以显式开启：\n\n```bash\nMODEL_PROBE_ENABLED=\"1\"\n```\n\n开启后，看门狗会先读取 OpenClaw 当前模型 provider 的 `baseUrl`，做一次不带凭据、不消耗 token 的入口连通性探测。然后再执行端到端模型探针：\n\n```bash\nopenclaw agent --session-id \"$MODEL_PROBE_SESSION_ID\" \\\n  --thinking \"$MODEL_PROBE_THINKING\" \\\n  --timeout \"$MODEL_PROBE_TIMEOUT\" \\\n  --json \\\n  --message \"$MODEL_PROBE_MESSAGE\"\n```\n\n如果 `MODEL_PROBE_MODEL` 为空，就使用 OpenClaw 当前配置的默认模型。结果会写入主日志，以及：\n\n```text\n~/.local/state/openclaw-gateway-watchdog/model-probe-history.jsonl\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-model-probe.json\n~/.local/state/openclaw-gateway-watchdog/last-model-api-edge-probe.txt\n```\n\n排查凌晨模型 provider 超时，可以先用这组设置：\n\n```bash\nMODEL_PROBE_ENABLED=\"1\"\nMODEL_EDGE_PROBE_ENABLED=\"1\"\nMODEL_PROBE_INTERVAL=\"600\"\nMODEL_PROBE_TIMEOUT=\"120\"\nMODEL_PROBE_ACTION=\"log\"\nMODEL_PROBE_THINKING=\"off\"\nMODEL_PROBE_MESSAGE=\"Reply with exactly OK.\"\n```\n\n`MODEL_PROBE_ACTION` 支持：\n\n- `log`：只记录证据，默认策略。\n- `restart`：连续失败达到 `MODEL_PROBE_FAILURES_BEFORE_ACTION` 后重启 gateway，但会先确认外部网络不是全局断网。\n- `command`：连续失败后执行 `MODEL_PROBE_COMMAND`。\n\n这个功能会真实调用模型，可能消耗额度或费用。日志不会打印 API key，但会记录 provider/model 名称、耗时、退出状态和第一行错误摘要。\n`MODEL_EDGE_PROBE_ENABLED` 不使用凭据，也不调用 `/chat/completions`；它只检查 provider API 入口，比如 `https://api.deepseek.com`，是否能快速完成 DNS/TLS/HTTP 连接。\n\n## 安全边界\n\n这个项目不修改 OpenClaw 配置、不改微信插件源码、不处理消息内容。默认情况下它只做三件事：\n\n1. 探测本机 gateway 和外部 URL。\n2. 写自己的日志和状态文件。\n3. 在满足保护条件后执行配置好的 gateway 重启命令。\n\n可选模型探针只有在你显式开启后才会发起真实模型请求。\n\n分享日志前，请检查里面是否包含本机路径、服务名或私有通道地址。\n\n## 开源许可\n\nMIT-0。这个许可证符合 ClawHub skill 发布要求，也方便别人直接复用、改造和分发。\n\n## 发布到 ClawHub\n\n已发布包：\n\n- ClawHub：<https://clawhub.ai/zc-kama/gateway-resilience-guard>\n- Slug：`gateway-resilience-guard`\n\n本仓库带 `SKILL.md`，也可以重新发布为 OpenClaw skill：\n\n```bash\nclawhub publish . \\\n  --slug gateway-resilience-guard \\\n  --name \"OpenClaw Gateway Resilience Guard\" \\\n  --version 1.4.2 \\\n  --changelog \"Redesign dashboard navigation, alignment, trend chart axis, strategy controls, and visual system\"\n```\n\n发布前需要先执行 `clawhub login` 完成 CLI 登录。\n\nFile v1.4.2:openclaw-plugin/openclaw.plugin.json\n\n{\n  \"id\": \"resilience-guard\",\n  \"name\": \"Resilience Guard\",\n  \"description\": \"Adds a Gateway Control UI entry that opens the external OpenClaw watchdog dashboard.\",\n  \"version\": \"1.4.2\",\n  \"configSchema\": {\n    \"type\": \"object\",\n    \"additionalProperties\": false,\n    \"properties\": {\n      \"dashboardUrl\": {\n        \"type\": \"string\",\n        \"default\": \"http://127.0.0.1:18790/\"\n      }\n    }\n  },\n  \"uiHints\": {\n    \"dashboardUrl\": {\n      \"label\": \"Dashboard URL\",\n      \"help\": \"External watchdog dashboard URL. It remains available even when Gateway is down.\",\n      \"placeholder\": \"http://127.0.0.1:18790/\"\n    }\n  }\n}\n\nFile v1.4.2:openclaw-plugin/package.json\n\n{\n  \"name\": \"@zc-kama/openclaw-resilience-guard\",\n  \"version\": \"1.4.2\",\n  \"type\": \"module\",\n  \"description\": \"OpenClaw plugin bridge for the external Gateway Resilience Guard dashboard.\",\n  \"license\": \"MIT-0\",\n  \"openclaw\": {\n    \"extensions\": [\n      \"./index.js\"\n    ],\n    \"compat\": {\n      \"pluginApi\": \">=2026.3.24-beta.2\",\n      \"minGatewayVersion\": \"2026.3.24-beta.2\"\n    },\n    \"build\": {\n      \"openclawVersion\": \"2026.5.18\",\n      \"pluginSdkVersion\": \"2026.5.18\"\n    }\n  }\n}\n\nArchive v1.4.1: 19 files, 59097 bytes\n\nFiles: CHANGELOG.md (3450b), dashboard/server.py (17990b), dashboard/static/app.js (17752b), dashboard/static/index.html (5822b), dashboard/static/styles.css (9504b), gateway-watchdog.ps1 (29656b), gateway-watchdog.sh (31639b), install-watchdog.ps1 (5020b), install-watchdog.sh (10659b), openclaw-plugin/index.js (1493b), openclaw-plugin/openclaw.plugin.json (624b), openclaw-plugin/package.json (485b), README-watchdog.md (1588b), README.md (15516b), README.zh-CN.md (15409b), SKILL.md (4331b), uninstall-watchdog.ps1 (1496b), uninstall-watchdog.sh (2009b), _meta.json (143b)\n\nFile v1.4.1:SKILL.md\n\n---\nname: gateway-resilience-guard\ndescription: OpenClaw Gateway Resilience Guard keeps Gateway, channels, logs, and optional model-provider probes observable after Wi-Fi changes, WSL sleep/resume, session expiry, provider timeouts, or partial outages. It adds a localhost dashboard with charts, event trends, language/theme switching, strategy presets, guarded action unlock, native OpenClaw health probes, restart safeguards, systemd, LaunchAgent, Task Scheduler, and an optional OpenClaw plugin bridge.\ntags:\n  - openclaw\n  - gateway\n  - watchdog\n  - resilience\n  - wechat\n  - wsl\n  - systemd\n  - macos\n  - windows\nrequirements:\n  tools:\n    - bash\n    - curl\n    - systemctl optional\n    - launchctl optional\n    - powershell optional\n    - openclaw recommended\npermissions:\n  - Writes a user-level systemd service, macOS LaunchAgent, or Windows scheduled task when requested.\n  - Writes config and logs under user-level config/state directories.\n  - Starts a localhost dashboard on 127.0.0.1:18790 when Python is available.\n  - Restarts OpenClaw Gateway through systemctl --user or openclaw gateway restart.\n  - Optional model probe sends real OpenClaw model requests only when explicitly enabled.\n---\n\n# OpenClaw Gateway Resilience Guard\n\nUse this skill when a user wants to keep OpenClaw Gateway and message channels online after network drops, WSL sleep/resume, macOS/Windows wake events, long-lived connection failures, or recurring model-provider timeouts.\n\n## Install\n\nLinux, WSL, or macOS:\n\n```bash\nbash install-watchdog.sh\n```\n\nThe installer works with defaults. It prompts for the main channel probe URL when running interactively, but pressing Enter is enough for the default WeChat probe.\n\nFor unattended install:\n\n```bash\nbash install-watchdog.sh --yes\n```\n\nFor a custom channel:\n\n```bash\nbash install-watchdog.sh --channel-url \"https://your-channel.example.com/health\"\n```\n\nWindows PowerShell:\n\n```powershell\npowershell -ExecutionPolicy Bypass -File .\\install-watchdog.ps1\n```\n\nAfter install, open the standalone dashboard:\n\n```text\nhttp://127.0.0.1:18790/\n```\n\nThe dashboard is served by the watchdog process, not Gateway, so it remains the recovery entry when Gateway is down.\n\nOptional Gateway-side bridge while Gateway is healthy:\n\n```bash\nopenclaw plugins install ./openclaw-plugin\nopenclaw plugins enable resilience-guard\nopenclaw gateway restart\n```\n\n## Operate\n\n```bash\nsystemctl --user status gateway-watchdog\njournalctl --user -u gateway-watchdog -f\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh\n```\n\nOn macOS, inspect `launchctl print gui/$(id -u)/ai.clawhub.gateway-resilience-guard`.\nOn Windows, inspect `Get-ScheduledTask -TaskName \"OpenClaw Gateway Resilience Guard\"`.\n\nIf user systemd is unavailable, the installer starts a direct background fallback and stores its pid under `~/.local/state/openclaw-gateway-watchdog/watchdog.pid`.\n\n## Safety Model\n\nThe watchdog restarts only after layered checks:\n\n1. OpenClaw native health probes: `openclaw gateway status --require-rpc`, `openclaw health --json --verbose`, and `openclaw status --deep` when available.\n2. Runtime diagnostics: `openclaw models status` plus `openclaw logs --plain` signal scanning for provider timeout, proxy/network, rate limit, auth, channel session, gateway degraded, config reload, and task runtime warnings.\n3. Local gateway health URL/TCP and main channel URL fallbacks.\n4. General network URLs to avoid restarting during whole-machine network failure.\n5. Optional model-provider probe via `openclaw agent --json`; default action is evidence logging only.\n6. Dashboard action token, backoff, and hourly restart limits to avoid restart storms.\n\nModel probing is disabled by default because it consumes real provider quota. To diagnose provider timeouts, set `MODEL_PROBE_ENABLED=1`, keep `MODEL_PROBE_ACTION=log`, and inspect `model-probe-history.jsonl` the next day.\nDiagnostic log scanning is enabled by default, but `OPENCLAW_DIAG_ACTION=log` keeps it non-invasive unless the operator explicitly opts into a restart or custom command.\nUse the dashboard strategy buttons for common modes: observe, overnight diagnosis, channel recovery, and conservative circuit breaker.\n\nTell users to review `~/.config/openclaw-gateway-watchdog/watchdog.env` before publishing, sharing logs, or reporting issues.\n\nFile v1.4.1:README.md\n\n# OpenClaw Gateway Resilience Guard\n\nExternal recovery guard for OpenClaw Gateway and long-lived channel plugins such as `openclaw-weixin`.\n\nThis project is for people who run OpenClaw continuously on Linux, WSL, macOS, or Windows and need the gateway to recover from channel disconnects, network sleep/resume, and long-lived session failures without babysitting the terminal.\n\n## Problem\n\nOpenClaw Gateway can still be alive while an individual channel is no longer healthy. This is common after laptop sleep, Wi-Fi changes, WSL network hiccups, or long idle periods.\n\nThe WeChat plugin is especially sensitive because it depends on a long-poll `getUpdates` loop. In the upstream `Tencent/openclaw-weixin` code, the monitor has a limited retry loop and session guard:\n\n- `monitor.ts` defines `MAX_CONSECUTIVE_FAILURES = 3` and `BACKOFF_DELAY_MS = 30_000`.\n- `session-guard.ts` defines `SESSION_PAUSE_DURATION_MS = 60 * 60 * 1000` and `SESSION_EXPIRED_ERRCODE = -14`.\n- Issue [Tencent/openclaw-weixin#141](https://github.com/Tencent/openclaw-weixin/issues/141) reports that after a config hot reload the monitor can end without starting again; the workaround is `openclaw gateway restart`.\n- Issue [Tencent/openclaw-weixin#155](https://github.com/Tencent/openclaw-weixin/issues/155) reports that `errcode=-14` can enter a 60-minute session pause loop and block outbound messages.\n\nThis watchdog does not replace the official plugin. It is an external safety net: when the gateway or channel stops behaving like a live system, it restarts the gateway with guardrails.\n\n## Design\n\nThe script uses a layered health model before it restarts anything:\n\n| Layer | Probe | Purpose |\n| --- | --- | --- |\n| Gateway | `openclaw gateway status --json --require-rpc`, local health URL, local TCP port, service/process fallback | Detect whether OpenClaw Gateway is down or locally unreachable. |\n| Channel | `openclaw health --json --verbose`, `openclaw status --deep`, optional `openclaw channels status --probe`, then URL fallback | Prefer OpenClaw's own per-channel health model, then fall back to a configured URL when native probes are unavailable. |\n| Runtime diagnostics | `openclaw models status --json`, `openclaw logs --plain`, warning classification | Separate provider, proxy/network, auth/rate-limit, gateway, channel-session, config reload, and task-runtime evidence before choosing an action. |\n| Model API, optional | `openclaw agent --json` with the configured model provider | Detect whether the configured model path is timing out while Gateway and channels still look healthy. Disabled by default because it makes real model calls. |\n| Network | multiple independent URLs, default Baidu/QQ/Weixin | Avoid restarting the gateway during whole-machine or ISP network failure. |\n\nOnly gateway failures restart immediately. Channel failures go through confirmation, network split-brain protection, exponential backoff, and restart-rate limits.\nModel probe failures default to evidence logging only; users can opt in to restart or a custom command after consecutive failures.\n\n## Recovery Policy\n\n- Gateway down: restart immediately.\n- Gateway running but channel probe fails: confirm with general network probes.\n- General network also fails: do nothing except wait; restarting will not fix an offline machine.\n- General network works but channel stays down: wait with exponential backoff, re-check, then restart.\n- OpenClaw warning logs: classify and record evidence first; default action is log-only.\n- Five consecutive successful channel probes reset the failure state.\n- Restart storm protection limits gateway restarts per hour.\n- Night hours can use a slower probe interval to reduce noise.\n\n## Files\n\n| File | Purpose |\n| --- | --- |\n| `gateway-watchdog.sh` | Main daemon loop: probes, backoff, restart decisions, log rotation, single-instance lock. |\n| `gateway-watchdog.ps1` | Windows-native daemon loop for Task Scheduler. |\n| `dashboard/` | Standalone local Web UI and API served by the watchdog, independent of Gateway. |\n| `openclaw-plugin/` | Optional native OpenClaw plugin bridge that redirects `/resilience-guard` to the standalone dashboard. |\n| `install-watchdog.sh` | Linux/WSL/macOS installer: copies files, writes config, creates systemd user service or macOS LaunchAgent. |\n| `install-watchdog.ps1` | Windows installer: creates config and a Task Scheduler job. |\n| `uninstall-watchdog.sh` | Stops and removes the service and installed scripts. |\n| `uninstall-watchdog.ps1` | Windows uninstaller. |\n| `SKILL.md` | ClawHub/OpenClaw skill metadata and operator instructions. |\n| `README.zh-CN.md` | Chinese documentation. |\n\n## Install\n\nInstall from ClawHub:\n\n```bash\nclawhub install gateway-resilience-guard\n```\n\nOr use this repository directly.\n\nLinux, WSL, or macOS:\n\n```bash\nbash install-watchdog.sh\n```\n\nFor unattended install:\n\n```bash\nbash install-watchdog.sh --yes\n```\n\nFor a custom channel probe:\n\n```bash\nbash install-watchdog.sh --channel-url \"https://your-channel.example.com/health\"\n```\n\nWindows PowerShell:\n\n```powershell\npowershell -ExecutionPolicy Bypass -File .\\install-watchdog.ps1\n```\n\n## Dashboard\n\nThe installer enables a standalone local dashboard:\n\n```text\nhttp://127.0.0.1:18790/\n```\n\nThis dashboard is served by the watchdog, not by OpenClaw Gateway. If Gateway is down, the dashboard can still open and show the last known evidence.\n\nIt includes:\n\n- Gateway, channel, network, OpenClaw log, and model-provider status.\n- Category charts for provider timeout, proxy/network, rate-limit, auth, channel session, Gateway degraded, config reload, and task runtime warnings.\n- Event trend chart with separate lanes for API failures, log warnings, successful model probes, and healthy diagnostics.\n- Chinese/English language switching and Light, Dark, Ocean, and Forest themes.\n- Status-file freshness checks so stale data is obvious.\n- Quick strategy buttons: observe, overnight diagnosis, channel recovery, and conservative circuit breaker.\n- Guarded actions with an unlock flow: run diagnostics, restart Gateway, apply presets, and export a diagnostic JSON bundle.\n\nDashboard actions are bound to localhost and protected with the generated `DASHBOARD_TOKEN`. The token is injected only into the same-origin dashboard page. A random token is written during install; the value is not shown in logs.\n\nOptional OpenClaw plugin bridge:\n\n```bash\nopenclaw plugins install ./openclaw-plugin\nopenclaw plugins enable resilience-guard\nopenclaw gateway restart\n```\n\nThen open the Gateway route while Gateway is healthy:\n\n```text\nhttp://127.0.0.1:18789/resilience-guard\n```\n\nThat route redirects to the external dashboard. It is a convenience entry only; the external dashboard remains the recovery entry when Gateway is unavailable.\n\nThe installer writes:\n\n- scripts to `~/.local/share/openclaw-gateway-watchdog`;\n- config to `~/.config/openclaw-gateway-watchdog/watchdog.env`;\n- logs to `~/.local/state/openclaw-gateway-watchdog/watchdog.log`;\n- a Linux/WSL user service to `~/.config/systemd/user/gateway-watchdog.service`;\n- a macOS LaunchAgent to `~/Library/LaunchAgents/ai.clawhub.gateway-resilience-guard.plist`;\n- a Windows scheduled task named `OpenClaw Gateway Resilience Guard`.\n\nIf user systemd is unavailable, the installer starts a direct background fallback process and stores its pid in the state directory.\n\n## Manage\n\n```bash\nsystemctl --user status gateway-watchdog\njournalctl --user -u gateway-watchdog -f\nsystemctl --user restart gateway-watchdog\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh\n```\n\nmacOS:\n\n```bash\nlaunchctl print gui/$(id -u)/ai.clawhub.gateway-resilience-guard\ntail -f ~/.local/state/openclaw-gateway-watchdog/watchdog.log\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh\n```\n\nWindows:\n\n```powershell\nGet-ScheduledTask -TaskName \"OpenClaw Gateway Resilience Guard\"\nGet-Content \"$env:LOCALAPPDATA\\openclaw-gateway-watchdog\\watchdog.log\" -Wait\npowershell -ExecutionPolicy Bypass -File \"$env:LOCALAPPDATA\\openclaw-gateway-watchdog\\uninstall-watchdog.ps1\"\n```\n\nRemove config and logs too:\n\n```bash\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh --purge\n```\n\n## Configuration\n\nMost users can keep the generated defaults. Advanced settings live in:\n\n```text\n~/.config/openclaw-gateway-watchdog/watchdog.env\n```\n\nCommon keys:\n\n```bash\nGATEWAY_SERVICE=\"openclaw-gateway\"\nGATEWAY_HEALTH_URL=\"http://127.0.0.1:18789/healthz\"\nGATEWAY_HOST=\"127.0.0.1\"\nGATEWAY_PORT=\"18789\"\nCHANNEL_URL=\"https://ilinkai.weixin.qq.com\"\nNETWORK_URLS=\"https://www.baidu.com https://www.qq.com https://api.weixin.qq.com\"\nRESTART_COMMAND=\"systemctl --user restart openclaw-gateway\"\nOPENCLAW_NATIVE_PROBES=\"auto\"\nOPENCLAW_HEALTH_TIMEOUT_MS=\"12000\"\nOPENCLAW_GATEWAY_STRICT=\"0\"\nOPENCLAW_CHANNELS_PROBE=\"1\"\nOPENCLAW_DIAG_ENABLED=\"1\"\nOPENCLAW_DIAG_INTERVAL=\"300\"\nOPENCLAW_LOG_SCAN_ENABLED=\"1\"\nOPENCLAW_LOG_LIMIT=\"200\"\nOPENCLAW_LOG_SIGNAL_LIMIT=\"40\"\nOPENCLAW_LOG_TIMEOUT_MS=\"15000\"\nOPENCLAW_LOG_WARN_PATTERNS=\"fetch failed|fetch timeout|LLM idle timeout|model silent|...\"\nOPENCLAW_DIAG_ACTION=\"log\"\nOPENCLAW_DIAG_FAILURES_BEFORE_ACTION=\"2\"\nOPENCLAW_DIAG_COMMAND=\"\"\nDASHBOARD_ENABLED=\"1\"\nDASHBOARD_HOST=\"127.0.0.1\"\nDASHBOARD_PORT=\"18790\"\nDASHBOARD_ACTIONS_ENABLED=\"1\"\nDASHBOARD_TOKEN=\"generated-at-install\"\nDASHBOARD_DIR=\"~/.local/share/openclaw-gateway-watchdog/dashboard\"\nMODEL_PROBE_ENABLED=\"0\"\nMODEL_EDGE_PROBE_ENABLED=\"1\"\nMODEL_PROBE_INTERVAL=\"1800\"\nMODEL_PROBE_TIMEOUT=\"120\"\nMODEL_PROBE_FAILURES_BEFORE_ACTION=\"2\"\nMODEL_PROBE_ACTION=\"log\"\nMODEL_PROBE_COMMAND=\"\"\nMODEL_PROBE_MODEL=\"\"\nMODEL_PROBE_THINKING=\"off\"\nMODEL_PROBE_SESSION_ID=\"watchdog-model-probe\"\nMODEL_PROBE_MESSAGE=\"Reply with exactly OK.\"\nBASE_INTERVAL=\"60\"\nNIGHT_INTERVAL=\"300\"\nMAX_INTERVAL=\"1800\"\nCHANNEL_FAILURES_BEFORE_RESTART=\"2\"\nSUCCESS_COUNT_TO_RESET=\"5\"\nMAX_RESTARTS_PER_HOUR=\"6\"\n```\n\nUse `RESTART_COMMAND` if your OpenClaw install is not managed by a user-level systemd unit. Example:\n\n```bash\nRESTART_COMMAND=\"openclaw gateway restart\"\n```\n\nOn Windows, the generated config is JSON:\n\n```text\n%APPDATA%\\openclaw-gateway-watchdog\\watchdog.json\n```\n\nSet `OpenClawNativeProbes` to `false` if your OpenClaw CLI is too old for `openclaw health` or `openclaw status --deep`.\n\n### OpenClaw diagnostics and log signals\n\nThe watchdog does more than ping one URL. Every `OPENCLAW_DIAG_INTERVAL` seconds it collects an OpenClaw diagnostic snapshot:\n\n```text\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-gateway-status.json\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-health.json\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-model-status.json\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-status-deep.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-logs.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-log-signals.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-log-signal-categories.txt\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-diagnostics.jsonl\n```\n\n`last-openclaw-log-signals.txt` is filtered from `openclaw logs --plain`. It classifies common failure families such as `provider_timeout`, `proxy_or_network`, `provider_rate_limit`, `provider_auth`, `abort_stuck`, `memory_dream_timeout`, `channel_session`, `gateway_degraded`, `config_reload`, and `task_runtime`.\n\nDefault action is `OPENCLAW_DIAG_ACTION=\"log\"` because warnings are evidence, not always proof that a restart is correct. If you explicitly set `OPENCLAW_DIAG_ACTION=\"restart\"` or `command`, the action only runs after consecutive diagnostic warnings and only when the general network probes still pass.\n\nPractical interpretation:\n\n| Evidence | Likely scope | Default strategy |\n| --- | --- | --- |\n| Gateway status/health is down | Gateway process or RPC path | Restart Gateway immediately through the normal gateway policy. |\n| Channel probe fails, network probes pass | Channel/session path | Backoff, re-check, then restart Gateway if still failed. |\n| OpenClaw logs show provider timeout, model probe also fails | Provider/API path | Log evidence; optional custom action. A Gateway restart may not fix provider outage. |\n| OpenClaw logs show provider timeout, model probe succeeds | OpenClaw runtime, task, session, or specific request path | Keep evidence, inspect logs; avoid blaming the provider alone. |\n| Logs show proxy/DNS/TLS errors | Local proxy, DNS, TLS, or ISP route | Log evidence and avoid restart storms; fix network/proxy route first. |\n| Logs show session expiry or monitor stopped | Channel plugin/session | Gateway restart is often useful after confirmation. |\n\n### Optional model probe\n\nSet `MODEL_PROBE_ENABLED=\"1\"` when you need to prove whether failures are in the model-provider path instead of Gateway or channel health.\n\nWhen enabled, the watchdog first reads OpenClaw's configured model provider and probes its provider `baseUrl` without credentials. Then it runs the end-to-end model probe:\n\n```bash\nopenclaw agent --session-id \"$MODEL_PROBE_SESSION_ID\" \\\n  --thinking \"$MODEL_PROBE_THINKING\" \\\n  --timeout \"$MODEL_PROBE_TIMEOUT\" \\\n  --json \\\n  --message \"$MODEL_PROBE_MESSAGE\"\n```\n\nIf `MODEL_PROBE_MODEL` is empty, OpenClaw's configured default model is used. Results are written to the main log and to:\n\n```text\n~/.local/state/openclaw-gateway-watchdog/model-probe-history.jsonl\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-model-probe.json\n~/.local/state/openclaw-gateway-watchdog/last-model-api-edge-probe.txt\n```\n\nRecommended diagnostic settings for overnight provider issues:\n\n```bash\nMODEL_PROBE_ENABLED=\"1\"\nMODEL_EDGE_PROBE_ENABLED=\"1\"\nMODEL_PROBE_INTERVAL=\"600\"\nMODEL_PROBE_TIMEOUT=\"120\"\nMODEL_PROBE_ACTION=\"log\"\nMODEL_PROBE_THINKING=\"off\"\nMODEL_PROBE_MESSAGE=\"Reply with exactly OK.\"\n```\n\n`MODEL_PROBE_ACTION` can be:\n\n- `log`: record evidence only. This is the default.\n- `restart`: restart Gateway after `MODEL_PROBE_FAILURES_BEFORE_ACTION` consecutive model failures, but only if general network probes still pass.\n- `command`: run `MODEL_PROBE_COMMAND` after consecutive failures.\n\nThis feature sends real model requests and may consume provider quota or money. It does not print API keys, but it does store provider/model names, timing, exit status, and the first non-empty error line.\n`MODEL_EDGE_PROBE_ENABLED` does not use credentials and does not call `/chat/completions`; it only checks whether the provider API edge such as `https://api.deepseek.com` is reachable quickly.\n\n## Safety Notes\n\nThis project intentionally avoids destructive behavior. It does not edit OpenClaw configuration, tokens, sessions, or plugin files. By default it only probes URLs and restarts the gateway through the configured command. The optional model probe makes real model calls only after you explicitly enable it.\n\nBefore sharing logs, review them for local paths, service names, and channel URLs.\n\n## License\n\nMIT-0. This matches ClawHub's skill publishing requirement and allows reuse without attribution requirements.\n\n## Publish To ClawHub\n\nPublished package:\n\n- ClawHub: <https://clawhub.ai/zc-kama/gateway-resilience-guard>\n- Slug: `gateway-resilience-guard`\n\nThis repository includes `SKILL.md`, so it can also be republished as an OpenClaw skill bundle:\n\n```bash\nclawhub publish . \\\n  --slug gateway-resilience-guard \\\n  --name \"OpenClaw Gateway Resilience Guard\" \\\n  --version 1.4.1 \\\n  --changelog \"Polish dashboard layout, language, themes, charts, logs, and guarded action unlock\"\n```\n\nClawHub requires CLI authentication. Run `clawhub login` first.\n\nFile v1.4.1:_meta.json\n\n{\n  \"ownerId\": \"kn70e6hn9pv8yrpzjn5kqczygd8708vw\",\n  \"slug\": \"gateway-resilience-guard\",\n  \"version\": \"1.4.1\",\n  \"publishedAt\": 1779378675499\n}\n\nFile v1.4.1:CHANGELOG.md\n\n# Changelog\n\n## 1.4.1\n\n- Redesign the dashboard with a sidebar layout, stronger visual grouping, and selectable Light, Dark, Ocean, and Forest themes.\n- Add Chinese/English language switching for dashboard labels, buttons, hints, legends, and notifications.\n- Fix canvas redraw sizing so repeated refreshes do not stretch dashboard panels.\n- Replace \"overnight timeline\" with a general event trend chart that includes lanes, legends, and time-axis ticks.\n- Auto-follow the newest watchdog log lines while keeping a toggle for manual scrolling.\n- Add an unlock/lock action flow and strategy hover hints so guarded controls are discoverable.\n\n## 1.4.0\n\n- Add a standalone local dashboard at `http://127.0.0.1:18790` that remains available when OpenClaw Gateway is down.\n- Add dashboard API summaries, log-signal charts, overnight timelines, status file freshness, safe config summaries, diagnostic export, and quick strategy presets.\n- Add guarded dashboard actions for run diagnostics, restart Gateway, and apply presets using a generated local action token.\n- Add an OpenClaw native plugin bridge that registers `/resilience-guard` and redirects to the external dashboard when Gateway is healthy.\n- Install dashboard files on Linux/WSL, macOS, and Windows.\n\n## 1.3.1\n\n- Publish a ClawHub package that includes the Windows PowerShell installer, watchdog, and uninstaller files.\n\n## 1.3.0\n\n- Add an opt-in model-provider probe that uses `openclaw agent --json` against the configured OpenClaw model path.\n- Add a no-credential provider edge probe that checks the configured provider `baseUrl` before the end-to-end model call.\n- Add OpenClaw runtime diagnostics that snapshot gateway status, health, model status, deep status, and recent OpenClaw logs.\n- Classify log signals for provider timeout, proxy/network, rate limit, auth, channel session, gateway degraded, config reload, task runtime, and related warning families.\n- Record model probe evidence in the main log and `model-probe-history.jsonl` without logging API keys.\n- Add configurable model failure actions: log-only, gateway restart, or a custom command after consecutive failures.\n- Document safe overnight diagnostics for separating Gateway/channel failures from model-provider timeouts.\n\n## 1.2.0\n\n- Add OpenClaw-native health probing via `openclaw gateway status --json --require-rpc`, `openclaw health --json --verbose`, and `openclaw status --deep`.\n- Add Windows Task Scheduler support with native PowerShell install, watchdog, and uninstall scripts.\n- Add macOS LaunchAgent support to the Bash installer and uninstaller.\n- Document cross-platform installation and native probe configuration.\n\n## 1.1.0\n\n- Rename the ClawHub package to `gateway-resilience-guard`.\n- Expand the public summary to describe the layered probes, restart guardrails, and advantages over simple restart loops or one-URL monitors.\n\n## 1.0.1\n\n- Add the published ClawHub URL and install command to the README files.\n\n## 1.0.0\n\n- First public zero-config release.\n- Install from the current folder instead of a hard-coded workspace path.\n- Generate user config and systemd service automatically.\n- Add layered gateway/channel/network probes.\n- Add exponential backoff, hourly restart limits, lock protection, and log rotation.\n- Reset failure state after five consecutive successful channel probes.\n- Add ClawHub-ready `SKILL.md`.\n- Initial ClawHub publication used a temporary slug before the 1.1.0 rename.\n\nFile v1.4.1:README-watchdog.md\n\n# OpenClaw Gateway 看门狗\n\n这是项目的中文快捷入口。完整中文文档见 [README.zh-CN.md](README.zh-CN.md)，英文文档见 [README.md](README.md)。\n\n最简单安装：\n\n```bash\nbash install-watchdog.sh\n```\n\nWindows：\n\n```powershell\npowershell -ExecutionPolicy Bypass -File .\\install-watchd\n\nArchive v1.4.0: 19 files, 53948 bytes\n\nFiles: CHANGELOG.md (2822b), dashboard/server.py (17990b), dashboard/static/app.js (9022b), dashboard/static/index.html (3624b), dashboard/static/styles.css (5868b), gateway-watchdog.ps1 (29656b), gateway-watchdog.sh (31639b), install-watchdog.ps1 (5020b), install-watchdog.sh (10659b), openclaw-plugin/index.js (1493b), openclaw-plugin/openclaw.plugin.json (624b), openclaw-plugin/package.json (485b), README-watchdog.md (1479b), README.md (15331b), README.zh-CN.md (15220b), SKILL.md (4343b), uninstall-watchdog.ps1 (1496b), uninstall-watchdog.sh (2009b), _meta.json (143b)\n\nArchive v1.3.1: 12 files, 36186 bytes\n\nFiles: CHANGELOG.md (2205b), gateway-watchdog.ps1 (26436b), gateway-watchdog.sh (29235b), install-watchdog.ps1 (4492b), install-watchdog.sh (9987b), README-watchdog.md (1232b), README.md (13418b), README.zh-CN.md (13446b), SKILL.md (3702b), uninstall-watchdog.ps1 (1120b), uninstall-watchdog.sh (1759b), _meta.json (143b)\n\nArchive v1.3.0: 9 files, 26751 bytes\n\nFiles: CHANGELOG.md (2085b), gateway-watchdog.sh (29235b), install-watchdog.sh (9987b), README-watchdog.md (1232b), README.md (13393b), README.zh-CN.md (13421b), SKILL.md (3702b), uninstall-watchdog.sh (1759b), _meta.json (143b)\n\nArchive v1.2.0: 13 files, 24248 bytes\n\nFiles: CHANGELOG.md (1210b), gateway-watchdog.ps1 (10629b), gateway-watchdog.sh (14598b), install-watchdog.ps1 (3254b), install-watchdog.sh (8816b), LICENSE (897b), README-watchdog.md (476b), README.md (7708b), README.zh-CN.md (7841b), SKILL.md (2905b), uninstall-watchdog.ps1 (1120b), uninstall-watchdog.sh (1759b), _meta.json (143b)\n\nArchive v1.1.0: 10 files, 15283 bytes\n\nFiles: CHANGELOG.md (800b), gateway-watchdog.sh (8506b), install-watchdog.sh (6374b), LICENSE (897b), README-watchdog.md (381b), README.md (6182b), README.zh-CN.md (6330b), SKILL.md (2365b), uninstall-watchdog.sh (1430b), _meta.json (143b)","readmeExcerpt":"Skill: OpenClaw Gateway Resilience Guard Owner: zc-kama Summary: OpenClaw Gateway Resilience Guard keeps Gateway, channels, logs, and optional model-provider probes observable after Wi-Fi changes, WSL sleep/resume, session... Tags: bash:1.2.0, channels:1.2.0, dashboard:1.4.4, gateway:1.4.4, guardian:1.2.0, latest:1.4.4, macos:1.4.4, openclaw:1.4.4, resilience:1.4.4, systemd:1.2.0, watchdog:1.4.4, wechat:1.4.4, weixin","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"bash install-watchdog.sh"},{"language":"bash","snippet":"bash install-watchdog.sh --yes"},{"language":"bash","snippet":"bash install-watchdog.sh --channel-url \"https://your-channel.example.com/health\""},{"language":"powershell","snippet":"powershell -ExecutionPolicy Bypass -File .\\install-watchdog.ps1"},{"language":"text","snippet":"http://127.0.0.1:18790/"},{"language":"bash","snippet":"openclaw plugins install ./openclaw-plugin\nopenclaw plugins enable resilience-guard\nopenclaw gateway restart"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: gateway-resilience-guard\ndescription: OpenClaw Gateway Resilience Guard keeps Gateway, channels, logs, and optional model-provider probes observable after Wi-Fi changes, WSL sleep/resume, session expiry, provider timeouts, or partial outages. It adds a localhost dashboard with charts, event trends, language/theme switching, strategy presets, guarded action unlock, native OpenClaw health probes, restart safeguards, systemd, LaunchAgent, Task Scheduler, and an optional OpenClaw plugin bridge.\ntags:\n  - openclaw\n  - gateway\n  - watchdog\n  - resilience\n  - wechat\n  - wsl\n  - systemd\n  - macos\n  - windows\nrequirements:\n  tools:\n    - bash\n    - curl\n    - systemctl optional\n    - launchctl optional\n    - powershell optional\n    - openclaw recommended\npermissions:\n  - Writes a user-level systemd service, macOS LaunchAgent, or Windows scheduled task when requested.\n  - Writes config and logs under user-level config/state directories.\n  - Starts a localhost dashboard on 127.0.0.1:18790 when Python is available.\n  - Restarts OpenClaw Gateway through systemctl --user or openclaw gateway restart.\n  - Optional model probe sends real OpenClaw model requests only when explicitly enabled.\n---\n\n# OpenClaw Gateway Resilience Guard\n\nUse this skill when a user wants to keep OpenClaw Gateway and message channels online after network drops, WSL sleep/resume, macOS/Windows wake events, long-lived connection failures, or recurring model-provider timeouts.\n\n## Install\n\nLinux, WSL, or macOS:\n\n```bash\nbash install-watchdog.sh\n```\n\nThe installer works with defaults. It prompts for the main channel probe URL when running interactively, but pressing Enter is enough for the default WeChat probe.\n\nFor unattended install:\n\n```bash\nbash install-watchdog.sh --yes\n```\n\nFor a custom channel:\n\n```bash\nbash install-watchdog.sh --channel-url \"https://your-channel.example.com/health\"\n```\n\nWindows PowerShell:\n\n```powershell\npowershell -ExecutionPolicy Bypass -File .\\install-watchdog.ps1\n```\n\nAfter install, open the standalone dashboard:\n\n```text\nhttp://127.0.0.1:18790/\n```\n\nThe dashboard is served by the watchdog process, not Gateway, so it remains the recovery entry when Gateway is down.\n\nOptional Gateway-side bridge while Gateway is healthy:\n\n```bash\nopenclaw plugins install ./openclaw-plugin\nopenclaw plugins enable resilience-guard\nopenclaw gateway restart\n```\n\n## Operate\n\n```bash\nsystemctl --user status gateway-watchdog\njournalctl --user -u gateway-watchdog -f\nbash ~/.local/share/openclaw-gateway-watchdog/uninstall-watchdog.sh\n```\n\nOn macOS, inspect `launchctl print gui/$(id -u)/ai.clawhub.gateway-resilience-guard`.\nOn Windows, inspect `Get-ScheduledTask -TaskName \"OpenClaw Gateway Resilience Guard\"`.\n\nIf user systemd is unavailable, the installer starts a direct background fallback and stores its pid under `~/.local/state/openclaw-gateway-watchdog/watchdog.pid`.\n\n## Safety Model\n\nThe watchdog restarts only after layered checks:\n\n1. OpenClaw native health probes: `openclaw"},{"path":"README.md","content":"# OpenClaw Gateway Resilience Guard\n\nExternal recovery guard for OpenClaw Gateway and long-lived channel plugins such as `openclaw-weixin`.\n\nThis project is for people who run OpenClaw continuously on Linux, WSL, macOS, or Windows and need the gateway to recover from channel disconnects, network sleep/resume, and long-lived session failures without babysitting the terminal.\n\n## Problem\n\nOpenClaw Gateway can still be alive while an individual channel is no longer healthy. This is common after laptop sleep, Wi-Fi changes, WSL network hiccups, or long idle periods.\n\nThe WeChat plugin is especially sensitive because it depends on a long-poll `getUpdates` loop. In the upstream `Tencent/openclaw-weixin` code, the monitor has a limited retry loop and session guard:\n\n- `monitor.ts` defines `MAX_CONSECUTIVE_FAILURES = 3` and `BACKOFF_DELAY_MS = 30_000`.\n- `session-guard.ts` defines `SESSION_PAUSE_DURATION_MS = 60 * 60 * 1000` and `SESSION_EXPIRED_ERRCODE = -14`.\n- Issue [Tencent/openclaw-weixin#141](https://github.com/Tencent/openclaw-weixin/issues/141) reports that after a config hot reload the monitor can end without starting again; the workaround is `openclaw gateway restart`.\n- Issue [Tencent/openclaw-weixin#155](https://github.com/Tencent/openclaw-weixin/issues/155) reports that `errcode=-14` can enter a 60-minute session pause loop and block outbound messages.\n\nThis watchdog does not replace the official plugin. It is an external safety net: when the gateway or channel stops behaving like a live system, it restarts the gateway with guardrails.\n\n## Design\n\nThe script uses a layered health model before it restarts anything:\n\n| Layer | Probe | Purpose |\n| --- | --- | --- |\n| Gateway | `openclaw gateway status --json --require-rpc`, local health URL, local TCP port, service/process fallback | Detect whether OpenClaw Gateway is down or locally unreachable. |\n| Channel | `openclaw health --json --verbose`, `openclaw status --deep`, optional `openclaw channels status --probe`, then URL fallback | Prefer OpenClaw's own per-channel health model, then fall back to a configured URL when native probes are unavailable. |\n| Runtime diagnostics | `openclaw models status --json`, `openclaw logs --plain`, warning classification | Separate provider, proxy/network, auth/rate-limit, gateway, channel-session, config reload, and task-runtime evidence before choosing an action. |\n| Model API, optional | `openclaw agent --json` with the configured model provider | Detect whether the configured model path is timing out while Gateway and channels still look healthy. Disabled by default because it makes real model calls. |\n| Network | multiple independent URLs, default Baidu/QQ/Weixin | Avoid restarting the gateway during whole-machine or ISP network failure. |\n\nOnly gateway failures restart immediately. Channel failures go through confirmation, network split-brain protection, exponential backoff, and restart-rate limits.\nModel probe failures default to evidence logging only;"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn70e6hn9pv8yrpzjn5kqczygd8708vw\",\n  \"slug\": \"gateway-resilience-guard\",\n  \"version\": \"1.4.4\",\n  \"publishedAt\": 1779383203702\n}"},{"path":"CHANGELOG.md","content":"# Changelog\n\n## 1.4.4\n\n- Replace fragile Web Component controls with semantic native buttons and selects styled by the local dashboard design system.\n- Fix the top action bar, sidebar navigation, refresh/export controls, and strategy buttons so they always render as clickable UI even if component scripts are blocked or stale.\n- Center the brand mark text and restore pointer cursor, hover, focus, and active states across navigation and action controls.\n- Fix config summary row layout so long environment keys no longer overlap values.\n\n## 1.4.3\n\n- Rebuild the dashboard visual system on local Fluent UI Web Components instead of plain native controls.\n- Replace the old dropdowns and buttons with componentized controls, dark glass panels, animated ambient background, and refined hover states.\n- Rebalance every dashboard page so Overview, Trends, Strategy, Logs, and Config each have complete work areas rather than sparse split-up cards.\n- Add richer summaries for trend digest, latest signals, strategy matrix, action center, and runtime map.\n- Serve bundled local component assets under `/vendor/` so the recovery UI does not depend on CDN availability.\n\n## 1.4.2\n\n- Rework the dashboard sidebar into real view switching for Overview, Trends, Strategy, Logs, and Config instead of one long page.\n- Tighten the dashboard grid, panel sizing, button alignment, and sidebar polish for a cleaner operations-console layout.\n- Fix the trend chart axis label collision by removing the redundant x-axis title and reserving more space for tick labels.\n- Keep charts stable across refreshes by drawing only the active view and ignoring hidden canvases.\n- Move strategy explanations into centered button content and remove the sidebar localhost note.\n\n## 1.4.1\n\n- Redesign the dashboard with a sidebar layout, stronger visual grouping, and selectable Light, Dark, Ocean, and Forest themes.\n- Add Chinese/English language switching for dashboard labels, buttons, hints, legends, and notifications.\n- Fix canvas redraw sizing so repeated refreshes do not stretch dashboard panels.\n- Replace \"overnight timeline\" with a general event trend chart that includes lanes, legends, and time-axis ticks.\n- Auto-follow the newest watchdog log lines while keeping a toggle for manual scrolling.\n- Add an unlock/lock action flow and strategy hover hints so guarded controls are discoverable.\n\n## 1.4.0\n\n- Add a standalone local dashboard at `http://127.0.0.1:18790` that remains available when OpenClaw Gateway is down.\n- Add dashboard API summaries, log-signal charts, overnight timelines, status file freshness, safe config summaries, diagnostic export, and quick strategy presets.\n- Add guarded dashboard actions for run diagnostics, restart Gateway, and apply presets using a generated local action token.\n- Add an OpenClaw native plugin bridge that registers `/resilience-guard` and redirects to the external dashboard when Gateway is healthy.\n- Install dashboard files on Linux/WSL, macOS, and Windows.\n\n## 1.3"},{"path":"README-watchdog.md","content":"# OpenClaw Gateway 看门狗\n\n这是项目的中文快捷入口。完整中文文档见 [README.zh-CN.md](README.zh-CN.md)，英文文档见 [README.md](README.md)。\n\n最简单安装：\n\n```bash\nbash install-watchdog.sh\n```\n\nWindows：\n\n```powershell\npowershell -ExecutionPolicy Bypass -File .\\install-watchdog.ps1\n```\n\n默认配置即可启动；需要指定自己的通道地址时：\n\n```bash\nbash install-watchdog.sh --channel-url \"https://你的通道地址/health\"\n```\n\n安装后还有一个独立图形化控制台：\n\n```text\nhttp://127.0.0.1:18790/\n```\n\n它由 watchdog 自己提供，不依赖 OpenClaw Gateway。Gateway 挂掉时，这个页面仍然可以看状态、图表、策略、日志和诊断导出。Dashboard 支持中英文切换、主题切换、事件趋势图、异常分类图和受保护操作解锁。\n\n安装后它会默认采集 OpenClaw 诊断快照和日志信号，包括 gateway status、health、models status、status --deep 和 `openclaw logs --plain` 的 WARN/ERROR 摘要。关键证据在：\n\n```text\n~/.local/state/openclaw-gateway-watchdog/watchdog.log\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-diagnostics.jsonl\n~/.local/state/openclaw-gateway-watchdog/last-openclaw-log-signals.txt\n```\n\n可选模型探针默认关闭。需要排查模型 API 是否在某个时段超时时，可以在配置里显式开启 `MODEL_PROBE_ENABLED=\"1\"`；它会发送极小的 `openclaw agent --json` 请求，并把结果写入 watchdog 日志和 `model-probe-history.jsonl`。默认动作仍是只记录证据，不会擅自改 OpenClaw 配置。"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"OpenClaw Gateway Resilience Guard keeps Gateway, channels, logs, and optional model-provider probes observable after Wi-Fi changes, WSL sleep/resume, session... Skill: OpenClaw Gateway Resilience Guard Owner: zc-kama Summary: OpenClaw Gateway Resilience Guard keeps Gateway, channels, logs, and optional model-provider probes observable after Wi-Fi changes, WSL sleep/resume, session... Tags: bash:1.2.0, channels:1.2.0, dashboard:1.4.4, gateway:1.4.4, guardian:1.2.0, latest:1.4.4, macos:1.4.4, openclaw:1.4.4, resilience:1.4.4, systemd:1.2.0, watchdog:1.4.4, wechat:1.4.4, weixin","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1697,"uniquenessScore":47,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T03:00:00.984Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T03:00:00.984Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T05:43:22.635Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}