{"id":"e99cbc90-634b-4772-a45c-6ff2f12b7be8","entityType":"agent","slug":"clawhub-zw008-ceph-aiops","name":"ceph-aiops","canonicalUrl":"https://www.xpersona.co/agent/clawhub-zw008-ceph-aiops","canonicalPath":"/agent/clawhub-zw008-ceph-aiops","generatedAt":"2026-10-10T07:41:33.610Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T21:32:55.609Z","emptyReason":null},"description":"Use this skill whenever the user needs to operate or diagnose a Ceph cluster via its ceph-mgr Dashboard REST API — decode a HEALTH_WARN/ERR state into cause + action (cluster_health), read the cluster status, inspect OSDs (tree/df/perf), placement groups (summary/stuck/scrub), pools (list/usable capacity), RBD images and snapshots, CephFS/MDS and RGW status, monitors/managers, slow ops and capacity forecast — plus governed writes (set cluster flags, reweight/mark-in/mark-out/purge OSDs, trigger scrubs, set pool quota/pg_num/autoscale/size, create/delete pools, create/delete RBD images and snapshots, throttle recovery/backfill). Always use this skill for \"ceph health\", \"what does this HEALTH_WARN mean\", \"PG_DEGRADED / OSD_NEARFULL / SLOW_OPS / MON_DOWN\", \"ceph -s\", \"which OSD is most full\", \"drain an OSD\", \"purge an OSD\", \"stuck PGs\", \"overdue scrub\", \"pool usable capacity\", \"set pool size / quota\", \"rebalance is too slow / throttle backfill\", \"RBD image or snapshot\", \"MDS behind on trimming\", \"RGW large omap\", \"mon quorum\", or \"days to nearfull\" when the context is a Ceph cluster (cephadm, hypervisor-bundled Ceph, or MicroCeph). Do NOT use when the target is not Ceph — a hypervisor, a different storage appliance, a backup product, a Kubernetes cluster, or a network device. Route those to the appropriate other AIops-tools skill (negative routing hint only). Common Ceph ops with a built-in governance harness (audit, policy, token budget, undo, risk-tiers).","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 2K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s171xgnmqse0nqvgqvqnaq5f9183kyre:ceph-aiops","sourceUrl":"https://clawhub.ai/zw008/ceph-aiops","homepage":"https://clawhub.ai/zw008/skills/ceph-aiops","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/zw008/ceph-aiops","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/zw008/skills/ceph-aiops","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":66,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"ceph-aiops technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T21:32:55.609Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T21:32:55.609Z","emptyReason":null},"stars":null,"forks":null,"downloads":1976,"packageName":null,"latestVersion":"0.11.5","tractionLabel":"2K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T21:32:55.609Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T21:32:55.609Z","lastCrawledAt":"2026-10-09T21:32:55.609Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T21:32:55.609Z","lastVerifiedAt":null,"highlights":[{"version":"0.11.5","createdAt":"2026-09-16T23:25:19.214Z","changelog":"- Removed the obsolete skill-card.md file. - Updated references/agent-guardrails.md documentation. - No changes to functionality or user-facing features.","fileCount":7,"zipByteSize":18004},{"version":"0.11.4","createdAt":"2026-09-15T05:50:07.558Z","changelog":"- Removed the sample file skill-card.md. - No functional or user-facing changes; housekeeping only.","fileCount":7,"zipByteSize":17583},{"version":"0.11.3","createdAt":"2026-09-12T23:38:35.627Z","changelog":"- Removed the redundant skill-card.md file. - Updated references/setup-guide.md (details not specified). - No changes to functionality or available tools. - General housekeeping and documentation cleanup.","fileCount":7,"zipByteSize":17829},{"version":"0.11.2","createdAt":"2026-09-12T14:05:01.564Z","changelog":"ceph-aiops 0.11.2 - Updated SKILL.md with minor OpenClaw plugin installation changes (now references clawhub:@zw008/ceph-aiops). - Removed skill-card.md from the project.","fileCount":7,"zipByteSize":17296},{"version":"0.11.1","createdAt":"2026-09-12T09:57:36.189Z","changelog":"- Added OpenClaw plugin installation instructions and usage notes to documentation. - Clarified that the skill requires the uvx binary on PATH for MCP server installation. - Removed the sample file \"skill-card.md\" from the project. - No functional or API changes; update is documentation-only.","fileCount":7,"zipByteSize":17386},{"version":"0.11.0","createdAt":"2026-09-12T00:46:19.902Z","changelog":"- Updated skill metadata: dropped mention of specific files and improved compatibility/env/bin requirements. - Removed obsolete file: skill-card.md. - Updated metadata: now accepts either ceph-aiops or uvx as required binaries, and clarified optional environment variables. - Improved negative routing and usage guidance in documentation. - No functional changes to tools or workflow; all 37 MCP tools remain available.","fileCount":7,"zipByteSize":16957},{"version":"0.10.0","createdAt":"2026-08-10T06:49:56.092Z","changelog":"- Removed the file: skill-card.md - No changes to functionality or features.","fileCount":7,"zipByteSize":17088},{"version":"0.9.0","createdAt":"2026-08-03T05:51:44.739Z","changelog":"- Removed the file skill-card.md. - No user-visible changes to skill functionality or documentation. - Version update to 0.9.0.","fileCount":7,"zipByteSize":17065}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s171xgnmqse0nqvgqvqnaq5f9183kyre:ceph-aiops","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s171xgnmqse0nqvgqvqnaq5f9183kyre:ceph-aiops` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/zw008/ceph-aiops before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zw008-ceph-aiops/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zw008-ceph-aiops/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zw008-ceph-aiops/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-zw008-ceph-aiops/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-zw008-ceph-aiops/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-zw008-ceph-aiops/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T07:41:33.605Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zw008-ceph-aiops/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zw008-ceph-aiops/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zw008-ceph-aiops/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zw008-ceph-aiops/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T21:32:55.609Z","emptyReason":null},"readme":"Skill: ceph-aiops\n\nOwner: zw008\n\nSummary: Use this skill whenever the user needs to operate or diagnose a Ceph cluster via its ceph-mgr Dashboard REST API — decode a HEALTH_WARN/ERR state into cause + action (cluster_health), read the cluster status, inspect OSDs (tree/df/perf), placement groups (summary/stuck/scrub), pools (list/usable capacity), RBD images and snapshots, CephFS/MDS and RGW status, monitors/managers, slow ops and capacity forecast — plus governed writes (set cluster flags, reweight/mark-in/mark-out/purge OSDs, trigger scrubs, set pool quota/pg_num/autoscale/size, create/delete pools, create/delete RBD images and snapshots, throttle recovery/backfill). Always use this skill for \"ceph health\", \"what does this HEALTH_WARN mean\", \"PG_DEGRADED / OSD_NEARFULL / SLOW_OPS / MON_DOWN\", \"ceph -s\", \"which OSD is most full\", \"drain an OSD\", \"purge an OSD\", \"stuck PGs\", \"overdue scrub\", \"pool usable capacity\", \"set pool size / quota\", \"rebalance is too slow / throttle backfill\", \"RBD image or snapshot\", \"MDS behind on trimming\", \"RGW large omap\", \"mon quorum\", or \"days to nearfull\" when the context is a Ceph cluster (cephadm, hypervisor-bundled Ceph, or MicroCeph). Do NOT use when the target is not Ceph — a hypervisor, a different storage appliance, a backup product, a Kubernetes cluster, or a network device. Route those to the appropriate other AIops-tools skill (negative routing hint only). Common Ceph ops with a built-in governance harness (audit, policy, token budget, undo, risk-tiers).\n\nTags: agent-skills:0.1.0, ai-ops:0.1.0, ceph:0.1.0, latest:0.11.5, mcp:0.1.0, storage:0.1.0\n\nVersion history:\n\nv0.11.5 | 2026-09-16T23:25:19.214Z | auto\n\n- Removed the obsolete skill-card.md file.\n- Updated references/agent-guardrails.md documentation.\n- No changes to functionality or user-facing features.\n\nv0.11.4 | 2026-09-15T05:50:07.558Z | auto\n\n- Removed the sample file skill-card.md.\n- No functional or user-facing changes; housekeeping only.\n\nv0.11.3 | 2026-09-12T23:38:35.627Z | auto\n\n- Removed the redundant skill-card.md file.\n- Updated references/setup-guide.md (details not specified).\n- No changes to functionality or available tools.\n- General housekeeping and documentation cleanup.\n\nv0.11.2 | 2026-09-12T14:05:01.564Z | auto\n\nceph-aiops 0.11.2\n\n- Updated SKILL.md with minor OpenClaw plugin installation changes (now references clawhub:@zw008/ceph-aiops).\n- Removed skill-card.md from the project.\n\nv0.11.1 | 2026-09-12T09:57:36.189Z | auto\n\n- Added OpenClaw plugin installation instructions and usage notes to documentation.\n- Clarified that the skill requires the uvx binary on PATH for MCP server installation.\n- Removed the sample file \"skill-card.md\" from the project.\n- No functional or API changes; update is documentation-only.\n\nv0.11.0 | 2026-09-12T00:46:19.902Z | auto\n\n- Updated skill metadata: dropped mention of specific files and improved compatibility/env/bin requirements.\n- Removed obsolete file: skill-card.md.\n- Updated metadata: now accepts either ceph-aiops or uvx as required binaries, and clarified optional environment variables.\n- Improved negative routing and usage guidance in documentation.\n- No functional changes to tools or workflow; all 37 MCP tools remain available.\n\nv0.10.0 | 2026-08-10T06:49:56.092Z | auto\n\n- Removed the file: skill-card.md\n- No changes to functionality or features.\n\nv0.9.0 | 2026-08-03T05:51:44.739Z | auto\n\n- Removed the file skill-card.md.\n- No user-visible changes to skill functionality or documentation.\n- Version update to 0.9.0.\n\nv0.8.0 | 2026-08-02T09:37:50.668Z | auto\n\n- Removed the sample skill-card.md file.\n- No changes to feature set or functionality.\n\nv0.7.0 | 2026-07-21T15:31:00.295Z | auto\n\n- skill-card.md documentation file has been removed.\n- No changes to user-facing features or functionality.\n\nv0.6.0 | 2026-07-21T09:40:03.364Z | auto\n\nceph-aiops 0.6.0\n\n- Updated documentation: SKILL.md, setup-guide.md, and agent-guardrails.md received edits for accuracy and clarity.\n- The \"skill-card.md\" file was removed.\n- No functional or breaking changes noted; this release focuses on content and documentation improvements.\n\nv0.5.0 | 2026-07-20T11:14:09.854Z | auto\n\n- Removed the file skill-card.md.\n- No functional changes or new features have been introduced in this version.\n\nv0.4.1 | 2026-07-20T03:07:05.339Z | auto\n\n- Removed the file skill-card.md.\n- No changes to functionality or user experience.\n- Documentation and feature set remain the same.\n\nv0.4.0 | 2026-07-19T03:50:16.404Z | auto\n\nceph-aiops 0.4.0\n\n- Added two new \"undo\" operations: undo_list and undo_apply, expanding the toolset to 37 total tools (17 read, 18 write, 2 undo).\n- Introduced agent guardrails documentation (references/agent-guardrails.md).\n- Improved documentation with updated capability, CLI reference, and setup guide files.\n- Simplified and clarified the SKILL.md metadata and descriptions.\n- Removed the obsolete skill-card.md file.\n\nv0.3.0 | 2026-07-17T05:55:16.858Z | auto\n\n- Removed the sample file skill-card.md.\n- No functional or visible changes to the skill or its documentation.\n- Internal cleanup only; user experience and features remain unchanged.\n\nv0.2.0 | 2026-07-13T13:08:26.875Z | auto\n\n- Dropped support wording for \"Proxmox-hosted Ceph\"; now refers to \"hypervisor-bundled Ceph\" in both usage and compatibility.\n- Removed sample file skill-card.md from the repository.\n- No changes to toolset or governance; remains preview with mock validation only.\n- Documentation improvements for clarity in supported environments and usage recommendations.\n\nv0.1.0 | 2026-07-12T06:58:54.190Z | auto\n\nceph-aiops v0.1.0 (initial preview release)\n\n- Introduces standalone, governed Ceph operations via the ceph-mgr Dashboard REST API.\n- Provides 35 tools (17 read, 18 write) covering Ceph health, OSDs, placement groups, pools, RBD, CephFS, RGW, monitors/managers, slow ops, and capacity forecasting.\n- All write operations are audited, gated by policy, risk-tiered, and support undo/dry-run with double confirmation.\n- Secrets (dashboard passwords) are securely stored encrypted, never in plaintext.\n- Compatible with vanilla ceph-mgr (cephadm, Proxmox, MicroCeph); no Kubernetes or croit requirements.\n- PREVIEW: Tools are mock-validated; live cluster verification is recommended for production use.\n\nArchive index:\n\nArchive v0.11.5: 7 files, 18004 bytes\n\nFiles: references/agent-guardrails.md (7564b), references/capabilities.md (4566b), references/cli-reference.md (2640b), references/setup-guide.md (4891b), skill-card.md (2837b), SKILL.md (14394b), _meta.json (130b)\n\nFile v0.11.5:SKILL.md\n\n---\nname: ceph-aiops\nslug: ceph-aiops\ndisplayName: \"Ceph AIops\"\nsummary: \"Governed Ceph mgr ops: HEALTH_WARN RCA, OSD/PG/pool/RBD/CephFS/RGW, 37 tools.\"\nlicense: MIT\nhomepage: https://github.com/AIops-tools/Ceph-AIops\ntags: [aiops, mcp, governance, ceph]\ndescription: >\n  Use this skill whenever the user needs to operate or diagnose a Ceph cluster via its ceph-mgr Dashboard REST API — decode a HEALTH_WARN/ERR state into cause + action (cluster_health), read the cluster status, inspect OSDs (tree/df/perf), placement groups (summary/stuck/scrub), pools (list/usable capacity), RBD images and snapshots, CephFS/MDS and RGW status, monitors/managers, slow ops and capacity forecast — plus governed writes (set cluster flags, reweight/mark-in/mark-out/purge OSDs, trigger scrubs, set pool quota/pg_num/autoscale/size, create/delete pools, create/delete RBD images and snapshots, throttle recovery/backfill).\n  Always use this skill for \"ceph health\", \"what does this HEALTH_WARN mean\", \"PG_DEGRADED / OSD_NEARFULL / SLOW_OPS / MON_DOWN\", \"ceph -s\", \"which OSD is most full\", \"drain an OSD\", \"purge an OSD\", \"stuck PGs\", \"overdue scrub\", \"pool usable capacity\", \"set pool size / quota\", \"rebalance is too slow / throttle backfill\", \"RBD image or snapshot\", \"MDS behind on trimming\", \"RGW large omap\", \"mon quorum\", or \"days to nearfull\" when the context is a Ceph cluster (cephadm, hypervisor-bundled Ceph, or MicroCeph).\n  Do NOT use when the target is not Ceph — a hypervisor, a different storage appliance, a backup product, a Kubernetes cluster, or a network device. Route those to the appropriate other AIops-tools skill (negative routing hint only).\n  Common Ceph ops with a built-in governance harness (audit, policy, token budget, undo, risk-tiers).\ninstaller:\n  kind: uv\n  package: ceph-aiops\nargument-hint: \"[ceph question or describe your cluster task]\"\nallowed-tools:\n  - Bash\nmetadata: {\"openclaw\":{\"requires\":{\"anyBins\":[\"ceph-aiops\",\"uvx\"]},\"optional\":{\"env\":[\"CEPH_AIOPS_CONFIG\",\"CEPH_AIOPS_MASTER_PASSWORD\"]},\"homepage\":\"https://github.com/AIops-tools/Ceph-AIops\",\"emoji\":\"🐙\",\"os\":[\"macos\",\"linux\"]}}\ncompatibility: >\n  Standalone, self-governed Ceph operations. The governance harness (audit, policy, token/runaway budget, undo, risk-tiers) is bundled in the package — no external skill-family dependency. Works against vanilla ceph-mgr (cephadm / hypervisor-bundled Ceph / MicroCeph); no croit and no Kubernetes dependency.\n  All write operations are audited to a local SQLite DB under ~/.ceph-aiops/ (relocatable via CEPH_AIOPS_HOME).\n  Connection: the ceph-mgr Dashboard REST API over HTTPS (default port 8443). Authentication is username + password exchanged for a short-lived JWT at POST /api/auth; the mgr 'dashboard' module must be enabled. The username lives in config.yaml; the password is stored ENCRYPTED in ~/.ceph-aiops/secrets.enc (Fernet/AES-128 + scrypt-derived key) — never plaintext on disk. Run 'ceph-aiops init' to onboard, or 'ceph-aiops secret set <target>' to add one. The store is unlocked by a master password from CEPH_AIOPS_MASTER_PASSWORD (non-interactive/MCP/CI) or an interactive prompt (CLI on a TTY). A legacy plaintext env var CEPH_<TARGET_NAME_UPPER>_PASSWORD is still honoured as a fallback with a deprecation warning (migrate with 'ceph-aiops secret migrate'). The password is held only in memory and exchanged for a JWT at request time; secrets are never logged or echoed.\n  State-changing operations require double confirmation at the CLI layer and support --dry-run. All write tools pass through the @governed_tool decorator (pre-check + budget guard + audit + risk-tier label). High-risk destructive ops (osd_mark_out, osd_purge, pool_delete, set_pool_size, rbd_image_delete, rbd_snapshot_delete) require dry-run + double confirmation; reversible writes (osd_reweight, cluster_flag_set, set_pool_quota/pg_num/autoscale, throttle_recovery) capture the prior state and record an inverse undo descriptor.\n  Webhooks: none — no outbound network calls beyond the configured ceph-mgr Dashboard REST API.\n  SSL: verify_ssl defaults to true; disable only for self-signed lab certificates.\n  Transitive dependencies: httpx (HTTP client) and the MCP SDK. No post-install scripts or background services.\n  Validation status: behaviour is exercised against mocked Dashboard responses; multi-node rebalance behaviour and the write ops have not been run against a live cluster (a single-node MicroCeph running 'ceph-aiops doctor' is the cheapest live path; see docs/VERIFICATION.md). The Dashboard API has no ETag/pagination, so none are exposed.\n---\n\n# Ceph AIops\n\n> **Disclaimer**: Community-maintained open-source project, **not affiliated with, endorsed by, or sponsored by the Ceph project or any storage vendor.** Product and trademark names belong to their owners. Source at [github.com/AIops-tools/Ceph-AIops](https://github.com/AIops-tools/Ceph-AIops) under the MIT license.\n\nGoverned Ceph operations via the **ceph-mgr Dashboard REST API** — **37 MCP tools**, every one wrapped with the bundled `@governed_tool` harness: a local unified audit log under `~/.ceph-aiops/`, token/runaway budget guard, undo-token recording, and descriptive risk tiers. The Dashboard password is stored **encrypted** (`~/.ceph-aiops/secrets.enc`, Fernet + scrypt) — never plaintext on disk. The flagship `cluster_health` turns raw HEALTH_WARN/ERR check codes into plain-language cause + suggested action.\n\n> **Standalone**: the governance harness is bundled in the package (`ceph_aiops.governance`) — ceph-aiops has no external skill-family dependency. Works against vanilla ceph-mgr (cephadm / hypervisor-bundled / MicroCeph); no croit, no Kubernetes.\n\n## What This Skill Does\n\n| Group | Tools | Count | Read or Write |\n|-------|-------|:-----:|:-------------:|\n| **Health** | cluster_health (flagship RCA), cluster_status | 2 | 2 read |\n| **OSD** | osd_tree, osd_df, osd_perf | 3 | 3 read |\n| | cluster_flag_set, osd_reweight, osd_mark_in, osd_mark_out, osd_purge | 5 | 5 write |\n| **PG** | pg_summary, pg_dump_stuck, scrub_status | 3 | 3 read |\n| | trigger_scrub, trigger_deep_scrub | 2 | 2 write |\n| **Pool** | pool_ls, pool_df | 2 | 2 read |\n| | set_pool_quota, set_pool_pg_num, set_pool_autoscale, pool_create, set_pool_size, pool_delete | 6 | 6 write |\n| **RBD** | rbd_ls | 1 | 1 read |\n| | rbd_image_create, rbd_snapshot_create, rbd_image_delete, rbd_snapshot_delete | 4 | 4 write |\n| **CephFS / RGW** | cephfs_status, rgw_status | 2 | 2 read |\n| **Cluster-ops** | mon_status, mgr_status, slow_ops, capacity_forecast | 4 | 4 read |\n| | throttle_recovery | 1 | 1 write |\n| **Undo** | undo_list, undo_apply | 2 | 2 undo |\n\nTotals: **37 tools — 17 read, 18 write, 2 undo.** The MCP server exposes all 37; the CLI is a convenience subset.\n\n## Quick Install\n\n```bash\nuv tool install ceph-aiops\nceph-aiops init       # interactive wizard: mgr host/port/username + encrypted Dashboard password\nceph-aiops doctor\n```\n\nOr as an OpenClaw plugin, which installs this skill and its MCP server together:\n\n```bash\nopenclaw plugins install clawhub:@zw008/ceph-aiops\nopenclaw skills info ceph-aiops          # expect: Visible to model: yes\n```\n\nNeeds `uvx` on `PATH`: the MCP server is fetched with uv, pinned to this release.\n\n## When to Use This Skill\n\n- Decode a **HEALTH_WARN/ERR** state (`cluster_health` / `health detail`) — cause + action per active check (`PG_DEGRADED`, `OSD_NEARFULL`, `SLOW_OPS`, `MON_DOWN`, `LARGE_OMAP_OBJECTS`, …)\n- One-shot triage (`overview`): HEALTH status + active checks + OSD up/in counts\n- Inspect OSDs (`osd_tree` / `osd_df` most-full first / `osd_perf` slowest first), PGs (`pg_summary` / `pg_dump_stuck` / `scrub_status`), pools (`pool_ls` / `pool_df` usable capacity)\n- Investigate slow requests (`slow_ops`), MDS trimming lag (`cephfs_status`), RGW large-omap (`rgw_status`), mon quorum (`mon_status`), and days-to-nearfull (`capacity_forecast`)\n- Safely **drain + purge** an OSD, change **pool size/quota**, or **throttle a slow rebalance** (governed writes with dry-run + undo)\n\n**Do NOT use when** the target is not Ceph (a hypervisor, another storage appliance, a backup product, a container cluster, or a network device). Route those to the appropriate **other AIops-tools** skill.\n\n## Related Skills — Skill Routing\n\n| If the user wants… | Use |\n|--------------------|-----|\n| Ceph: HEALTH_WARN RCA, OSD/PG/pool/RBD/CephFS/RGW, rebalance, slow ops | **ceph-aiops** (this skill) |\n| Any non-Ceph target (hypervisor, other storage, backup, cluster, network) | the appropriate **other AIops-tools** skill |\n\n## Common Workflows\n\n### 1. \"The cluster went HEALTH_WARN overnight\" — decode it (read-only)\n\n1. `ceph-aiops doctor` → confirm the mgr Dashboard is reachable and the JWT login works before trusting anything else\n2. `ceph-aiops overview` → HEALTH status, the list of active check codes, and OSD up/in counts in one shot\n3. `ceph-aiops health detail` (MCP: `cluster_health`) → each **active** check translated into what it means, the likely cause, and a suggested action\n4. Drill into the implicated resource: `PG_DEGRADED` → `pg_dump_stuck`; `OSD_NEARFULL` → `ceph-aiops osd df` (most-full first); `SLOW_OPS` → `slow_ops`; `MON_DOWN` → `mon_status`; `LARGE_OMAP_OBJECTS` → `rgw_status`\n5. **Failure branch**: if `doctor` fails on auth, the Dashboard password is wrong or the store is locked — re-run `ceph-aiops secret set <target>` (or export `CEPH_AIOPS_MASTER_PASSWORD` for non-interactive use). If `doctor` fails on reachability, the mgr `dashboard` module is likely not enabled; no read is issued against an unauthenticated session.\n\n### 2. Retire a failing OSD: drain, mark out, purge (governed)\n\n1. `ceph-aiops health detail` → confirm the OSD is genuinely the problem (e.g. `OSD_SLOW_PING_TIME`, repeated `SLOW_OPS` on one id) rather than a cluster-wide symptom\n2. `ceph-aiops osd df` → confirm the id, and that the remaining OSDs have room to absorb its data before you drain anything\n3. `ceph-aiops osd reweight <id> 0.0` → start a gradual drain; reversible, the prior CRUSH weight is captured as the undo descriptor\n4. `ceph-aiops osd out <id> --dry-run`, then re-run without `--dry-run` → **high** risk, double confirmation, needs `CEPH_AUDIT_APPROVED_BY`\n5. Wait for `ceph-aiops health status` / `pg_summary` to show all PGs `active+clean` — do not purge while backfill is running\n6. `ceph-aiops osd purge <id> --dry-run`, then re-run without `--dry-run` → **high**, irreversible\n7. **Failure branch**: if client I/O tanks during the drain, stop and reverse — `ceph-aiops undo list` then `ceph-aiops undo apply <id>` restores the prior weight (and `osd_mark_in` reverses the mark-out). Purge has no undo, which is exactly why it comes last and after `active+clean`.\n\n### 3. Recovery is starving client I/O\n\n1. `ceph-aiops health detail` → confirm the cluster is actually backfilling/recovering (`PG_DEGRADED`, `PG_BACKFILL_FULL`) rather than hitting a different bottleneck\n2. `slow_ops` → check whether client requests are genuinely being blocked, and by which OSDs\n3. `throttle_recovery(max_backfills=1, recovery_max_active=1)` → **med** risk, reversible; the prior `osd_max_backfills` / `osd_recovery_max_active` are captured as the undo descriptor\n4. Re-check `slow_ops` and `pg_summary` — recovery is slower but client latency should recover\n5. Once the cluster is quiet, raise the values back (or `ceph-aiops undo apply <id>` to restore the exact prior settings)\n6. **Failure branch**: if throttling does not help, the bottleneck is not recovery — go back to `osd_perf` (slowest OSDs first) and `mon_status`; do not keep lowering the throttle, you will only extend the degraded window.\n\n### 4. A pool is running out of usable capacity\n\n1. `ceph-aiops overview` → look for `POOL_NEARFULL` / `OSD_NEARFULL` among the active checks\n2. `pool_df` → per-pool usage with **usable capacity = raw ÷ size** (a `size=3` pool reports a third of raw — this is where most \"but the disks aren't full\" confusion comes from)\n3. `capacity_forecast` → days-to-nearfull at the current fill rate, so you know whether this is a this-week problem or a this-quarter one\n4. Buy time reversibly first: `set_pool_quota` (med, undo → prior quota) or `set_pool_autoscale` (med, undo) to let pg_num track the new size\n5. Only if a replica change is genuinely the answer: `set_pool_size --dry-run` then the real call — **high** risk, because lowering `size` reduces durability and any change forces cluster-wide data movement\n6. **Failure branch**: if the resulting rebalance saturates the cluster, apply workflow 3 (`throttle_recovery`) rather than reverting the size mid-flight; if the size change itself was wrong, `ceph-aiops undo apply <id>` replays the recorded prior value — but expect a second full rebalance.\n\n## Governance & Safety\n\nThe skill delivers reads and writes and records them; it does **not** decide\nwhether a write is permitted. That is your agent's judgement, or the permission\nof the account you connect it with (a ceph-mgr Dashboard account with a\nread-only role — writes then fail at the mgr). There is no read-only switch,\npolicy file, or approval gate.\n\n- **Audit is the guarantee, and it is not bypassable.** Every operation — MCP and CLI alike — is logged to `~/.ceph-aiops/audit.db` (relocatable via `CEPH_AIOPS_HOME`): params, result, status, duration, and the risk tier. The CLI writes the same row the MCP path does.\n- `CEPH_AUDIT_APPROVED_BY` / `CEPH_AUDIT_RATIONALE` are optional annotations recorded on the audit row (who/why); they are never required and never block.\n- **Runaway guard** — a safety backstop, not authorization: the same call looped in a tight window trips a circuit breaker. Disable with `CEPH_RUNAWAY_MAX=0`.\n- Destructive writes support `--dry-run` / `dry_run=True` and double confirmation at the CLI.\n- Reversible writes fetch the real before-state and record an inverse descriptor (`osd_reweight`→restore prior weight, `cluster_flag_set`→toggle back); irreversible ops (`osd_purge`, `pool_delete`, RBD deletes) record only the before-state.\n\n## References\n\n- `references/capabilities.md` — full tool → API-path → returns reference\n- `references/cli-reference.md` — CLI command reference\n- `references/setup-guide.md` — onboarding, credentials, and connectivity\n\nFile v0.11.5:_meta.json\n\n{\n  \"ownerId\": \"kn7b067awq2s97bn3d7p5qfhw5827pxc\",\n  \"slug\": \"ceph-aiops\",\n  \"version\": \"0.11.5\",\n  \"publishedAt\": 1789601119214\n}\n\nFile v0.11.5:references/agent-guardrails.md\n\n# Agent guardrails — running ceph-aiops with a smaller / local model\n\nIf you drive these tools with a local model (Llama, Qwen, Mistral … via Goose,\nOllama, LM Studio, or any OpenAI-compatible runtime), you will get noticeably\nbetter results with a short system prompt. This page gives you one, and — more\nimportantly — tells you which guardrails you **no longer need to write**, because\nthe tool now enforces them itself.\n\nThe distinction matters. A guardrail in a prompt is a request. A guardrail in the\nharness is a guarantee. Anything below that we could move into the harness, we did.\n\n## Authorization is not this tool's job — decide it where it belongs\n\nWhether a write should happen is your decision, or the account's. The tool does\nnot gate it — there is no read-only switch and no approval prompt to configure.\nThe two right places to control read vs write:\n\n- **The account you connect with.** Give it a ceph-mgr Dashboard account with a\n  read-only role. A write then fails at the mgr, which is the only place the\n  permission actually lives — no skill-side flag can be argued around by a\n  model, but a revoked permission cannot be.\n- **Your agent's system prompt.** If you want an observe-only session, tell the\n  model not to call the write tools (they are clearly tagged `[WRITE]`).\n\nWhat the tool *does* guarantee is that you can always see what happened:\n\n## What the tool enforces — do not waste prompt budget on these\n\n| You might be tempted to prompt | Why you don't need to |\n|---|---|\n| \"Log everything you do, over both MCP and the CLI\" | Every call is audited to `~/.ceph-aiops/audit.db` regardless of what the model says it did — and the CLI writes the same row the MCP path does, so there is no unaudited entry point. Reversible writes also record an undo token capturing the *prior* state. |\n| \"Don't invent a value when a field is missing\" | A field the Dashboard did not return comes back as `null`, never as `\"\"`. A missing `deviceClass`, `host`, MDS `state`, or `pg_autoscale_mode` is distinguishable from an empty one in the payload. |\n| \"Tell me if the output was cut off\" | `pg_dump_stuck` returns `{\"stuck\": [...], \"returned\": N, \"limit\": L, \"truncated\": true/false}` and `pg_summary` the same shape under `unhealthy` (plus `states`) — the list key differs, so look it up per tool rather than expecting `stuck` everywhere. Truncation is measured (one extra row is collected), not guessed. `pg_summary` also keeps `unhealthyCount` as the true total even when the list is capped. |\n| \"Explain what HEALTH_WARN means\" | `cluster_health` already folds each active check code (`PG_DEGRADED`, `OSD_NEARFULL`, `SLOW_OPS`, `LARGE_OMAP_OBJECTS`, …) into a plain-language `cause` and `suggestedAction`. The model should quote those, not compose its own. |\n| \"Confirm before anything destructive\" | Every destructive tool (`osd_purge`, `osd_mark_out`, `pool_delete`, `rbd_image_delete`, `rbd_snapshot_delete`, `set_pool_size`) takes `dry_run=True` for a preview and is `risk=high`, tagged `review` on its audit row. ⚠️ **The double confirmation is a CLI feature, and only `osd_purge` and `osd_mark_out` have CLI commands** — the other four are reachable only over MCP, where nothing prompts. Keep your own confirmation for those. |\n| \"Don't get stuck retrying\" | The runaway guard trips a circuit breaker if the same call is hammered in a tight loop — a stuck agent is stopped rather than left to burn calls and time. |\n\n## What still needs a prompt\n\nThese are model-behaviour problems the harness cannot fix from the outside.\nCopy this into your agent's system prompt:\n\n```text\nYou operate a Ceph cluster through the ceph-aiops MCP tools, which talk to the\nceph-mgr Dashboard REST API.\n\nTOOL USE\n- Before answering any question about the current cluster, you MUST call a tool.\n  Never answer from memory or assumption.\n- Actually invoke the tool. Do not describe the call you would make, and do not\n  emit an example JSON response in place of calling it.\n- If a tool call fails, report the real error verbatim. Never fill the gap with\n  a plausible-sounding answer. A read that fails returns an \"error\" field rather\n  than raising — treat that as \"unknown\", not as \"healthy\".\n\nREADING RESULTS\n- Read the whole result before concluding. If a result contains a \"truncated\"\n  field that is true, say so and re-run with a higher limit instead of treating\n  the partial result as complete.\n- A null field means the Dashboard did not return that value. Report it as \"not\n  available\" — never infer it.\n- Report values exactly as returned. Do not normalise, translate, or prettify\n  status strings (HEALTH_WARN, active+undersized+degraded), PG ids, or OSD ids.\n- When cluster_health returns findings, quote each finding's \"cause\" and\n  \"suggestedAction\" rather than composing your own explanation of the check code.\n\n- `pool_delete`, `rbd_image_delete`, `rbd_snapshot_delete` and `set_pool_size` have no CLI\n  command, so nothing will ask you to confirm them. Call them with `dry_run=True` first,\n  show the operator what would change, and wait for an explicit go-ahead.\n\nSCOPE\n- Separate observation from interpretation. State what the tools returned, then\n  any interpretation, clearly marked as such.\n- Do not assert a capacity, performance, or data-loss problem unless a tool\n  result supports it. HEALTH_WARN is not automatically an emergency —\n  PG_NOT_DEEP_SCRUBBED on a small cluster is routine.\n- Do not confuse the identifier kinds: an OSD id is a number (3), a PG id is\n  pool.hex (\"2.1a\"), a pool name is a string, and an RBD image is addressed as\n  pool/name. Never pass one where another is expected.\n- capacity_forecast is arithmetic extrapolation from a growth rate you supply.\n  With no growth rate it reports \"insufficient-data\" — do not present that as a\n  prediction.\n```\n\n## Recommended setup for a local model\n\nStart with a connection that *cannot* write, verify, and widen the account's\npermission only when you trust the setup — the destructive operations on a Ceph\ncluster are unusually cheap to invoke and unusually expensive to undo\n(`pool_delete` and `rbd_image_delete` destroy data no undo token can bring back):\n\n```bash\n# e.g. use a ceph-mgr Dashboard account with a read-only role. Then:\nceph-aiops doctor\n```\n\nOptionally annotate the audit trail with who is operating and why — recorded on\nevery row, never required:\n\n```bash\nexport CEPH_AUDIT_APPROVED_BY=\"your.name@example.com\"\nexport CEPH_AUDIT_RATIONALE=\"draining osd.7 for disk replacement 2026-07-20\"\n```\n\n## If your model still struggles\n\nSome behaviours are model-capacity limits rather than prompt problems:\n\n- **Multi-tool workflows time out or drift.** Prefer `cluster_health` and\n  `fleet_overview` — they do the multi-step correlation inside one call, so the\n  model does not have to chain reads and keep OSD/PG ids straight.\n- **The model ignores later tool results in a long context.** Ask narrower\n  questions and use `limit` deliberately rather than dumping every PG in the\n  cluster; `pg_summary`'s histogram is usually the right level of detail.\n- **The model describes calls instead of making them.** This is usually a\n  runtime/tool-calling-format mismatch, not a prompt problem — check that your\n  client advertises the tools in the format your model was trained on.\n\nFeedback on running this with a specific local model is genuinely useful —\nopen an issue at\n[github.com/AIops-tools/Ceph-AIops](https://github.com/AIops-tools/Ceph-AIops/issues)\nwith the model, runtime, and what went wrong.\n\nFile v0.11.5:references/capabilities.md\n\n# ceph-aiops capabilities\n\n> 37 MCP tools (17 read, 18 write, 2 undo) over the **ceph-mgr\n> Dashboard REST API** (`https://<host>:8443`, JWT via `POST /api/auth`).\n> Multi-node rebalance behaviour and the write ops need live verification\n> (see `docs/VERIFICATION.md`).\n\n## Read tools (17)\n\n| Tool | API path | Returns |\n|------|----------|---------|\n| `cluster_health` | `GET /api/health/full` | **flagship RCA** — per active HEALTH_WARN/ERR check: code, plain-language meaning, likely cause, suggested action |\n| `cluster_status` | `GET /api/health/minimal` | `ceph -s` summary: health status, mon/mgr/osd/pg counts |\n| `osd_tree` | `GET /api/osd` | OSD tree: up/in, CRUSH weight, host, device class |\n| `osd_df` | `GET /api/osd` | per-OSD utilization %, **most-full first**, near-full / backfill-full flags |\n| `osd_perf` | `GET /api/osd` | commit/apply latency per OSD, **slowest first** |\n| `pg_summary` | `GET /api/pg` (+ `/api/health/full`) | PG **state histogram** + list of non-active+clean PGs |\n| `pg_dump_stuck` | `GET /api/pg` | stuck PGs (inactive/unclean/stale/undersized) + implicated OSDs |\n| `scrub_status` | `GET /api/health/full` | PGs overdue for scrub / deep-scrub |\n| `pool_ls` | `GET /api/pool` | pools: name, id, size, pg_num, autoscale mode, application |\n| `pool_df` | `GET /api/pool` | per-pool usage; **usable capacity = raw ÷ size** |\n| `rbd_ls` | `GET /api/block/image` | RBD images (optionally filtered by pool): name, size, pool |\n| `cephfs_status` | `GET /api/cephfs` | MDS ranks + **\"behind on trimming\"** + client count |\n| `rgw_status` | `GET /api/rgw/daemon` + `GET /api/rgw/bucket` | RGW daemons + buckets + **LARGE_OMAP / unsharded-index** findings |\n| `mon_status` | `GET /api/monitor` | monitors: in-quorum vs **out-of-quorum** |\n| `mgr_status` | `GET /api/health/full` | active mgr, standbys, enabled modules |\n| `slow_ops` | `GET /api/health/full` | blocked / slow requests grouped **by OSD** |\n| `capacity_forecast` | `GET /api/osd` (+ df) | raw/used/avail + **days-to-nearfull** projection |\n\n## Write tools (18)\n\n| Tool | Risk | API path | Undo / safety |\n|------|------|----------|---------------|\n| `cluster_flag_set` | medium | `GET`+`PUT /api/osd/flags` | set/unset noout/noscrub/nobackfill/norecover; captures prior flag set (undo) |\n| `osd_reweight` | medium | `POST /api/osd/{id}/reweight` | 0.0 = drain; captures prior weight (undo) |\n| `osd_mark_in` | medium | `POST /api/osd/{id}/mark` | captures prior up/in state (undo) |\n| `osd_mark_out` | **high** | `POST /api/osd/{id}/mark` | drains data; CLI double-confirm + dry-run; captures prior state |\n| `osd_purge` | **high** | `DELETE /api/osd/{id}` | destroy + crush rm + auth del; **irreversible**; dry-run + double-confirm |\n| `trigger_scrub` | medium | `POST /api/pg/{pgid}/scrub` | schedule a shallow scrub; no prior state |\n| `trigger_deep_scrub` | medium | `POST /api/pg/{pgid}/deep_scrub` | schedule a deep (data-integrity) scrub |\n| `set_pool_quota` | medium | `PUT /api/pool/{name}` | captures prior max_bytes/max_objects (undo) |\n| `set_pool_pg_num` | medium | `PUT /api/pool/{name}` | captures prior pg_num (undo) |\n| `set_pool_autoscale` | medium | `PUT /api/pool/{name}` | captures prior autoscale mode (undo) |\n| `pool_create` | medium | `POST /api/pool` | create a new pool |\n| `set_pool_size` | **high** | `PUT /api/pool/{name}` | replica change **forces data movement**; dry-run + double-confirm; captures prior size |\n| `pool_delete` | **high** | `DELETE /api/pool/{name}` | **destroys all data**; dry-run + double-confirm |\n| `rbd_image_create` | medium | `POST /api/block/image` | create an RBD image |\n| `rbd_snapshot_create` | medium | `POST /api/block/image/{spec}/snap` | reversible → delete the snapshot |\n| `rbd_image_delete` | **high** | `DELETE /api/block/image/{spec}` | **irreversible**; dry-run + double-confirm |\n| `rbd_snapshot_delete` | **high** | `DELETE /api/block/image/{spec}/snap/{snap}` | **irreversible**; dry-run + double-confirm |\n| `throttle_recovery` | medium | `GET`+`POST /api/cluster_conf` | tunes `osd_max_backfills` / `osd_recovery_max_active`; captures prior values (undo) |\n\n## Out of scope (by design)\n\n- RGW **multisite** replication and zone/zonegroup management\n- **NFS-Ganesha** exports\n- **cephadm orchestrator** host/daemon management (add/remove hosts, deploy daemons)\n- Per-daemon config sprawl beyond the recovery-tuning keys above\n\nThe ceph-mgr Dashboard API has no ETag / pagination, so this tool exposes none.\n\nWant one of these? Open an issue or PR — feedback and contributions welcome.\n\nFile v0.11.5:references/cli-reference.md\n\n# ceph-aiops CLI reference\n\n> The CLI is a convenience subset; the full 37-tool surface\n> is via the MCP server (`ceph-aiops mcp`). Talks to the ceph-mgr Dashboard REST\n> API (`https://<host>:8443`, JWT via `POST /api/auth`).\n\n## Setup & diagnostics\n\n```bash\nceph-aiops init                      # interactive onboarding wizard\nceph-aiops doctor [--skip-auth]      # config + secret store + JWT login + mgr-dashboard reachability\nceph-aiops mcp                       # start the MCP server (stdio transport)\n```\n\n## Secrets (encrypted store ~/.ceph-aiops/secrets.enc)\n\n```bash\nceph-aiops secret set <target> [--value <password>]  # store Dashboard password (hidden prompt if no --value)\nceph-aiops secret list                               # names only — values never shown\nceph-aiops secret rm <target>\nceph-aiops secret migrate                            # import legacy plaintext .env (CEPH_<T>_PASSWORD)\nceph-aiops secret rotate-password                    # re-encrypt under a new master password\n```\n\n## Read commands\n\n```bash\nceph-aiops overview [--target <t>]        # HEALTH status + active checks + OSD up/in\nceph-aiops health detail                  # decode active HEALTH_WARN/ERR checks → cause + action (RCA)\nceph-aiops health status                  # ceph -s summary\nceph-aiops osd tree                        # OSD tree: up/in, weight, host, device class\nceph-aiops osd df                          # per-OSD utilization, most-full first, near/backfill-full flags\n```\n\n## Write commands (governed; risk tier in parentheses)\n\n```bash\nceph-aiops osd reweight <osd_id> <weight> [--dry-run]   # (med) 0.0 = drain; reversible → prior weight\nceph-aiops osd out <osd_id> [--dry-run]                 # (high) mark out — drains data; double confirm\nceph-aiops osd purge <osd_id> [--dry-run]               # (high) purge — irreversible; double confirm\n```\n\nThe remaining writes (cluster flags, pool quota/pg_num/autoscale/size/create/delete,\nRBD image/snapshot create/delete, trigger scrubs, throttle recovery) are exposed\nthrough the **MCP server**, not the CLI.\n\n## Common options\n\n- `--target, -t <name>` — target name from `config.yaml` (omit to use the default/first target)\n- `--dry-run` — print the API call that would be made, change nothing\n- Destructive commands (`osd out`, `osd purge`) require `--dry-run` review + double confirmation\n- `doctor --skip-auth` — skip the JWT login / connectivity check (config + secret-store checks only)\n\n## Approver env vars (high-risk ops)\n\n```bash\nexport CEPH_AUDIT_APPROVED_BY='you@example.com'\nexport CEPH_AUDIT_RATIONALE='draining failed OSD 7 per ticket OPS-123'\n```\n\nFile v0.11.5:references/setup-guide.md\n\n# ceph-aiops setup & security guide\n\n> The cheapest live check is a single-node MicroCeph running `ceph-aiops doctor`.\n> See `docs/VERIFICATION.md` for the full live-verification checklist.\n\n## 1. Install\n\n```bash\nuv tool install ceph-aiops\n```\n\n## 2. Enable the ceph-mgr Dashboard module\n\nceph-aiops talks to the **ceph-mgr Dashboard REST API** (HTTPS, default port\n`8443`). The mgr **dashboard** module must be enabled and a Dashboard user must\nexist:\n\n```bash\nceph mgr module enable dashboard\n# Start read-only. This tool does not decide whether a write is allowed — the\n# Dashboard role does — so the role you pick here IS the authorization boundary.\nceph dashboard ac-user-create <username> -i <password-file> read-only\n# find the URL/port: ceph mgr services   → e.g. https://<host>:8443/\n```\n\nGrant `administrator` only if you intend the agent to perform writes (set flags,\nreweight/mark-out/purge OSDs, scrub, pool and RBD create/delete), and prefer a\ndedicated account for it rather than reusing a human's.\n\n> **Which Dashboard role each endpoint needs has not been verified per endpoint\n> against a live cluster** — only that the role, not this tool, is what decides.\n> If a read is refused under `read-only`, that is Ceph's role boundary doing its\n> job: widen the role deliberately, or report the endpoint on the issue tracker\n> so this note can be replaced with a measured list.\n\nceph-aiops authenticates by exchanging the **username + password** for a\nshort-lived **JWT** at `POST /api/auth`; the token is cached in memory and used\nas a Bearer header for subsequent calls.\n\n## 3. Onboard\n\n```bash\nceph-aiops init\n```\n\nThe wizard collects (non-secret) connection details into\n`~/.ceph-aiops/config.yaml` and stores the Dashboard **password** encrypted into\n`~/.ceph-aiops/secrets.enc`. Example config:\n\n```yaml\ntargets:\n  - name: ceph1\n    host: 10.0.0.30\n    port: 8443\n    username: ceph-aiops       # the read-only Dashboard user created above\n    verify_ssl: true           # false only for self-signed lab certs\n```\n\nThe `username` lives in the config file (it is not a secret); the password never\ndoes.\n\n## 4. Non-interactive use (MCP server / CI / cron)\n\nExport the master password so the encrypted store can be unlocked without a\nprompt:\n\n```bash\nexport CEPH_AIOPS_MASTER_PASSWORD='your-master-password'\n```\n\n## Credential security\n\n- The Dashboard password is **never** written to disk in plaintext. It lives only\n  in `~/.ceph-aiops/secrets.enc`, encrypted with Fernet (AES-128-CBC + HMAC), the\n  key derived from your master password via scrypt. Only a per-store random salt\n  and the ciphertext are on disk (chmod 600); the master password itself is never\n  stored.\n- A legacy plaintext env var `CEPH_<TARGET_NAME_UPPER>_PASSWORD` is still honoured\n  as a fallback with a deprecation warning — migrate with\n  `ceph-aiops secret migrate` (it imports then renames the old `.env`).\n- The password is held only in memory, exchanged for a JWT at request time, and\n  is never logged or echoed; exception text and tracebacks are scrubbed of\n  secret-shaped strings before being written to the audit log.\n\n## Audit-annotation env vars (optional)\n\nThe skill does not decide whether a write is permitted — that is the agent's\njudgement or the connecting Dashboard account's role. If you want the audit trail\nto record *who* ran a destructive op and *why*, set these; they are recorded on\nthe row, never required, and gate nothing:\n\n```bash\nexport CEPH_AUDIT_APPROVED_BY='you@example.com'\nexport CEPH_AUDIT_RATIONALE='why this destructive op is justified'\n```\n\n## Governance harness state\n\nState lives under `~/.ceph-aiops/` (relocate with `CEPH_AIOPS_HOME`):\n\n- `audit.db` — every tool call (SQLite), with risk tier and any approver/rationale\n- `undo.db` — inverse descriptors for reversible writes (e.g. `osd_reweight`,\n  `set_pool_quota`, `throttle_recovery`)\n- budget / runaway guard — caps cumulative tool calls and wall-time; trips on\n  tight poll/retry loops\n\n## Self-test free with MicroCeph\n\nThe cheapest **live** path — a single-node cluster on one box:\n\n```bash\nsnap install microceph\nmicroceph cluster bootstrap\nmicroceph disk add loop,4G,3        # 3 loop-file OSDs\n# enable dashboard + create a user (see step 2), then:\nceph-aiops init\nceph-aiops doctor\n```\n\nA 3-node Vagrant cluster exercises real rebalance/backfill behaviour (draining an\nOSD, changing pool size) that a single node cannot.\n\n## Note: no ETag / pagination\n\nThe ceph-mgr Dashboard API offers neither ETag caching nor pagination, so\nceph-aiops exposes none — nothing is missing, the upstream API simply doesn't\nprovide them.\n\n## Verify\n\n```bash\nceph-aiops doctor\n```\n\n`doctor` checks the config file, the encrypted store and its permissions, that a\npassword is present per target, and (unless `--skip-auth`) connectivity by\nperforming the JWT login against the mgr Dashboard.\n\nFile v0.11.5:skill-card.md\n\n## Description:\n\nceph-aiops helps agents diagnose and operate Ceph clusters through the ceph-mgr Dashboard REST API, including HEALTH_WARN and HEALTH_ERR root-cause analysis, OSD, PG, pool, RBD, CephFS, and RGW inspection, capacity forecasting, and governed write workflows.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[zw008](https://clawhub.ai/user/zw008)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nStorage operators, SREs, and infrastructure engineers use this skill to inspect Ceph cluster health, diagnose active warnings and errors, and carry out governed maintenance actions such as OSD drains, pool changes, RBD operations, recovery throttling, and capacity checks.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: MCP tools can perform irreversible Ceph cluster changes without an enforced confirmation or read-only gate.\n\nMitigation: Use a dedicated read-only Ceph Dashboard account by default, enable write privileges only for controlled sessions, and require explicit operator approval before destructive MCP calls.\n\nRisk: Destructive operations such as pool deletion, RBD deletion, snapshot deletion, and pool size changes can cause data loss or cluster-wide data movement.\n\nMitigation: Run dry-run previews first, review the planned change with the operator, and prefer reversible actions with undo records before irreversible changes.\n\nRisk: Write operations and multi-node rebalance behavior have not been fully verified against a live multi-node cluster in the provided artifact evidence.\n\nMitigation: Validate behavior in a lab Ceph environment before production use and start with low-risk read-only diagnostics such as doctor, overview, and health detail.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/zw008/skills/ceph-aiops)\n- [Project homepage](https://github.com/AIops-tools/Ceph-AIops)\n- [Capabilities reference](references/capabilities.md)\n- [CLI reference](references/cli-reference.md)\n- [Setup and security guide](references/setup-guide.md)\n- [Agent guardrails](references/agent-guardrails.md)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, Shell commands, Configuration, Guidance]\n\n**Output Format:** [Markdown guidance with inline shell commands and structured Ceph operation summaries.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May include risk-tier labels, dry-run recommendations, undo guidance, and references to Ceph Dashboard API-derived results.]\n\n## Skill Version(s):\n\n0.11.5 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v0.11.4: 7 files, 17583 bytes\n\nFiles: references/agent-guardrails.md (6913b), references/capabilities.md (4566b), references/cli-reference.md (2640b), references/setup-guide.md (4891b), skill-card.md (2499b), SKILL.md (14394b), _meta.json (130b)\n\nFile v0.11.4:SKILL.md\n\n---\nname: ceph-aiops\nslug: ceph-aiops\ndisplayName: \"Ceph AIops\"\nsummary: \"Governed Ceph mgr ops: HEALTH_WARN RCA, OSD/PG/pool/RBD/CephFS/RGW, 37 tools.\"\nlicense: MIT\nhomepage: https://github.com/AIops-tools/Ceph-AIops\ntags: [aiops, mcp, governance, ceph]\ndescription: >\n  Use this skill whenever the user needs to operate or diagnose a Ceph cluster via its ceph-mgr Dashboard REST API — decode a HEALTH_WARN/ERR state into cause + action (cluster_health), read the cluster status, inspect OSDs (tree/df/perf), placement groups (summary/stuck/scrub), pools (list/usable capacity), RBD images and snapshots, CephFS/MDS and RGW status, monitors/managers, slow ops and capacity forecast — plus governed writes (set cluster flags, reweight/mark-in/mark-out/purge OSDs, trigger scrubs, set pool quota/pg_num/autoscale/size, create/delete pools, create/delete RBD images and snapshots, throttle recovery/backfill).\n  Always use this skill for \"ceph health\", \"what does this HEALTH_WARN mean\", \"PG_DEGRADED / OSD_NEARFULL / SLOW_OPS / MON_DOWN\", \"ceph -s\", \"which OSD is most full\", \"drain an OSD\", \"purge an OSD\", \"stuck PGs\", \"overdue scrub\", \"pool usable capacity\", \"set pool size / quota\", \"rebalance is too slow / throttle backfill\", \"RBD image or snapshot\", \"MDS behind on trimming\", \"RGW large omap\", \"mon quorum\", or \"days to nearfull\" when the context is a Ceph cluster (cephadm, hypervisor-bundled Ceph, or MicroCeph).\n  Do NOT use when the target is not Ceph — a hypervisor, a different storage appliance, a backup product, a Kubernetes cluster, or a network device. Route those to the appropriate other AIops-tools skill (negative routing hint only).\n  Common Ceph ops with a built-in governance harness (audit, policy, token budget, undo, risk-tiers).\ninstaller:\n  kind: uv\n  package: ceph-aiops\nargument-hint: \"[ceph question or describe your cluster task]\"\nallowed-tools:\n  - Bash\nmetadata: {\"openclaw\":{\"requires\":{\"anyBins\":[\"ceph-aiops\",\"uvx\"]},\"optional\":{\"env\":[\"CEPH_AIOPS_CONFIG\",\"CEPH_AIOPS_MASTER_PASSWORD\"]},\"homepage\":\"https://github.com/AIops-tools/Ceph-AIops\",\"emoji\":\"🐙\",\"os\":[\"macos\",\"linux\"]}}\ncompatibility: >\n  Standalone, self-governed Ceph operations. The governance harness (audit, policy, token/runaway budget, undo, risk-tiers) is bundled in the package — no external skill-family dependency. Works against vanilla ceph-mgr (cephadm / hypervisor-bundled Ceph / MicroCeph); no croit and no Kubernetes dependency.\n  All write operations are audited to a local SQLite DB under ~/.ceph-aiops/ (relocatable via CEPH_AIOPS_HOME).\n  Connection: the ceph-mgr Dashboard REST API over HTTPS (default port 8443). Authentication is username + password exchanged for a short-lived JWT at POST /api/auth; the mgr 'dashboard' module must be enabled. The username lives in config.yaml; the password is stored ENCRYPTED in ~/.ceph-aiops/secrets.enc (Fernet/AES-128 + scrypt-derived key) — never plaintext on disk. Run 'ceph-aiops init' to onboard, or 'ceph-aiops secret set <target>' to add one. The store is unlocked by a master password from CEPH_AIOPS_MASTER_PASSWORD (non-interactive/MCP/CI) or an interactive prompt (CLI on a TTY). A legacy plaintext env var CEPH_<TARGET_NAME_UPPER>_PASSWORD is still honoured as a fallback with a deprecation warning (migrate with 'ceph-aiops secret migrate'). The password is held only in memory and exchanged for a JWT at request time; secrets are never logged or echoed.\n  State-changing operations require double confirmation at the CLI layer and support --dry-run. All write tools pass through the @governed_tool decorator (pre-check + budget guard + audit + risk-tier label). High-risk destructive ops (osd_mark_out, osd_purge, pool_delete, set_pool_size, rbd_image_delete, rbd_snapshot_delete) require dry-run + double confirmation; reversible writes (osd_reweight, cluster_flag_set, set_pool_quota/pg_num/autoscale, throttle_recovery) capture the prior state and record an inverse undo descriptor.\n  Webhooks: none — no outbound network calls beyond the configured ceph-mgr Dashboard REST API.\n  SSL: verify_ssl defaults to true; disable only for self-signed lab certificates.\n  Transitive dependencies: httpx (HTTP client) and the MCP SDK. No post-install scripts or background services.\n  Validation status: behaviour is exercised against mocked Dashboard responses; multi-node rebalance behaviour and the write ops have not been run against a live cluster (a single-node MicroCeph running 'ceph-aiops doctor' is the cheapest live path; see docs/VERIFICATION.md). The Dashboard API has no ETag/pagination, so none are exposed.\n---\n\n# Ceph AIops\n\n> **Disclaimer**: Community-maintained open-source project, **not affiliated with, endorsed by, or sponsored by the Ceph project or any storage vendor.** Product and trademark names belong to their owners. Source at [github.com/AIops-tools/Ceph-AIops](https://github.com/AIops-tools/Ceph-AIops) under the MIT license.\n\nGoverned Ceph operations via the **ceph-mgr Dashboard REST API** — **37 MCP tools**, every one wrapped with the bundled `@governed_tool` harness: a local unified audit log under `~/.ceph-aiops/`, token/runaway budget guard, undo-token recording, and descriptive risk tiers. The Dashboard password is stored **encrypted** (`~/.ceph-aiops/secrets.enc`, Fernet + scrypt) — never plaintext on disk. The flagship `cluster_health` turns raw HEALTH_WARN/ERR check codes into plain-language cause + suggested action.\n\n> **Standalone**: the governance harness is bundled in the package (`ceph_aiops.governance`) — ceph-aiops has no external skill-family dependency. Works against vanilla ceph-mgr (cephadm / hypervisor-bundled / MicroCeph); no croit, no Kubernetes.\n\n## What This Skill Does\n\n| Group | Tools | Count | Read or Write |\n|-------|-------|:-----:|:-------------:|\n| **Health** | cluster_health (flagship RCA), cluster_status | 2 | 2 read |\n| **OSD** | osd_tree, osd_df, osd_perf | 3 | 3 read |\n| | cluster_flag_set, osd_reweight, osd_mark_in, osd_mark_out, osd_purge | 5 | 5 write |\n| **PG** | pg_summary, pg_dump_stuck, scrub_status | 3 | 3 read |\n| | trigger_scrub, trigger_deep_scrub | 2 | 2 write |\n| **Pool** | pool_ls, pool_df | 2 | 2 read |\n| | set_pool_quota, set_pool_pg_num, set_pool_autoscale, pool_create, set_pool_size, pool_delete | 6 | 6 write |\n| **RBD** | rbd_ls | 1 | 1 read |\n| | rbd_image_create, rbd_snapshot_create, rbd_image_delete, rbd_snapshot_delete | 4 | 4 write |\n| **CephFS / RGW** | cephfs_status, rgw_status | 2 | 2 read |\n| **Cluster-ops** | mon_status, mgr_status, slow_ops, capacity_forecast | 4 | 4 read |\n| | throttle_recovery | 1 | 1 write |\n| **Undo** | undo_list, undo_apply | 2 | 2 undo |\n\nTotals: **37 tools — 17 read, 18 write, 2 undo.** The MCP server exposes all 37; the CLI is a convenience subset.\n\n## Quick Install\n\n```bash\nuv tool install ceph-aiops\nceph-aiops init       # interactive wizard: mgr host/port/username + encrypted Dashboard password\nceph-aiops doctor\n```\n\nOr as an OpenClaw plugin, which installs this skill and its MCP server together:\n\n```bash\nopenclaw plugins install clawhub:@zw008/ceph-aiops\nopenclaw skills info ceph-aiops          # expect: Visible to model: yes\n```\n\nNeeds `uvx` on `PATH`: the MCP server is fetched with uv, pinned to this release.\n\n## When to Use This Skill\n\n- Decode a **HEALTH_WARN/ERR** state (`cluster_health` / `health detail`) — cause + action per active check (`PG_DEGRADED`, `OSD_NEARFULL`, `SLOW_OPS`, `MON_DOWN`, `LARGE_OMAP_OBJECTS`, …)\n- One-shot triage (`overview`): HEALTH status + active checks + OSD up/in counts\n- Inspect OSDs (`osd_tree` / `osd_df` most-full first / `osd_perf` slowest first), PGs (`pg_summary` / `pg_dump_stuck` / `scrub_status`), pools (`pool_ls` / `pool_df` usable capacity)\n- Investigate slow requests (`slow_ops`), MDS trimming lag (`cephfs_status`), RGW large-omap (`rgw_status`), mon quorum (`mon_status`), and days-to-nearfull (`capacity_forecast`)\n- Safely **drain + purge** an OSD, change **pool size/quota**, or **throttle a slow rebalance** (governed writes with dry-run + undo)\n\n**Do NOT use when** the target is not Ceph (a hypervisor, another storage appliance, a backup product, a container cluster, or a network device). Route those to the appropriate **other AIops-tools** skill.\n\n## Related Skills — Skill Routing\n\n| If the user wants… | Use |\n|--------------------|-----|\n| Ceph: HEALTH_WARN RCA, OSD/PG/pool/RBD/CephFS/RGW, rebalance, slow ops | **ceph-aiops** (this skill) |\n| Any non-Ceph target (hypervisor, other storage, backup, cluster, network) | the appropriate **other AIops-tools** skill |\n\n## Common Workflows\n\n### 1. \"The cluster went HEALTH_WARN overnight\" — decode it (read-only)\n\n1. `ceph-aiops doctor` → confirm the mgr Dashboard is reachable and the JWT login works before trusting anything else\n2. `ceph-aiops overview` → HEALTH status, the list of active check codes, and OSD up/in counts in one shot\n3. `ceph-aiops health detail` (MCP: `cluster_health`) → each **active** check translated into what it means, the likely cause, and a suggested action\n4. Drill into the implicated resource: `PG_DEGRADED` → `pg_dump_stuck`; `OSD_NEARFULL` → `ceph-aiops osd df` (most-full first); `SLOW_OPS` → `slow_ops`; `MON_DOWN` → `mon_status`; `LARGE_OMAP_OBJECTS` → `rgw_status`\n5. **Failure branch**: if `doctor` fails on auth, the Dashboard password is wrong or the store is locked — re-run `ceph-aiops secret set <target>` (or export `CEPH_AIOPS_MASTER_PASSWORD` for non-interactive use). If `doctor` fails on reachability, the mgr `dashboard` module is likely not enabled; no read is issued against an unauthenticated session.\n\n### 2. Retire a failing OSD: drain, mark out, purge (governed)\n\n1. `ceph-aiops health detail` → confirm the OSD is genuinely the problem (e.g. `OSD_SLOW_PING_TIME`, repeated `SLOW_OPS` on one id) rather than a cluster-wide symptom\n2. `ceph-aiops osd df` → confirm the id, and that the remaining OSDs have room to absorb its data before you drain anything\n3. `ceph-aiops osd reweight <id> 0.0` → start a gradual drain; reversible, the prior CRUSH weight is captured as the undo descriptor\n4. `ceph-aiops osd out <id> --dry-run`, then re-run without `--dry-run` → **high** risk, double confirmation, needs `CEPH_AUDIT_APPROVED_BY`\n5. Wait for `ceph-aiops health status` / `pg_summary` to show all PGs `active+clean` — do not purge while backfill is running\n6. `ceph-aiops osd purge <id> --dry-run`, then re-run without `--dry-run` → **high**, irreversible\n7. **Failure branch**: if client I/O tanks during the drain, stop and reverse — `ceph-aiops undo list` then `ceph-aiops undo apply <id>` restores the prior weight (and `osd_mark_in` reverses the mark-out). Purge has no undo, which is exactly why it comes last and after `active+clean`.\n\n### 3. Recovery is starving client I/O\n\n1. `ceph-aiops health detail` → confirm the cluster is actually backfilling/recovering (`PG_DEGRADED`, `PG_BACKFILL_FULL`) rather than hitting a different bottleneck\n2. `slow_ops` → check whether client requests are genuinely being blocked, and by which OSDs\n3. `throttle_recovery(max_backfills=1, recovery_max_active=1)` → **med** risk, reversible; the prior `osd_max_backfills` / `osd_recovery_max_active` are captured as the undo descriptor\n4. Re-check `slow_ops` and `pg_summary` — recovery is slower but client latency should recover\n5. Once the cluster is quiet, raise the values back (or `ceph-aiops undo apply <id>` to restore the exact prior settings)\n6. **Failure branch**: if throttling does not help, the bottleneck is not recovery — go back to `osd_perf` (slowest OSDs first) and `mon_status`; do not keep lowering the throttle, you will only extend the degraded window.\n\n### 4. A pool is running out of usable capacity\n\n1. `ceph-aiops overview` → look for `POOL_NEARFULL` / `OSD_NEARFULL` among the active checks\n2. `pool_df` → per-pool usage with **usable capacity = raw ÷ size** (a `size=3` pool reports a third of raw — this is where most \"but the disks aren't full\" confusion comes from)\n3. `capacity_forecast` → days-to-nearfull at the current fill rate, so you know whether this is a this-week problem or a this-quarter one\n4. Buy time reversibly first: `set_pool_quota` (med, undo → prior quota) or `set_pool_autoscale` (med, undo) to let pg_num track the new size\n5. Only if a replica change is genuinely the answer: `set_pool_size --dry-run` then the real call — **high** risk, because lowering `size` reduces durability and any change forces cluster-wide data movement\n6. **Failure branch**: if the resulting rebalance saturates the cluster, apply workflow 3 (`throttle_recovery`) rather than reverting the size mid-flight; if the size change itself was wrong, `ceph-aiops undo apply <id>` replays the recorded prior value — but expect a second full rebalance.\n\n## Governance & Safety\n\nThe skill delivers reads and writes and records them; it does **not** decide\nwhether a write is permitted. That is your agent's judgement, or the permission\nof the account you connect it with (a ceph-mgr Dashboard account with a\nread-only role — writes then fail at the mgr). There is no read-only switch,\npolicy file, or approval gate.\n\n- **Audit is the guarantee, and it is not bypassable.** Every operation — MCP and CLI alike — is logged to `~/.ceph-aiops/audit.db` (relocatable via `CEPH_AIOPS_HOME`): params, result, status, duration, and the risk tier. The CLI writes the same row the MCP path does.\n- `CEPH_AUDIT_APPROVED_BY` / `CEPH_AUDIT_RATIONALE` are optional annotations recorded on the audit row (who/why); they are never required and never block.\n- **Runaway guard** — a safety backstop, not authorization: the same call looped in a tight window trips a circuit breaker. Disable with `CEPH_RUNAWAY_MAX=0`.\n- Destructive writes support `--dry-run` / `dry_run=True` and double confirmation at the CLI.\n- Reversible writes fetch the real before-state and record an inverse descriptor (`osd_reweight`→restore prior weight, `cluster_flag_set`→toggle back); irreversible ops (`osd_purge`, `pool_delete`, RBD deletes) record only the before-state.\n\n## References\n\n- `references/capabilities.md` — full tool → API-path → returns reference\n- `references/cli-reference.md` — CLI command reference\n- `references/setup-guide.md` — onboarding, credentials, and connectivity\n\nFile v0.11.4:_meta.json\n\n{\n  \"ownerId\": \"kn7b067awq2s97bn3d7p5qfhw5827pxc\",\n  \"slug\": \"ceph-aiops\",\n  \"version\": \"0.11.4\",\n  \"publishedAt\": 1789451407558\n}\n\nFile v0.11.4:references/agent-guardrails.md\n\n# Agent guardrails — running ceph-aiops with a smaller / local model\n\nIf you drive these tools with a local model (Llama, Qwen, Mistral … via Goose,\nOllama, LM Studio, or any OpenAI-compatible runtime), you will get noticeably\nbetter results with a short system prompt. This page gives you one, and — more\nimportantly — tells you which guardrails you **no longer need to write**, because\nthe tool now enforces them itself.\n\nThe distinction matters. A guardrail in a prompt is a request. A guardrail in the\nharness is a guarantee. Anything below that we could move into the harness, we did.\n\n## Authorization is not this tool's job — decide it where it belongs\n\nWhether a write should happen is your decision, or the account's. The tool does\nnot gate it — there is no read-only switch and no approval prompt to configure.\nThe two right places to control read vs write:\n\n- **The account you connect with.** Give it a ceph-mgr Dashboard account with a\n  read-only role. A write then fails at the mgr, which is the only place the\n  permission actually lives — no skill-side flag can be argued around by a\n  model, but a revoked permission cannot be.\n- **Your agent's system prompt.** If you want an observe-only session, tell the\n  model not to call the write tools (they are clearly tagged `[WRITE]`).\n\nWhat the tool *does* guarantee is that you can always see what happened:\n\n## What the tool enforces — do not waste prompt budget on these\n\n| You might be tempted to prompt | Why you don't need to |\n|---|---|\n| \"Log everything you do, over both MCP and the CLI\" | Every call is audited to `~/.ceph-aiops/audit.db` regardless of what the model says it did — and the CLI writes the same row the MCP path does, so there is no unaudited entry point. Reversible writes also record an undo token capturing the *prior* state. |\n| \"Don't invent a value when a field is missing\" | A field the Dashboard did not return comes back as `null`, never as `\"\"`. A missing `deviceClass`, `host`, MDS `state`, or `pg_autoscale_mode` is distinguishable from an empty one in the payload. |\n| \"Tell me if the output was cut off\" | `pg_summary` and `pg_dump_stuck` return `{\"stuck\": [...], \"returned\": N, \"limit\": L, \"truncated\": true/false}`. Truncation is measured (one extra row is collected), not guessed. `pg_summary` also keeps `unhealthyCount` as the true total even when the list is capped. |\n| \"Explain what HEALTH_WARN means\" | `cluster_health` already folds each active check code (`PG_DEGRADED`, `OSD_NEARFULL`, `SLOW_OPS`, `LARGE_OMAP_OBJECTS`, …) into a plain-language `cause` and `suggestedAction`. The model should quote those, not compose its own. |\n| \"Confirm before anything destructive\" | Destructive operations (`osd_purge`, `pool_delete`, `rbd_image_delete`, `rbd_snapshot_delete`, `set_pool_size`) require a `--dry-run`-able preview + double confirmation at the CLI. |\n| \"Don't get stuck retrying\" | The runaway guard trips a circuit breaker if the same call is hammered in a tight loop — a stuck agent is stopped rather than left to burn calls and time. |\n\n## What still needs a prompt\n\nThese are model-behaviour problems the harness cannot fix from the outside.\nCopy this into your agent's system prompt:\n\n```text\nYou operate a Ceph cluster through the ceph-aiops MCP tools, which talk to the\nceph-mgr Dashboard REST API.\n\nTOOL USE\n- Before answering any question about the current cluster, you MUST call a tool.\n  Never answer from memory or assumption.\n- Actually invoke the tool. Do not describe the call you would make, and do not\n  emit an example JSON response in place of calling it.\n- If a tool call fails, report the real error verbatim. Never fill the gap with\n  a plausible-sounding answer. A read that fails returns an \"error\" field rather\n  than raising — treat that as \"unknown\", not as \"healthy\".\n\nREADING RESULTS\n- Read the whole result before concluding. If a result contains a \"truncated\"\n  field that is true, say so and re-run with a higher limit instead of treating\n  the partial result as complete.\n- A null field means the Dashboard did not return that value. Report it as \"not\n  available\" — never infer it.\n- Report values exactly as returned. Do not normalise, translate, or prettify\n  status strings (HEALTH_WARN, active+undersized+degraded), PG ids, or OSD ids.\n- When cluster_health returns findings, quote each finding's \"cause\" and\n  \"suggestedAction\" rather than composing your own explanation of the check code.\n\nSCOPE\n- Separate observation from interpretation. State what the tools returned, then\n  any interpretation, clearly marked as such.\n- Do not assert a capacity, performance, or data-loss problem unless a tool\n  result supports it. HEALTH_WARN is not automatically an emergency —\n  PG_NOT_DEEP_SCRUBBED on a small cluster is routine.\n- Do not confuse the identifier kinds: an OSD id is a number (3), a PG id is\n  pool.hex (\"2.1a\"), a pool name is a string, and an RBD image is addressed as\n  pool/name. Never pass one where another is expected.\n- capacity_forecast is arithmetic extrapolation from a growth rate you supply.\n  With no growth rate it reports \"insufficient-data\" — do not present that as a\n  prediction.\n```\n\n## Recommended setup for a local model\n\nStart with a connection that *cannot* write, verify, and widen the account's\npermission only when you trust the setup — the destructive operations on a Ceph\ncluster are unusually cheap to invoke and unusually expensive to undo\n(`pool_delete` and `rbd_image_delete` destroy data no undo token can bring back):\n\n```bash\n# e.g. use a ceph-mgr Dashboard account with a read-only role. Then:\nceph-aiops doctor\n```\n\nOptionally annotate the audit trail with who is operating and why — recorded on\nevery row, never required:\n\n```bash\nexport CEPH_AUDIT_APPROVED_BY=\"your.name@example.com\"\nexport CEPH_AUDIT_RATIONALE=\"draining osd.7 for disk replacement 2026-07-20\"\n```\n\n## If your model still struggles\n\nSome behaviours are model-capacity limits rather than prompt problems:\n\n- **Multi-tool workflows time out or drift.** Prefer `cluster_health` and\n  `fleet_overview` — they do the multi-step correlation inside one call, so the\n  model does not have to chain reads and keep OSD/PG ids straight.\n- **The model ignores later tool results in a long context.** Ask narrower\n  questions and use `limit` deliberately rather than dumping every PG in the\n  cluster; `pg_summary`'s histogram is usually the right level of detail.\n- **The model describes calls instead of making them.** This is usually a\n  runtime/tool-calling-format mismatch, not a prompt problem — check that your\n  client advertises the tools in the format your model was trained on.\n\nFeedback on running this with a specific local model is genuinely useful —\nopen an issue at\n[github.com/AIops-tools/Ceph-AIops](https://github.com/AIops-tools/Ceph-AIops/issues)\nwith the model, runtime, and what went wrong.\n\nFile v0.11.4:references/capabilities.md\n\n# ceph-aiops capabilities\n\n> 37 MCP tools (17 read, 18 write, 2 undo) over the **ceph-mgr\n> Dashboard REST API** (`https://<host>:8443`, JWT via `POST /api/auth`).\n> Multi-node rebalance behaviour and the write ops need live verification\n> (see `docs/VERIFICATION.md`).\n\n## Read tools (17)\n\n| Tool | API path | Returns |\n|------|----------|---------|\n| `cluster_health` | `GET /api/health/full` | **flagship RCA** — per active HEALTH_WARN/ERR check: code, plain-language meaning, likely cause, suggested action |\n| `cluster_status` | `GET /api/health/minimal` | `ceph -s` summary: health status, mon/mgr/osd/pg counts |\n| `osd_tree` | `GET /api/osd` | OSD tree: up/in, CRUSH weight, host, device class |\n| `osd_df` | `GET /api/osd` | per-OSD utilization %, **most-full first**, near-full / backfill-full flags |\n| `osd_perf` | `GET /api/osd` | commit/apply latency per OSD, **slowest first** |\n| `pg_summary` | `GET /api/pg` (+ `/api/health/full`) | PG **state histogram** + list of non-active+clean PGs |\n| `pg_dump_stuck` | `GET /api/pg` | stuck PGs (inactive/unclean/stale/undersized) + implicated OSDs |\n| `scrub_status` | `GET /api/health/full` | PGs overdue for scrub / deep-scrub |\n| `pool_ls` | `GET /api/pool` | pools: name, id, size, pg_num, autoscale mode, application |\n| `pool_df` | `GET /api/pool` | per-pool usage; **usable capacity = raw ÷ size** |\n| `rbd_ls` | `GET /api/block/image` | RBD images (optionally filtered by pool): name, size, pool |\n| `cephfs_status` | `GET /api/cephfs` | MDS ranks + **\"behind on trimming\"** + client count |\n| `rgw_status` | `GET /api/rgw/daemon` + `GET /api/rgw/bucket` | RGW daemons + buckets + **LARGE_OMAP / unsharded-index** findings |\n| `mon_status` | `GET /api/monitor` | monitors: in-quorum vs **out-of-quorum** |\n| `mgr_status` | `GET /api/health/full` | active mgr, standbys, enabled modules |\n| `slow_ops` | `GET /api/health/full` | blocked / slow requests grouped **by OSD** |\n| `capacity_forecast` | `GET /api/osd` (+ df) | raw/used/avail + **days-to-nearfull** projection |\n\n## Write tools (18)\n\n| Tool | Risk | API path | Undo / safety |\n|------|------|----------|---------------|\n| `cluster_flag_set` | medium | `GET`+`PUT /api/osd/flags` | set/unset noout/noscrub/nobackfill/norecover; captures prior flag set (undo) |\n| `osd_reweight` | medium | `POST /api/osd/{id}/reweight` | 0.0 = drain; captures prior weight (undo) |\n| `osd_mark_in` | medium | `POST /api/osd/{id}/mark` | captures prior up/in state (undo) |\n| `osd_mark_out` | **high** | `POST /api/osd/{id}/mark` | drains data; CLI double-confirm + dry-run; captures prior state |\n| `osd_purge` | **high** | `DELETE /api/osd/{id}` | destroy + crush rm + auth del; **irreversible**; dry-run + double-confirm |\n| `trigger_scrub` | medium | `POST /api/pg/{pgid}/scrub` | schedule a shallow scrub; no prior state |\n| `trigger_deep_scrub` | medium | `POST /api/pg/{pgid}/deep_scrub` | schedule a deep (data-integrity) scrub |\n| `set_pool_quota` | medium | `PUT /api/pool/{name}` | captures prior max_bytes/max_objects (undo) |\n| `set_pool_pg_num` | medium | `PUT /api/pool/{name}` | captures prior pg_num (undo) |\n| `set_pool_autoscale` | medium | `PUT /api/pool/{name}` | captures prior autoscale mode (undo) |\n| `pool_create` | medium | `POST /api/pool` | create a new pool |\n| `set_pool_size` | **high** | `PUT /api/pool/{name}` | replica change **forces data movement**; dry-run + double-confirm; captures prior size |\n| `pool_delete` | **high** | `DELETE /api/pool/{name}` | **destroys all data**; dry-run + double-confirm |\n| `rbd_image_create` | medium | `POST /api/block/image` | create an RBD image |\n| `rbd_snapshot_create` | medium | `POST /api/block/image/{spec}/snap` | reversible → delete the snapshot |\n| `rbd_image_delete` | **high** | `DELETE /api/block/image/{spec}` | **irreversible**; dry-run + double-confirm |\n| `rbd_snapshot_delete` | **high** | `DELETE /api/block/image/{spec}/snap/{snap}` | **irreversible**; dry-run + double-confirm |\n| `throttle_recovery` | medium | `GET`+`POST /api/cluster_conf` | tunes `osd_max_backfills` / `osd_recovery_max_active`; captures prior values (undo) |\n\n## Out of scope (by design)\n\n- RGW **multisite** replication and zone/zonegroup management\n- **NFS-Ganesha** exports\n- **cephadm orchestrator** host/daemon management (add/remove hosts, deploy daemons)\n- Per-daemon config sprawl beyond the recovery-tuning keys above\n\nThe ceph-mgr Dashboard API has no ETag / pagination, so this tool exposes none.\n\nWant one of these? Open an issue or PR — feedback and contributions welcome.\n\nFile v0.11.4:references/cli-reference.md\n\n# ceph-aiops CLI reference\n\n> The CLI is a convenience subset; the full 37-tool surface\n> is via the MCP server (`ceph-aiops mcp`). Talks to the ceph-mgr Dashboard REST\n> API (`https://<host>:8443`, JWT via `POST /api/auth`).\n\n## Setup & diagnostics\n\n```bash\nceph-aiops init                      # interactive onboarding wizard\nceph-aiops doctor [--skip-auth]      # config + secret store + JWT login + mgr-dashboard reachability\nceph-aiops mcp                       # start the MCP server (stdio transport)\n```\n\n## Secrets (encrypted store ~/.ceph-aiops/secrets.enc)\n\n```bash\nceph-aiops secret set <target> [--value <password>]  # store Dashboard password (hidden prompt if no --value)\nceph-aiops secret list                               # names only — values never shown\nceph-aiops secret rm <target>\nceph-aiops secret migrate                            # import legacy plaintext .env (CEPH_<T>_PASSWORD)\nceph-aiops secret rotate-password                    # re-encrypt under a new master password\n```\n\n## Read commands\n\n```bash\nceph-aiops overview [--target <t>]        # HEALTH status + active checks + OSD up/in\nceph-aiops health detail                  # decode active HEALTH_WARN/ERR checks → cause + action (RCA)\nceph-aiops health status                  # ceph -s summary\nceph-aiops osd tree                        # OSD tree: up/in, weight, host, device class\nceph-aiops osd df                          # per-OSD utilization, most-full first, near/backfill-full flags\n```\n\n## Write commands (governed; risk tier in parentheses)\n\n```bash\nceph-aiops osd reweight <osd_id> <weight> [--dry-run]   # (med) 0.0 = drain; reversible → prior weight\nceph-aiops osd out <osd_id> [--dry-run]                 # (high) mark out — drains data; double confirm\nceph-aiops osd purge <osd_id> [--dry-run]               # (high) purge — irreversible; double confirm\n```\n\nThe remaining writes (cluster flags, pool quota/pg_num/autoscale/size/create/delete,\nRBD image/snapshot create/delete, trigger scrubs, throttle recovery) are exposed\nthrough the **MCP server**, not the CLI.\n\n## Common options\n\n- `--target, -t <name>` — target name from `config.yaml` (omit to use the default/first target)\n- `--dry-run` — print the API call that would be made, change nothing\n- Destructive commands (`osd out`, `osd purge`) require `--dry-run` review + double confirmation\n- `doctor --skip-auth` — skip the JWT login / connectivity check (config + secret-store checks only)\n\n## Approver env vars (high-risk ops)\n\n```bash\nexport CEPH_AUDIT_APPROVED_BY='you@example.com'\nexport CEPH_AUDIT_RATIONALE='draining failed OSD 7 per ticket OPS-123'\n```\n\nFile v0.11.4:references/setup-guide.md\n\n# ceph-aiops setup & security guide\n\n> The cheapest live check is a single-node MicroCeph running `ceph-aiops doctor`.\n> See `docs/VERIFICATION.md` for the full live-verification checklist.\n\n## 1. Install\n\n```bash\nuv tool install ceph-aiops\n```\n\n## 2. Enable the ceph-mgr Dashboard module\n\nceph-aiops talks to the **ceph-mgr Dashboard REST API** (HTTPS, default port\n`8443`). The mgr **dashboard** module must be enabled and a Dashboard user must\nexist:\n\n```bash\nceph mgr module enable dashboard\n# Start read-only. This tool does not decide whether a write is allowed — the\n# Dashboard role does — so the role you pick here IS the authorization boundary.\nceph dashboard ac-user-create <username> -i <password-file> read-only\n# find the URL/port: ceph mgr services   → e.g. https://<host>:8443/\n```\n\nGrant `administrator` only if you intend the agent to perform writes (set flags,\nreweight/mark-out/purge OSDs, scrub, pool and RBD create/delete), and prefer a\ndedicated account for it rather than reusing a human's.\n\n> **Which Dashboard role each endpoint needs has not been verified per endpoint\n> against a live cluster** — only that the role, not this tool, is what decides.\n> If a read is refused under `read-only`, that is Ceph's role boundary doing its\n> job: widen the role deliberately, or report the endpoint on the issue tracker\n> so this note can be replaced with a measured list.\n\nceph-aiops authenticates by exchanging the **username + password** for a\nshort-lived **JWT** at `POST /api/auth`; the token is cached in memory and used\nas a Bearer header for subsequent calls.\n\n## 3. Onboard\n\n```bash\nceph-aiops init\n```\n\nThe wizard collects (non-secret) connection details into\n`~/.ceph-aiops/config.yaml` and stores the Dashboard **password** encrypted into\n`~/.ceph-aiops/secrets.enc`. Example config:\n\n```yaml\ntargets:\n  - name: ceph1\n    host: 10.0.0.30\n    port: 8443\n    username: ceph-aiops       # the read-only Dashboard user created above\n    verify_ssl: true           # false only for self-signed lab certs\n```\n\nThe `username` lives in the config file (it is not a secret); the password never\ndoes.\n\n## 4. Non-interactive use (MCP server / CI / cron)\n\nExport the master password so the encrypted store can be unlocked without a\nprompt:\n\n```bash\nexport CEPH_AIOPS_MASTER_PASSWORD='your-master-password'\n```\n\n## Credential security\n\n- The Dashboard password is **never** written to disk in plaintext. It lives only\n  in `~/.ceph-aiops/secrets.enc`, encrypted with Fernet (AES-128-CBC + HMAC), the\n  key derived from your master password via scrypt. Only a per-store random salt\n  and the ciphertext are on disk (chmod 600); the master password itself is never\n  stored.\n- A legacy plaintext env var `CEPH_<TARGET_NAME_UPPER>_PASSWORD` is still honoured\n  as a fallback with a deprecation warning — migrate with\n  `ceph-aiops secret migrate` (it imports then renames the old `.env`).\n- The password is held only in memory, exchanged for a JWT at request time, and\n  is never logged or echoed; exception text and tracebacks are scrubbed of\n  secret-shaped strings before being written to the audit log.\n\n## Audit-annotation env vars (optional)\n\nThe skill does not decide whether a write is permitted — that is the agent's\njudgement or the connecting Dashboard account's role. If you want the audit trail\nto record *who* ran a destructive op and *why*, set these; they are recorded on\nthe row, never required, and gate nothing:\n\n```bash\nexport CEPH_AUDIT_APPROVED_BY='you@example.com'\nexport CEPH_AUDIT_RATIONALE='why this destructive op is justified'\n```\n\n## Governance harness state\n\nState lives under `~/.ceph-aiops/` (relocate with `CEPH_AIOPS_HOME`):\n\n- `audit.db` — every tool call (SQLite), with risk tier and any approver/rationale\n- `undo.db` — inverse descriptors for reversible writes (e.g. `osd_reweight`,\n  `set_pool_quota`, `throttle_recovery`)\n- budget / runaway guard — caps cumulative tool calls and wall-time; trips on\n  tight poll/retry loops\n\n## Self-test free with MicroCeph\n\nThe cheapest **live** path — a single-node cluster on one box:\n\n```bash\nsnap install microceph\nmicroceph cluster bootstrap\nmicroceph disk add loop,4G,3        # 3 loop-file OSDs\n# enable dashboard + create a user (see step 2), then:\nceph-aiops init\nceph-aiops doctor\n```\n\nA 3-node Vagrant cluster exercises real rebalance/backfill behaviour (draining an\nOSD, changing pool size) that a single node cannot.\n\n## Note: no ETag / pagination\n\nThe ceph-mgr Dashboard API offers neither ETag caching nor pagination, so\nceph-aiops exposes none — nothing is missing, the upstream API simply doesn't\nprovide them.\n\n## Verify\n\n```bash\nceph-aiops doctor\n```\n\n`doctor` checks the config file, the encrypted store and its permissions, that a\npassword is present per target, and (unless `--skip-auth`) connectivity by\nperforming the JWT login against the mgr Dashboard.\n\nFile v0.11.4:skill-card.md\n\n## Description:\n\nceph-aiops helps agents diagnose Ceph cluster health and operate Ceph resources through the ceph-mgr Dashboard REST API, including read-only RCA and governed write workflows for OSDs, PGs, pools, RBD, CephFS, RGW, monitors, managers, slow ops, and capacity forecasting.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[zw008](https://clawhub.ai/user/zw008)\n\n### License/Terms of Use:\n\nMIT\n\n## Use Case:\n\nDevelopers and storage operators use this skill to investigate Ceph HEALTH_WARN or HEALTH_ERR states, inspect cluster resources, and run governed Ceph maintenance tasks through an MCP server or CLI.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Write-capable MCP tools can make irreversible Ceph cluster or data changes without an enforced skill-side approval gate.\n\nMitigation: Start with a dedicated read-only Ceph Dashboard account; expose write-capable credentials only with external approval controls and dry-run review for destructive operations.\n\nRisk: The release installs an unpinned external package.\n\nMitigation: Pin or verify the ceph-aiops package version before use and install only from trusted package sources.\n\nRisk: Multi-node rebalance behavior and write operations are not fully live-verified in the artifact evidence.\n\nMitigation: Test write workflows in a non-production Ceph or MicroCeph environment before using them on production clusters.\n\n## Reference(s):\n\n- [ClawHub ceph-aiops Skill Page](https://clawhub.ai/zw008/skills/ceph-aiops)\n- [Ceph-AIops Homepage](https://github.com/AIops-tools/Ceph-AIops)\n- [Capabilities](references/capabilities.md)\n- [Setup and Security Guide](references/setup-guide.md)\n- [CLI Reference](references/cli-reference.md)\n- [Agent Guardrails](references/agent-guardrails.md)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, Shell commands, Configuration, Guidance]\n\n**Output Format:** [Markdown with inline shell commands and structured operational guidance]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May include Ceph Dashboard observations, risk tiers, dry-run advice, and follow-up commands.]\n\n## Skill Version(s):\n\n0.11.4 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v0.11.3: 7 files, 17829 bytes\n\nFiles: references/agent-guardrails.md (6913b), references/capabilities.md (4566b), references/cli-reference.md (2640b), references/setup-guide.md (4891b), skill-card.md (3043b), SKILL.md (14394b), _meta.json (130b)\n\nFile v0.11.3:SKILL.md\n\n---\nname: ceph-aiops\nslug: ceph-aiops\ndisplayName: \"Ceph AIops\"\nsummary: \"Governed Ceph mgr ops: HEALTH_WARN RCA, OSD/PG/pool/RBD/CephFS/RGW, 37 tools.\"\nlicense: MIT\nhomepage: https://github.com/AIops-tools/Ceph-AIops\ntags: [aiops, mcp, governance, ceph]\ndescription: >\n  Use this skill whenever the user needs to operate or diagnose a Ceph cluster via its ceph-mgr Dashboard REST API — decode a HEALTH_WARN/ERR state into cause + action (cluster_health), read the cluster status, inspect OSDs (tree/df/perf), placement groups (summary/stuck/scrub), pools (list/usable capacity), RBD images and snapshots, CephFS/MDS and RGW status, monitors/managers, slow ops and capacity forecast — plus governed writes (set cluster flags, reweight/mark-in/mark-out/purge OSDs, trigger scrubs, set pool quota/pg_num/autoscale/size, create/delete pools, create/delete RBD images and snapshots, throttle recovery/backfill).\n  Always use this skill for \"ceph health\", \"what does this HEALTH_WARN mean\", \"PG_DEGRADED / OSD_NEARFULL / SLOW_OPS / MON_DOWN\", \"ceph -s\", \"which OSD is most full\", \"drain an OSD\", \"purge an OSD\", \"stuck PGs\", \"overdue scrub\", \"pool usable capacity\", \"set pool size / quota\", \"rebalance is too slow / throttle backfill\", \"RBD image or snapshot\", \"MDS behind on trimming\", \"RGW large omap\", \"mon quorum\", or \"days to nearfull\" when the context is a Ceph cluster (cephadm, hypervisor-bundled Ceph, or MicroCeph).\n  Do NOT use when the target is not Ceph — a hypervisor, a different storage appliance, a backup product, a Kubernetes cluster, or a network device. Route those to the appropriate other AIops-tools skill (negative routing hint only).\n  Common Ceph ops with a built-in governance harness (audit, policy, token budget, undo, risk-tiers).\ninstaller:\n  kind: uv\n  package: ceph-aiops\nargument-hint: \"[ceph question or describe your cluster task]\"\nallowed-tools:\n  - Bash\nmetadata: {\"openclaw\":{\"requires\":{\"anyBins\":[\"ceph-aiops\",\"uvx\"]},\"optional\":{\"env\":[\"CEPH_AIOPS_CONFIG\",\"CEPH_AIOPS_MASTER_PASSWORD\"]},\"homepage\":\"https://github.com/AIops-tools/Ceph-AIops\",\"emoji\":\"🐙\",\"os\":[\"macos\",\"linux\"]}}\ncompatibility: >\n  Standalone, self-governed Ceph operations. The governance harness (audit, policy, token/runaway budget, undo, risk-tiers) is bundled in the package — no external skill-family dependency. Works against vanilla ceph-mgr (cephadm / hypervisor-bundled Ceph / MicroCeph); no croit and no Kubernetes dependency.\n  All write operations are audited to a local SQLite DB under ~/.ceph-aiops/ (relocatable via CEPH_AIOPS_HOME).\n  Connection: the ceph-mgr Dashboard REST API over HTTPS (default port 8443). Authentication is username + password exchanged for a short-lived JWT at POST /api/auth; the mgr 'dashboard' module must be enabled. The username lives in config.yaml; the password is stored ENCRYPTED in ~/.ceph-aiops/secrets.enc (Fernet/AES-128 + scrypt-derived key) — never plaintext on disk. Run 'ceph-aiops init' to onboard, or 'ceph-aiops secret set <target>' to add one. The store is unlocked by a master password from CEPH_AIOPS_MASTER_PASSWORD (non-interactive/MCP/CI) or an interactive prompt (CLI on a TTY). A legacy plaintext env var CEPH_<TARGET_NAME_UPPER>_PASSWORD is still honoured as a fallback with a deprecation warning (migrate with 'ceph-aiops secret migrate'). The password is held only in memory and exchanged for a JWT at request time; secrets are never logged or echoed.\n  State-changing operations require double confirmation at the CLI layer and support --dry-run. All write tools pass through the @governed_tool decorator (pre-check + budget guard + audit + risk-tier label). High-risk destructive ops (osd_mark_out, osd_purge, pool_delete, set_pool_size, rbd_image_delete, rbd_snapshot_delete) require dry-run + double confirmation; reversible writes (osd_reweight, cluster_flag_set, set_pool_quota/pg_num/autoscale, throttle_recovery) capture the prior state and record an inverse undo descriptor.\n  Webhooks: none — no outbound network calls beyond the configured ceph-mgr Dashboard REST API.\n  SSL: verify_ssl defaults to true; disable only for self-signed lab certificates.\n  Transitive dependencies: httpx (HTTP client) and the MCP SDK. No post-install scripts or background services.\n  Validation status: behaviour is exercised against mocked Dashboard responses; multi-node rebalance behaviour and the write ops have not been run against a live cluster (a single-node MicroCeph running 'ceph-aiops doctor' is the cheapest live path; see docs/VERIFICATION.md). The Dashboard API has no ETag/pagination, so none are exposed.\n---\n\n# Ceph AIops\n\n> **Disclaimer**: Community-maintained open-source project, **not affiliated with, endorsed by, or sponsored by the Ceph project or any storage vendor.** Product and trademark names belong to their owners. Source at [github.com/AIops-tools/Ceph-AIops](https://github.com/AIops-tools/Ceph-AIops) under the MIT license.\n\nGoverned Ceph operations via the **ceph-mgr Dashboard REST API** — **37 MCP tools**, every one wrapped with the bundled `@governed_tool` harness: a local unified audit log under `~/.ceph-aiops/`, token/runaway budget guard, undo-token recording, and descriptive risk tiers. The Dashboard password is stored **encrypted** (`~/.ceph-aiops/secrets.enc`, Fernet + scrypt) — never plaintext on disk. The flagship `cluster_health` turns raw HEALTH_WARN/ERR check codes into plain-language cause + suggested action.\n\n> **Standalone**: the governance harness is bundled in the package (`ceph_aiops.governance`) — ceph-aiops has no external skill-family dependency. Works against vanilla ceph-mgr (cephadm / hypervisor-bundled / MicroCeph); no croit, no Kubernetes.\n\n## What This Skill Does\n\n| Group | Tools | Count | Read or Write |\n|-------|-------|:-----:|:-------------:|\n| **Health** | cluster_health (flagship RCA), cluster_status | 2 | 2 read |\n| **OSD** | osd_tree, osd_df, osd_perf | 3 | 3 read |\n| | cluster_flag_set, osd_reweight, osd_mark_in, osd_mark_out, osd_purge | 5 | 5 write |\n| **PG** | pg_summary, pg_dump_stuck, scrub_status | 3 | 3 read |\n| | trigger_scrub, trigger_deep_scrub | 2 | 2 write |\n| **Pool** | pool_ls, pool_df | 2 | 2 read |\n| | set_pool_quota, set_pool_pg_num, set_pool_autoscale, pool_create, set_pool_size, pool_delete | 6 | 6 write |\n| **RBD** | rbd_ls | 1 | 1 read |\n| | rbd_image_create, rbd_snapshot_create, rbd_image_delete, rbd_snapshot_delete | 4 | 4 write |\n| **CephFS / RGW** | cephfs_status, rgw_status | 2 | 2 read |\n| **Cluster-ops** | mon_status, mgr_status, slow_ops, capacity_forecast | 4 | 4 read |\n| | throttle_recovery | 1 | 1 write |\n| **Undo** | undo_list, undo_apply | 2 | 2 undo |\n\nTotals: **37 tools — 17 read, 18 write, 2 undo.** The MCP server exposes all 37; the CLI is a convenience subset.\n\n## Quick Install\n\n```bash\nuv tool install ceph-aiops\nceph-aiops init       # interactive wizard: mgr host/port/username + encrypted Dashboard password\nceph-aiops doctor\n```\n\nOr as an OpenClaw plugin, which installs this skill and its MCP server together:\n\n```bash\nopenclaw plugins install clawhub:@zw008/ceph-aiops\nopenclaw skills info ceph-aiops          # expect: Visible to model: yes\n```\n\nNeeds `uvx` on `PATH`: the MCP server is fetched with uv, pinned to this release.\n\n## When to Use This Skill\n\n- Decode a **HEALTH_WARN/ERR** state (`cluster_health` / `health detail`) — cause + action per active check (`PG_DEGRADED`, `OSD_NEARFULL`, `SLOW_OPS`, `MON_DOWN`, `LARGE_OMAP_OBJECTS`, …)\n- One-shot triage (`overview`): HEALTH status + active checks + OSD up/in counts\n- Inspect OSDs (`osd_tree` / `osd_df` most-full first / `osd_perf` slowest first), PGs (`pg_summary` / `pg_dump_stuck` / `scrub_status`), pools (`pool_ls` / `pool_df` usable capacity)\n- Investigate slow requests (`slow_ops`), MDS trimming lag (`cephfs_status`), RGW large-omap (`rgw_status`), mon quorum (`mon_status`), and days-to-nearfull (`capacity_forecast`)\n- Safely **drain + purge** an OSD, change **pool size/quota**, or **throttle a slow rebalance** (governed writes with dry-run + undo)\n\n**Do NOT use when** the target is not Ceph (a hypervisor, another storage appliance, a backup product, a container cluster, or a network device). Route those to the appropriate **other AIops-tools** skill.\n\n## Related Skills — Skill Routing\n\n| If the user wants… | Use |\n|--------------------|-----|\n| Ceph: HEALTH_WARN RCA, OSD/PG/pool/RBD/CephFS/RGW, rebalance, slow ops | **ceph-aiops** (this skill) |\n| Any non-Ceph target (hypervisor, other storage, backup, cluster, network) | the appropriate **other AIops-tools** skill |\n\n## Common Workflows\n\n### 1. \"The cluster went HEALTH_WARN overnight\" — decode it (read-only)\n\n1. `ceph-aiops doctor` → confirm the mgr Dashboard is reachable and the JWT login works before trusting anything else\n2. `ceph-aiops overview` → HEALTH status, the list of active check codes, and OSD up/in counts in one shot\n3. `ceph-aiops health detail` (MCP: `cluster_health`) → each **active** check translated into what it means, the likely cause, and a suggested action\n4. Drill into the implicated resource: `PG_DEGRADED` → `pg_dump_stuck`; `OSD_NEARFULL` → `ceph-aiops osd df` (most-full first); `SLOW_OPS` → `slow_ops`; `MON_DOWN` → `mon_status`; `LARGE_OMAP_OBJECTS` → `rgw_status`\n5. **Failure branch**: if `doctor` fails on auth, the Dashboard password is wrong or the store is locked — re-run `ceph-aiops secret set <target>` (or export `CEPH_AIOPS_MASTER_PASSWORD` for non-interactive use). If `doctor` fails on reachability, the mgr `dashboard` module is likely not enabled; no read is issued against an unauthenticated session.\n\n### 2. Retire a failing OSD: drain, mark out, purge (governed)\n\n1. `ceph-aiops health detail` → confirm the OSD is genuinely the problem (e.g. `OSD_SLOW_PING_TIME`, repeated `SLOW_OPS` on one id) rather than a cluster-wide symptom\n2. `ceph-aiops osd df` → confirm the id, and that the remaining OSDs have room to absorb its data before you drain anything\n3. `ceph-aiops osd reweight <id> 0.0` → start a gradual drain; reversible, the prior CRUSH weight is captured as the undo descriptor\n4. `ceph-aiops osd out <id> --dry-run`, then re-run without `--dry-run` → **high** risk, double confirmation, needs `CEPH_AUDIT_APPROVED_BY`\n5. Wait for `ceph-aiops health status` / `pg_summary` to show all PGs `active+clean` — do not purge while backfill is running\n6. `ceph-aiops osd purge <id> --dry-run`, then re-run without `--dry-run` → **high**, irreversible\n7. **Failure branch**: if client I/O tanks during the drain, stop and reverse — `ceph-aiops undo list` then `ceph-aiops undo apply <id>` restores the prior weight (and `osd_mark_in` reverses the mark-out). Purge has no undo, which is exactly why it comes last and after `active+clean`.\n\n### 3. Recovery is starving client I/O\n\n1. `ceph-aiops health detail` → confirm the cluster is actually backfilling/recovering (`PG_DEGRADED`, `PG_BACKFILL_FULL`) rather than hitting a different bottleneck\n2. `slow_ops` → check whether client requests are genuinely being blocked, and by which OSDs\n3. `throttle_recovery(max_backfills=1, recovery_max_active=1)` → **med** risk, reversible; the prior `osd_max_backfills` / `osd_recovery_max_active` are captured as the undo descriptor\n4. Re-check `slow_ops` and `pg_summary` — recovery is slower but client latency should recover\n5. Once the cluster is quiet, raise the values back (or `ceph-aiops undo apply <id>` to restore the exact prior settings)\n6. **Failure branch**: if throttling does not help, the bottleneck is not recovery — go back to `osd_perf` (slowest OSDs first) and `mon_status`; do not keep lowering the throttle, you will only extend the degraded window.\n\n### 4. A pool is running out of usable capacity\n\n1. `ceph-aiops overview` → look for `POOL_NEARFULL` / `OSD_NEARFULL` among the active checks\n2. `pool_df` → per-pool usage with **usable capacity = raw ÷ size** (a `size=3` pool reports a third of raw — this is where most \"but the disks aren't full\" confusion comes from)\n3. `capacity_forecast` → days-to-nearfull at the current fill rate, so you know whether this is a this-week problem or a this-quarter one\n4. Buy time reversibly first: `set_pool_quota` (med, undo → prior quota) or `set_pool_autoscale` (med, undo) to let pg_num track the new size\n5. Only if a replica change is genuinely the answer: `set_pool_size --dry-run` then the real call — **high** risk, because lowering `size` reduces durability and any change forces cluster-wide data movement\n6. **Failure branch**: if the resulting rebalance saturates the cluster, apply workflow 3 (`throttle_recovery`) rather than reverting the size mid-flight; if the size change itself was wrong, `ceph-aiops undo apply <id>` replays the recorded prior value — but expect a second full rebalance.\n\n## Governance & Safety\n\nThe skill delivers reads and writes and records them; it does **not** decide\nwhether a write is permitted. That is your agent's judgement, or the permission\nof the account you connect it with (a ceph-mgr Dashboard account with a\nread-only role — writes then fail at the mgr). There is no read-only switch,\npolicy file, or approval gate.\n\n- **Audit is the guarantee, and it is not bypassable.** Every operation — MCP and CLI alike — is logged to `~/.ceph-aiops/audit.db` (relocatable via `CEPH_AIOPS_HOME`): params, result, status, duration, and the risk tier. The CLI writes the same row the MCP path does.\n- `CEPH_AUDIT_APPROVED_BY` / `CEPH_AUDIT_RATIONALE` are optional annotations recorded on the audit row (who/why); they are never required and never block.\n- **Runaway guard** — a safety backstop, not authorization: the same call looped in a tight window trips a circuit breaker. Disable with `CEPH_RUNAWAY_MAX=0`.\n- Destructive writes support `--dry-run` / `dry_run=True` and double confirmation at the CLI.\n- Reversible writes fetch the real before-state and record an inverse descriptor (`osd_reweight`→restore prior weight, `cluster_flag_set`→toggle back); irreversible ops (`osd_purge`, `pool_delete`, RBD deletes) record only the before-state.\n\n## References\n\n- `references/capabilities.md` — full tool → API-path → returns reference\n- `references/cli-reference.md` — CLI command reference\n- `references/setup-guide.md` — onboarding, credentials, and connectivity\n\nFile v0.11.3:_meta.json\n\n{\n  \"ownerId\": \"kn7b067awq2s97bn3d7p5qfhw5827pxc\",\n  \"slug\": \"ceph-aiops\",\n  \"version\": \"0.11.3\",\n  \"publishedAt\": 1789256315627\n}\n\nFile v0.11.3:references/agent-guardrails.md\n\n# Agent guardrails — running ceph-aiops with a smaller / local model\n\nIf you drive these tools with a local model (Llama, Qwen, Mistral … via Goose,\nOllama, LM Studio, or any OpenAI-compatible runtime), you will get noticeably\nbetter results with a short system prompt. This page gives you one, and — more\nimportantly — tells you which guardrails you **no longer need to write**, because\nthe tool now enforces them itself.\n\nThe distinction matters. A guardrail in a prompt is a request. A guardrail in the\nharness is a guarantee. Anything below that we could move into the harness, we did.\n\n## Authorization is not this tool's job — decide it where it belongs\n\nWhether a write should happen is your decision, or the account's. The tool does\nnot gate it — there is no read-only switch and no approval prompt to configure.\nThe two right places to control read vs write:\n\n- **The account you connect with.** Give it a ceph-mgr Dashboard account with a\n  read-only role. A write then fails at the mgr, which is the only place the\n  permission actually lives — no skill-side flag can be argued around by a\n  model, but a revoked permission cannot be.\n- **Your agent's system prompt.** If you want an observe-only session, tell the\n  model not to call the write tools (they are clearly tagged `[WRITE]`).\n\nWhat the tool *does* guarantee is that you can always see what happened:\n\n## What the tool enforces — do not waste prompt budget on these\n\n| You might be tempted to prompt | Why you don't need to |\n|---|---|\n| \"Log everything you do, over both MCP and the CLI\" | Every call is audited to `~/.ceph-aiops/audit.db` regardless of what the model says it did — and the CLI writes the same row the MCP path does, so there is no unaudited entry point. Reversible writes also record an undo token capturing the *prior* state. |\n| \"Don't invent a value when a field is missing\" | A field the Dashboard did not return comes back as `null`, never as `\"\"`. A missing `deviceClass`, `host`, MDS `state`, or `pg_autoscale_mode` is distinguishable from an empty one in the payload. |\n| \"Tell me if the output was cut off\" | `pg_summary` and `pg_dump_stuck` return `{\"stuck\": [...], \"returned\": N, \"limit\": L, \"truncated\": true/false}`. Truncation is measured (one extra row is collected), not guessed. `pg_summary` also keeps `unhealthyCount` as the true total even when the list is capped. |\n| \"Explain what HEALTH_WARN means\" | `cluster_health` already folds each active check code (`PG_DEGRADED`, `OSD_NEARFULL`, `SLOW_OPS`, `LARGE_OMAP_OBJECTS`, …) into a plain-language `cause` and `suggestedAction`. The model should quote those, not compose its own. |\n| \"Confirm before anything destructive\" | Destructive operations (`osd_purge`, `pool_delete`, `rbd_image_delete`, `rbd_snapshot_delete`, `set_pool_size`) require a `--dry-run`-able preview + double confirmation at the CLI. |\n| \"Don't get stuck retrying\" | The runaway guard trips a circuit breaker if the same call is hammered in a tight loop — a stuck agent is stopped rather than left to burn calls and time. |\n\n## What still needs a prompt\n\nThese are model-behaviour problems the harness cannot fix from the outside.\nCopy this into your agent's system prompt:\n\n```text\nYou operate a Ceph cluster through the ceph-aiops MCP tools, which talk to the\nceph-mgr Dashboard REST API.\n\nTOOL USE\n- Before answering any question about the current cluster, you MUST call a tool.\n  Never answer from memory or assumption.\n- Actually invoke the tool. Do not describe the call you would make, and do not\n  emit an example JSON response in place of calling it.\n- If a tool call fails, report the real error verbatim. Never fill the gap with\n  a plausible-sounding answer. A read that fails returns an \"error\" field rather\n  than raising — treat that as \"unknown\", not as \"healthy\".\n\nREADING RESULTS\n- Read the whole result before concluding. If a result contains a \"truncated\"\n  field that is true, say so and re-run with a higher limit instead of treating\n  the partial result as complete.\n- A null field means the Dashboard did not return that value. Report it as \"not\n  available\" — never infer it.\n- Report values exactly as returned. Do not normalise, translate, or prettify\n  status strings (HEALTH_WARN, active+undersized+degraded), PG ids, or OSD ids.\n- When cluster_health returns findings, quote each finding's \"cause\" and\n  \"suggestedAction\" rather than composing your own explanation of the check code.\n\nSCOPE\n- Separate observation from interpretation. State what the tools returned, then\n  any interpretation, clearly marked as such.\n- Do not assert a capacity, performance, or data-loss problem unless a tool\n  result supports it. HEALTH_WARN is not automatically an emergency —\n  PG_NOT_DEEP_SCRUBBED on a small cluster is routine.\n- Do not confuse the identifier kinds: an OSD id is a number (3), a PG id is\n  pool.hex (\"2.1a\"), a pool name is a string, and an RBD image is addressed as\n  pool/name. Never pass one where another is expected.\n- capacity_forecast is arithmetic extrapolation from a growth rate you supply.\n  With no growth rate it reports \"insufficient-data\" — do not present that as a\n  prediction.\n```\n\n## Recommended setup for a local model\n\nStart with a connection that *cannot* write, verify, and widen the account's\npermission only when you trust the setup — the destructive operations on a Ceph\ncluster are unusually cheap to invoke and unusually expensive to undo\n(`pool_delete` and `rbd_image_delete` destroy data no undo token can bring back):\n\n```bash\n# e.g. use a ceph-mgr Dashboard account with a read-only role. Then:\nceph-aiops doctor\n```\n\nOptionally annotate the audit trail with who is operating and why — recorded on\nevery row, never required:\n\n```bash\nexport CEPH_AUDIT_APPROVED_BY=\"your.name@example.com\"\nexport CEPH_AUDIT_RATIONALE=\"draining osd.7 for disk replacement 2026-07-20\"\n```\n\n## If your model still struggles\n\nSome behaviours are model-capacity limits rather than prompt problems:\n\n- **Multi-tool workflows time out or drift.** Prefer `cluster_health` and\n  `fleet_overview` — they do the multi-step correlation inside one call, so the\n  model does not have to chain reads and keep OSD/PG ids straight.\n- **The model ignores later tool results in a long context.** Ask narrower\n  questions and use `limit` deliberately rather than dumping every PG in the\n  cluster; `pg_summary`'s histogram is usually the right level of detail.\n- **The model describes calls instead of making them.** This is usually a\n  runtime/tool-calling-format mismatch, not a prompt problem — check that your\n  client advertises the tools in the format your model was trained on.\n\nFeedback on running this with a specific local model is genuinely useful —\nopen an issue at\n[github.com/AIops-tools/Ceph-AIops](https://github.com/AIops-tools/Ceph-AIops/issues)\nwith the model, runtime, and what went wrong.\n\nFile v0.11.3:references/capabilities.md\n\n# ceph-aiops capabilities\n\n> 37 MCP tools (17 read, 18 write, 2 undo) over the **ceph-mgr\n> Dashboard REST API** (`https://<host>:8443`, JWT via `POST /api/auth`).\n> Multi-node rebalance behaviour and the write ops need live verification\n> (see `docs/VERIFICATION.md`).\n\n## Read tools (17)\n\n| Tool | API path | Returns |\n|------|----------|---------|\n| `cluster_health` | `GET /api/health/full` | **flagship RCA** — per active HEALTH_WARN/ERR check: code, plain-language meaning, likely cause, suggested action |\n| `cluster_status` | `GET /api/health/minimal` | `ceph -s` summary: health status, mon/mgr/osd/pg counts |\n| `osd_tree` | `GET /api/osd` | OSD tree: up/in, CRUSH weight, host, device class |\n| `osd_df` | `GET /api/osd` | per-OSD utilization %, **most-full first**, near-full / backfill-full flags |\n| `osd_perf` | `GET /api/osd` | commit/apply latency per OSD, **slowest first** |\n| `pg_summary` | `GET /api/pg` (+ `/api/health/full`) | PG **state histogram** + list of non-active+clean PGs |\n| `pg_dump_stuck` | `GET /api/pg` | stuck PGs (inactive/unclean/stale/undersized) + implicated OSDs |\n| `scrub_status` | `GET /api/health/full` | PGs overdue for scrub / deep-scrub |\n| `pool_ls` | `GET /api/pool` | pools: name, id, size, pg_num, autoscale mode, application |\n| `pool_df` | `GET /api/pool` | per-pool usage; **usable capacity = raw ÷ size** |\n| `rbd_ls` | `GET /api/block/image` | RBD images (optionally filtered by pool): name, size, pool |\n| `cephfs_status` | `GET /api/cephfs` | MDS ranks + **\"behind on trimming\"** + client count |\n| `rgw_status` | `GET /api/rgw/daemon` + `GET /api/rgw/bucket` | RGW daemons + buckets + **LARGE_OMAP / unsharded-index** findings |\n| `mon_status` | `GET /api/monitor` | monitors: in-quorum vs **out-of-quorum** |\n| `mgr_status` | `GET /api/health/full` | active mgr, standbys, enabled modules |\n| `slow_ops` | `GET /api/health/full` | blocked / slow requests grouped **by OSD** |\n| `capacity_forecast` | `GET /api/osd` (+ df) | raw/used/avail + **days-to-nearfull** projection |\n\n## Write tools (18)\n\n| Tool | Risk | API path | Undo / safety |\n|------|------|----------|---------------|\n| `cluster_flag_set` | medium | `GET`+`PUT /api/osd/flags` | set/unset noout/noscrub/nobackfill/norecover; captures prior flag set (undo) |\n| `osd_reweight` | medium | `POST /api/osd/{id}/reweight` | 0.0 = drain; captures prior weight (undo) |\n| `osd_mark_in` | medium | `POST /api/osd/{id}/mark` | captures prior up/in state (undo) |\n| `osd_mark_out` | **high** | `POST /api/osd/{id}/mark` | drains data; CLI double-confirm + dry-run; captures prior state |\n| `osd_purge` | **high** | `DELETE /api/osd/{id}` | destroy + crush rm + auth del; **irreversible**; dry-run + double-confirm |\n| `trigger_scrub` | medium | `POST /api/pg/{pgid}/scrub` | schedule a shallow scrub; no prior state |\n| `trigger_deep_scrub` | medium | `POST /api/pg/{pgid}/deep_scrub` | schedule a deep (data-integrity) scrub |\n| `set_pool_quota` | medium | `PUT /api/pool/{name}` | captures prior max_bytes/max_objects (undo) |\n| `set_pool_pg_num` | medium | `PUT /api/pool/{name}` | captures prior pg_num (undo) |\n| `set_pool_autoscale` | medium | `PUT /api/pool/{name}` | captures prior autoscale mode (undo) |\n| `pool_create` | medium | `POST /api/pool` | create a new pool |\n| `set_pool_size` | **high** | `PUT /api/pool/{name}` | replica change **forces data movement**; dry-run + double-confirm; captures prior size |\n| `pool_delete` | **high** | `DELETE /api/pool/{name}` | **destroys all data**; dry-run + double-confirm |\n| `rbd_image_create` | medium | `POST /api/block/image` | create an RBD image |\n| `rbd_snapshot_create` | medium | `POST /api/block/image/{spec}/snap` | reversible → delete the snapshot |\n| `rbd_image_delete` | **high** | `DELETE /api/block/image/{spec}` | **irreversible**; dry-run + double-confirm |\n| `rbd_snapshot_delete` | **high** | `DELETE /api/block/image/{spec}/snap/{snap}` | **irreversible**; dry-run + double-confirm |\n| `throttle_recovery` | medium | `GET`+`POST /api/cluster_conf` | tunes `osd_max_backfills` / `osd_recovery_max_active`; captures prior values (undo) |\n\n## Out of scope (by design)\n\n- RGW **multisite** replication and zone/zonegroup management\n- **NFS-Ganesha** exports\n- **cephadm orchestrator** host/daemon management (add/remove hosts, deploy daemons)\n- Per-daemon config sprawl beyond the recovery-tuning keys above\n\nThe ceph-mgr Dashboard API has no ETag / pagination, so this tool exposes none.\n\nWant one of these? Open an issue or PR — feedback and contributions welcome.\n\nFile v0.11.3:references/cli-reference.md\n\n# ceph-aiops CLI reference\n\n> The CLI is a convenience subset; the full 37-tool surface\n> is via the MCP server (`ceph-aiops mcp`). Talks to the ceph-mgr Dashboard REST\n> API (`https://<host>:8443`, JWT via `POST /api/auth`).\n\n## Setup & diagnostics\n\n```bash\nceph-aiops init                      # interactive onboarding wizard\nceph-aiops doctor [--skip-auth]      # config + secret store + JWT login + mgr-dashboard reachability\nceph-aiops mcp                       # start the MCP server (stdio transport)\n```\n\n## Secrets (encrypted store ~/.ceph-aiops/secrets.enc)\n\n```bash\nceph-aiops secret set <target> [--value <password>]  # store Dashboard password (hidden prompt if no --value)\nceph-aiops secret list                               # names only — values never shown\nceph-aiops secret rm <target>\nceph-aiops secret migrate                            # import legacy plaintext .env (CEPH_<T>_PASSWORD)\nceph-aiops secret rotate-password                    # re-encrypt under a new master password\n```\n\n## Read commands\n\n```bash\nceph-aiops overview [--target <t>]        # HEALTH status + active checks + OSD up/in\nceph-aiops health detail                  # decode active HEALTH_WARN/ERR checks → cause + action (RCA)\nceph-aiops health status                  # ceph -s summary\nceph-aiops osd tree                        # OSD tree: up/in, weight, host, device class\nceph-aiops osd df                          # per-OSD utilization, most-full first, near/backfill-full flags\n```\n\n## Write commands (governed; risk tier in parentheses)\n\n```bash\nceph-aiops osd reweight <osd_id> <weight> [--dry-run]   # (med) 0.0 = drain; reversible → prior weight\nceph-aiops osd out <osd_id> [--dry-run]                 # (high) mark out — drains data; double confirm\nceph-aiops osd purge <osd_id> [--dry-run]               # (high) purge — irreversible; double confirm\n```\n\nThe remaining writes (cluster flags, pool quota/pg_num/autoscale/size/create/delete,\nRBD image/snapshot create/delete, trigger scrubs, throttle recovery) are exposed\nthrough the **MCP server**, not the CLI.\n\n## Common options\n\n- `--target, -t <name>` — target name from `config.yaml` (omit to use the default/first target)\n- `--dry-run` — print the API call that would be made, change nothing\n- Destructive commands (`osd out`, `osd purge`) require `--dry-run` review + double confirmation\n- `doctor --skip-auth` — skip the JWT login / connectivity check (config + secret-store checks only)\n\n## Approver env vars (high-risk ops)\n\n```bash\nexport CEPH_AUDIT_APPROVED_BY='you@example.com'\nexport CEPH_AUDIT_RATIONALE='draining failed OSD 7 per ticket OPS-123'\n```\n\nFile v0.11.3:references/setup-guide.md\n\n# ceph-aiops setup & security guide\n\n> The cheapest live check is a single-node MicroCeph running `ceph-aiops doctor`.\n> See `docs/VERIFICATION.md` for the full live-verification checklist.\n\n## 1. Install\n\n```bash\nuv tool install ceph-aiops\n```\n\n## 2. Enable the ceph-mgr Dashboard module\n\nceph-aiops talks to the **ceph-mgr Dashboard REST API** (HTTPS, default port\n`8443`). The mgr **dashboard** module must be enabled and a Dashboard user must\nexist:\n\n```bash\nceph mgr module enable dashboard\n# Start read-only. This tool does not decide whether a write is allowed — the\n# Dashboard role does — so the role you pick here IS the authorization boundary.\nceph dashboard ac-user-create <username> -i <password-file> read-only\n# find the URL/port: ceph mgr services   → e.g. https://<host>:8443/\n```\n\nGrant `administrator` only if you intend the agent to perform writes (set flags,\nreweight/mark-out/purge OSDs, scrub, pool and RBD create/delete), and prefer a\ndedicated account for it rather than reusing a human's.\n\n> **Which Dashboard role each endpoint needs has not been verified per endpoint\n> against a live cluster** — only that the role, not this tool, is what decides.\n> If a read is refused under `read-only`, that is Ceph's role boundary doing its\n> job: widen the role deliberately, or report the endpoint on the issue tracker\n> so this note can be replaced with a measured list.\n\nceph-aiops authenticates by exchanging the **username + password** for a\nshort-lived **JWT** at `POST /api/auth`; the token is cached in memory and used\nas a Bearer header for subsequent calls.\n\n## 3. Onboard\n\n```bash\nceph-aiops init\n```\n\nThe wizard collects (non-secret) connection details into\n`~/.ceph-aiops/config.yaml` and stores the Dashboard **password** encrypted into\n`~/.ceph-aiops/secrets.enc`. Example config:\n\n```yaml\ntargets:\n  - name: ceph1\n    host: 10.0.0.30\n    port: 8443\n    username: ceph-aiops       # the read-only Dashboard user created above\n    verify_ssl: true           # false only for self-signed lab certs\n```\n\nThe `username` lives in the config file (it is not a secret); the password never\ndoes.\n\n## 4. Non-interactive use (MCP server / CI / cron)\n\nExport the master password so the encrypted store can be unlocked without a\nprompt:\n\n```bash\nexport CEPH_AIOPS_MASTER_PASSWORD='your-master-password'\n```\n\n## Credential security\n\n- The Dashboard password is **never** written to disk in plaintext. It lives only\n  in `~/.ceph-aiops/secrets.enc`, encrypted with Fernet (AES-128-CBC + HMAC), the\n  key derived from your master password via scrypt. Only a per-store random salt\n  and the ciphertext are on disk (chmod 600); the master password itself is never\n  stored.\n- A legacy plaintext env var `CEPH_<TARGET_NAME_UPPER>_PASSWORD` is still honoured\n  as a fallback with a deprecation warning — migrate with\n  `ceph-aiops secret migrate` (it imports then renames the old `.env`).\n- The password is held only in memory, exchanged for a JWT at request time, and\n  is never logged or echoed; exception text and tracebacks are scrubbed of\n  secret-shaped strings before being written to the audit log.\n\n## Audit-annotation env vars (optional)\n\nThe skill does not decide whether a write is permitted — that is the agent's\njudgement or the connecting Dashboard account's role. If you want the audit trail\nto record *who* ran a destructive op and *why*, set these; they are recorded on\nthe row, never required, and gate nothing:\n\n```bash\nexport CEPH_AUDIT_APPROVED_BY='you@example.com'\nexport CEPH_AUDIT_RATIONALE='why this destructive op is justified'\n```\n\n## Governance harness state\n\nState lives under `~/.ceph-aiops/` (relocate with `CEPH_AIOPS_HOME`):\n\n- `audit.db` — every tool call (SQLite), with risk tier and any approver/rationale\n- `undo.db` — inverse descriptors for reversible writes (e.g. `osd_reweight`,\n  `set_pool_quota`, `throttle_recovery`)\n- budget / runaway guard — caps cumulative tool calls and wall-time; trips on\n  tight poll/retry loops\n\n## Self-test free with MicroCeph\n\nThe cheapest **live** path — a single-node cluster on one box:\n\n```bash\nsnap install microceph\nmicroceph cluster bootstrap\nmicroceph disk add loop,4G,3        # 3 loop-file OSDs\n# enable dashboard + create a user (see step 2), then:\nceph-aiops init\nceph-aiops doctor\n```\n\nA 3-node Vagrant cluster exercises real rebalance/backfill behaviour (draining an\nOSD, changing pool size) that a single node cannot.\n\n## Note: no ETag / pagination\n\nThe ceph-mgr Dashboard API offers neither ETag caching nor pagination, so\nceph-aiops exposes none — nothing is missing, the upstream API simply doesn't\nprovide them.\n\n## Verify\n\n```bash\nceph-aiops doctor\n```\n\n`doctor` checks the config file, the encrypted store and its permissions, that a\npassword is present per target, and (unless `--skip-auth`) connectivity by\nperforming the JWT login against the mgr Dashboard.\n\nFile v0.11.3:skill-card.md\n\n## Description:\n\nCeph AIops helps agents diagnose and operate Ceph clusters through the ceph-mgr Dashboard REST API, including health analysis, OSD, PG, pool, RBD, CephFS, RGW inspection, and governed write operations.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[zw008](https://clawhub.ai/user/zw008)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers, storage operators, and SREs use this skill to triage Ceph health warnings, inspect cluster storage state, and plan operational actions. It can also invoke governed write workflows for cluster flags, OSD lifecycle, pool settings, RBD objects, scrubs, and recovery throttling when the connected Dashboard account permits them.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill installs and runs an external ceph-aiops executable that can access Ceph credentials and operate a live cluster.\n\nMitigation: Verify the package source and version out of band before installation, and start with a dedicated read-only ceph-mgr Dashboard account.\n\nRisk: Write tools can make destructive or irreversible Ceph changes, including OSD purge, pool deletion, and RBD deletion when the connected account has permission.\n\nMitigation: Grant administrator rights only for planned write sessions, require dry-run review where available, and confirm cluster state before executing high-risk operations.\n\nRisk: Credentials can be exposed if passwords are passed directly on command lines or reused from broad-privilege human accounts.\n\nMitigation: Use the encrypted secret store or non-interactive master password flow, avoid passing passwords with --value, and use a dedicated least-privilege Dashboard account.\n\nRisk: Some write and multi-node rebalance behavior has not been verified against a live cluster in the artifact evidence.\n\nMitigation: Run ceph-aiops doctor and validate intended workflows in a MicroCeph or staging cluster before using them on production data.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/zw008/skills/ceph-aiops)\n- [Project homepage](https://github.com/AIops-tools/Ceph-AIops)\n- [Capabilities reference](references/capabilities.md)\n- [CLI reference](references/cli-reference.md)\n- [Setup and security guide](references/setup-guide.md)\n- [Agent guardrails](references/agent-guardrails.md)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, Shell commands, Configuration, Guidance]\n\n**Output Format:** [Markdown with command snippets and structured operational recommendations]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May include Ceph health findings, risk-tier notes, dry-run steps, audit context, and undo guidance.]\n\n## Skill Version(s):\n\n0.11.3 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v0.11.2: 7 files, 17296 bytes\n\nFiles: references/agent-guardrails.md (6913b), references/capabilities.md (4566b), references/cli-reference.md (2640b), references/setup-guide.md (4071b), skill-card.md (2697b), SKILL.md (14394b), _meta.json (130b)\n\nFile v0.11.2:SKILL.md\n\n---\nname: ceph-aiops\nslug: ceph-aiops\ndisplayName: \"Ceph AIops\"\nsummary: \"Governed Ceph mgr ops: HEALTH_WARN RCA, OSD/PG/pool/RBD/CephFS/RGW, 37 tools.\"\nlicense: MIT\nhomepage: https://github.com/AIops-tools/Ceph-AIops\ntags: [aiops, mcp, governance, ceph]\ndescription: >\n  Use this skill whenever the user needs to operate or diagnose a Ceph cluster via its ceph-mgr Dashboard REST API — decode a HEALTH_WARN/ERR state into cause + action (cluster_health), read the cluster status, inspect OSDs (tree/df/perf), placement groups (summary/stuck/scrub), pools (list/usable capacity), RBD images and snapshots, CephFS/MDS and RGW status, monitors/managers, slow ops and capacity forecast — plus governed writes (set cluster flags, reweight/mark-in/mark-out/purge OSDs, trigger scrubs, set pool quota/pg_num/autoscale/size, create/delete pools, create/delete RBD images and snapshots, throttle recovery/backfill).\n  Always use this skill for \"ceph health\", \"what does this HEALTH_WARN mean\", \"PG_DEGRADED / OSD_NEARFULL / SLOW_OPS / MON_DOWN\", \"ceph -s\", \"which OSD is most full\", \"drain an OSD\", \"purge an OSD\", \"stuck PGs\", \"overdue scrub\", \"pool usable capacity\", \"set pool size / quota\", \"rebalance is too slow / throttle backfill\", \"RBD image or snapshot\", \"MDS behind on trimming\", \"RGW large omap\", \"mon quorum\", or \"days to nearfull\" when the context is a Ceph cluster (cephadm, hypervisor-bundled Ceph, or MicroCeph).\n  Do NOT use when the target is not Ceph — a hypervisor, a different storage appliance, a backup product, a Kubernetes cluster, or a network device. Route those to the appropriate other AIops-tools skill (negative routing hint only).\n  Common Ceph ops with a built-in governance harness (audit, policy, token budget, undo, risk-tiers).\ninstaller:\n  kind: uv\n  package: ceph-aiops\nargument-hint: \"[ceph question or describe your cluster task]\"\nallowed-tools:\n  - Bash\nmetadata: {\"openclaw\":{\"requires\":{\"anyBins\":[\"ceph-aiops\",\"uvx\"]},\"optional\":{\"env\":[\"CEPH_AIOPS_CONFIG\",\"CEPH_AIOPS_MASTER_PASSWORD\"]},\"homepage\":\"https://github.com/AIops-tools/Ceph-AIops\",\"emoji\":\"🐙\",\"os\":[\"macos\",\"linux\"]}}\ncompatibility: >\n  Standalone, self-governed Ceph operations. The governance harness (audit, policy, token/runaway budget, undo, risk-tiers) is bundled in the package — no external skill-family dependency. Works against vanilla ceph-mgr (cephadm / hypervisor-bundled Ceph / MicroCeph); no croit and no Kubernetes dependency.\n  All write operations are audited to a local SQLite DB under ~/.ceph-aiops/ (relocatable via CEPH_AIOPS_HOME).\n  Connection: the ceph-mgr Dashboard REST API over HTTPS (default port 8443). Authentication is username + password exchanged for a short-lived JWT at POST /api/auth; the mgr 'dashboard' module must be enabled. The username lives in config.yaml; the password is stored ENCRYPTED in ~/.ceph-aiops/secrets.enc (Fernet/AES-128 + scrypt-derived key) — never plaintext on disk. Run 'ceph-aiops init' to onboard, or 'ceph-aiops secret set <target>' to add one. The store is unlocked by a master password from CEPH_AIOPS_MASTER_PASSWORD (non-interactive/MCP/CI) or an interactive prompt (CLI on a TTY). A legacy plaintext env var CEPH_<TARGET_NAME_UPPER>_PASSWORD is still honoured as a fallback with a deprecation warning (migrate with 'ceph-aiops secret migrate'). The password is held only in memory and exchanged for a JWT at request time; secrets are never logged or echoed.\n  State-changing operations require double confirmation at the CLI layer and support --dry-run. All write tools pass through the @governed_tool decorator (pre-check + budget guard + audit + risk-tier label). High-risk destructive ops (osd_mark_out, osd_purge, pool_delete, set_pool_size, rbd_image_delete, rbd_snapshot_delete) require dry-run + double confirmation; reversible writes (osd_reweight, cluster_flag_set, set_pool_quota/pg_num/autoscale, throttle_recovery) capture the prior state and record an inverse undo descriptor.\n  Webhooks: none — no outbound network calls beyond the configured ceph-mgr Dashboard REST API.\n  SSL: verify_ssl defaults to true; disable only for self-signed lab certificates.\n  Transitive dependencies: httpx (HTTP client) and the MCP SDK. No post-install scripts or background services.\n  Validation status: behaviour is exercised against mocked Dashboard responses; multi-node rebalance behaviour and the write ops have not been run against a live cluster (a single-node MicroCeph running 'ceph-aiops doctor' is the cheapest live path; see docs/VERIFICATION.md). The Dashboard API has no ETag/pagination, so none are exposed.\n---\n\n# Ceph AIops\n\n> **Disclaimer**: Community-maintained open-source project, **not affiliated with, endorsed by, or sponsored by the Ceph project or any storage vendor.** Product and trademark names belong to their owners. Source at [github.com/AIops-tools/Ceph-AIops](https://github.com/AIops-tools/Ceph-AIops) under the MIT license.\n\nGoverned Ceph operations via the **ceph-mgr Dashboard REST API** — **37 MCP tools**, every one wrapped with the bundled `@governed_tool` harness: a local unified audit log under `~/.ceph-aiops/`, token/runaway budget guard, undo-token recording, and descriptive risk tiers. The Dashboard password is stored **encrypted** (`~/.ceph-aiops/secrets.enc`, Fernet + scrypt) — never plaintext on disk. The flagship `cluster_health` turns raw HEALTH_WARN/ERR check codes into plain-language cause + suggested action.\n\n> **Standalone**: the governance harness is bundled in the package (`ceph_aiops.governance`) — ceph-aiops has no external skill-family dependency. Works against vanilla ceph-mgr (cephadm / hypervisor-bundled / MicroCeph); no croit, no Kubernetes.\n\n## What This Skill Does\n\n| Group | Tools | Count | Read or Write |\n|-------|-------|:-----:|:-------------:|\n| **Health** | cluster_health (flagship RCA), cluster_status | 2 | 2 read |\n| **OSD** | osd_tree, osd_df, osd_perf | 3 | 3 read |\n| | cluster_flag_set, osd_reweight, osd_mark_in, osd_mark_out, osd_purge | 5 | 5 write |\n| **PG** | pg_summary, pg_dump_stuck, scrub_status | 3 | 3 read |\n| | trigger_scrub, trigger_deep_scrub | 2 | 2 write |\n| **Pool** | pool_ls, pool_df | 2 | 2 read |\n| | set_pool_quota, set_pool_pg_num, set_pool_autoscale, pool_create, set_pool_size, pool_delete | 6 | 6 write |\n| **RBD** | rbd_ls | 1 | 1 read |\n| | rbd_image_create, rbd_snapshot_create, rbd_image_delete, rbd_snapshot_delete | 4 | 4 write |\n| **CephFS / RGW** | cephfs_status, rgw_status | 2 | 2 read |\n| **Cluster-ops** | mon_status, mgr_status, slow_ops, capacity_forecast | 4 | 4 read |\n| | throttle_recovery | 1 | 1 write |\n| **Undo** | undo_list, undo_apply | 2 | 2 undo |\n\nTotals: **37 tools — 17 read, 18 write, 2 undo.** The MCP server exposes all 37; the CLI is a convenience subset.\n\n## Quick Install\n\n```bash\nuv tool install ceph-aiops\nceph-aiops init       # interactive wizard: mgr host/port/username + encrypted Dashboard password\nceph-aiops doctor\n```\n\nOr as an OpenClaw plugin, which installs this skill and its MCP server together:\n\n```bash\nopenclaw plugins install clawhub:@zw008/ceph-aiops\nopenclaw skills info ceph-aiops          # expect: Visible to model: yes\n```\n\nNeeds `uvx` on `PATH`: the MCP server is fetched with uv, pinned to this release.\n\n## When to Use This Skill\n\n- Decode a **HEALTH_WARN/ERR** state (`cluster_health` / `health detail`) — cause + action per active check (`PG_DEGRADED`, `OSD_NEARFULL`, `SLOW_OPS`, `MON_DOWN`, `LARGE_OMAP_OBJECTS`, …)\n- One-shot triage (`overview`): HEALTH status + active checks + OSD up/in counts\n- Inspect OSDs (`osd_tree` / `osd_df` most-full first / `osd_perf` slowest first), PGs (`pg_summary` / `pg_dump_stuck` / `scrub_status`), pools (`pool_ls` / `pool_df` usable capacity)\n- Investigate slow requests (`slow_ops`), MDS trimming lag (`cephfs_status`), RGW large-omap (`rgw_status`), mon quorum (`mon_status`), and days-to-nearfull (`capacity_forecast`)\n- Safely **drain + purge** an OSD, change **pool size/quota**, or **throttle a slow rebalance** (governed writes with dry-run + undo)\n\n**Do NOT use when** the target is not Ceph (a hypervisor, another storage appliance, a backup product, a container cluster, or a network device). Route those to the appropriate **other AIops-tools** skill.\n\n## Related Skills — Skill Routing\n\n| If the user wants… | Use |\n|--------------------|-----|\n| Ceph: HEALTH_WARN RCA, OSD/PG/pool/RBD/CephFS/RGW, rebalance, slow ops | **ceph-aiops** (this skill) |\n| Any non-Ceph target (hypervisor, other storage, backup, cluster, network) | the appropriate **other AIops-tools** skill |\n\n## Common Workflows\n\n### 1. \"The cluster went HEALTH_WARN overnight\" — decode it (read-only)\n\n1. `ceph-aiops doctor` → confirm the mgr Dashboard is reachable and the JWT login works before trusting anything else\n2. `ceph-aiops overview` → HEALTH status, the list of active check codes, and OSD up/in counts in one shot\n3. `ceph-aiops health detail` (MCP: `cluster_health`) → each **active** check translated into what it means, the likely cause, and a suggested action\n4. Drill into the implicated resource: `PG_DEGRADED` → `pg_dump_stuck`; `OSD_NEARFULL` → `ceph-aiops osd df` (most-full first); `SLOW_OPS` → `slow_ops`; `MON_DOWN` → `mon_status`; `LARGE_OMAP_OBJECTS` → `rgw_status`\n5. **Failure branch**: if `doctor` fails on auth, the Dashboard password is wrong or the store is locked — re-run `ceph-aiops secret set <target>` (or export `CEPH_AIOPS_MASTER_PASSWORD` for non-interactive use). If `doctor` fails on reachability, the mgr `dashboard` module is likely not enabled; no read is issued against an unauthenticated session.\n\n### 2. Retire a failing OSD: drain, mark out, purge (governed)\n\n1. `ceph-aiops health detail` → confirm the OSD is genuinely the problem (e.g. `OSD_SLOW_PING_TIME`, repeated `SLOW_OPS` on one id) rather than a cluster-wide symptom\n2. `ceph-aiops osd df` → confirm the id, and that the remaining OSDs have room to absorb its data before you drain anything\n3. `ceph-aiops osd reweight <id> 0.0` → start a gradual drain; reversible, the prior CRUSH weight is captured as the undo descriptor\n4. `ceph-aiops osd out <id> --dry-run`, then re-run without `--dry-run` → **high** risk, double confirmation, needs `CEPH_AUDIT_APPROVED_BY`\n5. Wait for `ceph-aiops health status` / `pg_summary` to show all PGs `active+clean` — do not purge while backfill is running\n6. `ceph-aiops osd purge <id> --dry-run`, then re-run without `--dry-run` → **high**, irreversible\n7. **Failure branch**: if client I/O tanks during the drain, stop and reverse — `ceph-aiops undo list` then `ceph-aiops undo apply <id>` restores the prior weight (and `osd_mark_in` reverses the mark-out). Purge has no undo, which is exactly why it comes last and after `active+clean`.\n\n### 3. Recovery is starving client I/O\n\n1. `ceph-aiops health detail` → confirm the cluster is actually backfilling/recovering (`PG_DEGRADED`, `PG_BACKFILL_FULL`) rather than hitting a different bottleneck\n2. `slow_ops` → check whether client requests are genuinely being blocked, and by which OSDs\n3. `throttle_recovery(max_backfills=1, recovery_max_active=1)` → **med** risk, reversible; the prior `osd_max_backfills` / `osd_recovery_max_active` are captured as the undo descriptor\n4. Re-check `slow_ops` and `pg_summary` — recovery is slower but client latency should recover\n5. Once the cluster is quiet, raise the values back (or `ceph-aiops undo apply <id>` to restore the exact prior settings)\n6. **Failure branch**: if throttling does not help, the bottleneck is not recovery — go back to `osd_perf` (slowest OSDs first) and `mon_status`; do not keep lowering the throttle, you will only extend the degraded window.\n\n### 4. A pool is running out of usable capacity\n\n1. `ceph-aiops overview` → look for `POOL_NEARFULL` / `OSD_NEARFULL` among the active checks\n2. `pool_df` → per-pool usage with **usable capacity = raw ÷ size** (a `size=3` pool reports a third of raw — this is where most \"but the disks aren't full\" confusion comes from)\n3. `capacity_forecast` → days-to-nearfull at the current fill rate, so you know whether this is a this-week problem or a this-quarter one\n4. Buy time reversibly first: `set_pool_quota` (med, undo → prior quota) or `set_pool_autoscale` (med, undo) to let pg_num track the new size\n5. Only if a replica change is genuinely the answer: `set_pool_size --dry-run` then the real call — **high** risk, because lowering `size` reduces durability and any change forces cluster-wide data movement\n6. **Failure branch**: if the resulting rebalance saturates the cluster, apply workflow 3 (`throttle_recovery`) rather than reverting the size mid-flight; if the size change itself was wrong, `ceph-aiops undo apply <id>` replays the recorded prior value — but expect a second full rebalance.\n\n## Governance & Safety\n\nThe skill delivers reads and writes and records them; it does **not** decide\nwhether a write is permitted. That is your agent's judgement, or the permission\nof the account you connect it with (a ceph-mgr Dashboard account with a\nread-only role — writes then fail at the mgr). There is no read-only switch,\npolicy file, or approval gate.\n\n- **Audit is the guarantee, and it is not bypassable.** Every operation — MCP and CLI alike — is logged to `~/.ceph-aiops/audit.db` (relocatable via `CEPH_AIOPS_HOME`): params, result, status, duration, and the risk tier. The CLI writes the same row the MCP path does.\n- `CEPH_AUDIT_APPROVED_BY` / `CEPH_AUDIT_RATIONALE` are optional annotations recorded on the audit row (who/why); they are never required and never block.\n- **Runaway guard** — a safety backstop, not authorization: the same call looped in a tight window trips a circuit breaker. Disable with `CEPH_RUNAWAY_MAX=0`.\n- Destructive writes support `--dry-run` / `dry_run=True` and double confirmation at the CLI.\n- Reversible writes fetch the real before-state and record an inverse descriptor (`osd_reweight`→restore prior weight, `cluster_flag_set`→toggle back); irreversible ops (`osd_purge`, `pool_delete`, RBD deletes) record only the before-state.\n\n## References\n\n- `references/capabilities.md` — full tool → API-path → returns reference\n- `references/cli-reference.md` — CLI command reference\n- `references/setup-guide.md` — onboarding, credentials, and connectivity\n\nFile v0.11.2:_meta.json\n\n{\n  \"ownerId\": \"kn7b067awq2s97bn3d7p5qfhw5827pxc\",\n  \"slug\": \"ceph-aiops\",\n  \"version\": \"0.11.2\",\n  \"publishedAt\": 1789221901564\n}\n\nFile v0.11.2:references/agent-guardrails.md\n\n# Agent guardrails — running ceph-aiops with a smaller / local model\n\nIf you drive these tools with a local model (Llama, Qwen, Mistral … via Goose,\nOllama, LM Studio, or any OpenAI-compatible runtime), you will get noticeably\nbetter results with a short system prompt. This page gives you one, and — more\nimportantly — tells you which guardrails you **no longer need to write**, because\nthe tool now enforces them itself.\n\nThe distinction matters. A guardrail in a prompt is a request. A guardrail in the\nharness is a guarantee. Anything below that we could move into the harness, we did.\n\n## Authorization is not this tool's job — decide it where it belongs\n\nWhether a write should happen is your decision, or the account's. The tool does\nnot gate it — there is no read-only switch and no approval prompt to configure.\nThe two right places to control read vs write:\n\n- **The account you connect with.** Give it a ceph-mgr Dashboard account with a\n  read-only role. A write then fails at the mgr, which is the only place the\n  permission actually lives — no skill-side flag can be argued around by a\n  model, but a revoked permission cannot be.\n- **Your agent's system prompt.** If you want an observe-only session, tell the\n  model not to call the write tools (they are clearly tagged `[WRITE]`).\n\nWhat the tool *does* guarantee is that you can always see what happened:\n\n## What the tool enforces — do not waste prompt budget on these\n\n| You might be tempted to prompt | Why you don't need to |\n|---|---|\n| \"Log everything you do, over both MCP and the CLI\" | Every call is audited to `~/.ceph-aiops/audit.db` regardless of what the model says it did — and the CLI writes the same row the MCP path does, so there is no unaudited entry point. Reversible writes also record an undo token capturing the *prior* state. |\n| \"Don't invent a value when a field is missing\" | A field the Dashboard did not return comes back as `null`, never as `\"\"`. A missing `deviceClass`, `host`, MDS `state`, or `pg_autoscale_mode` is distinguishable from an empty one in the payload. |\n| \"Tell me if the output was cut off\" | `pg_summary` and `pg_dump_stuck` return `{\"stuck\": [...], \"returned\": N, \"limit\": L, \"truncated\": true/false}`. Truncation is measured (one extra row is collected), not guessed. `pg_summary` also keeps `unhealthyCount` as the true total even when the list is capped. |\n| \"Explain what HEALTH_WARN means\" | `cluster_health` already folds each active check code (`PG_DEGRADED`, `OSD_NEARFULL`, `SLOW_OPS`, `LARGE_OMAP_OBJECTS`, …) into a plain-language `cause` and `suggestedAction`. The model should quote those, not compose its own. |\n| \"Confirm before anything destructive\" | Destructive operations (`osd_purge`, `pool_delete`, `rbd_image_delete`, `rbd_snapshot_delete`, `set_pool_size`) require a `--dry-run`-able preview + double confirmation at the CLI. |\n| \"Don't get stuck retrying\" | The runaway guard trips a circuit breaker if the same call is hammered in a tight loop — a stuck agent is stopped rather than left to burn calls and time. |\n\n## What still needs a prompt\n\nThese are model-behaviour problems the harness cannot fix from the outside.\nCopy this into your agent's system prompt:\n\n```text\nYou operate a Ceph cluster through the ceph-aiops MCP tools, which talk to the\nceph-mgr Dashboard REST API.\n\nTOOL USE\n- Before answering any question about the current cluster, you MUST call a tool.\n  Never answer from memory or assumption.\n- Actually invoke the tool. Do not describe the call you would make, and do not\n  emit an example JSON response in place of calling it.\n- If a tool call fails, report the real error verbatim. Never fill the gap with\n  a plausible-sounding answer. A read that fails returns an \"error\" field rather\n  than raising — treat that as \"unknown\", not as \"healthy\".\n\nREADING RESULTS\n- Read the whole result before concluding. If a result contains a \"truncated\"\n  field that is true, say so and re-run with a higher limit instead of treating\n  the partial result as complete.\n- A null field means the Dashboard did not return that value. Report it as \"not\n  available\" — never infer it.\n- Report values exactly as returned. Do not normalise, translate, or prettify\n  status strings (HEALTH_WARN, active+undersized+degraded), PG ids, or OSD ids.\n- When cluster_health returns findings, quote each finding's \"cause\" and\n  \"suggestedAction\" rather than composing your own explanation of the check code.\n\nSCOPE\n- Separate observation from interpretation. State what the tools returned, then\n  any interpretation, clearly marked as such.\n- Do not assert a capacity, performance, or data-loss problem unless a tool\n  result supports it. HEALTH_WARN is not automatically an emergency —\n  PG_NOT_DEEP_SCRUBBED on a small cluster is routine.\n- Do not confuse the identifier kinds: an OSD id is a number (3), a PG id is\n  pool.hex (\"2.1a\"), a pool name is a string, and an RBD image is addressed as\n  pool/name. Never pass one where another is expected.\n- capacity_forecast is arithmetic extrapolation from a growth rate you supply.\n  With no growth rate it reports \"insufficient-data\" — do not present that as a\n  prediction.\n```\n\n## Recommended setup for a local model\n\nStart with a connection that *cannot* write, verify, and widen the account's\npermission only when you trust the setup — the destructive operations on a Ceph\ncluster are unusually cheap to invoke and unusually expensive to undo\n(`pool_delete` and `rbd_image_delete` destroy data no undo token can bring back):\n\n```bash\n# e.g. use a ceph-mgr Dashboard account with a read-only role. Then:\nceph-aiops doctor\n```\n\nOptionally annotate the audit trail with who is operating and why — recorded on\nevery row, never required:\n\n```bash\nexport CEPH_AUDIT_APPROVED_BY=\"your.name@example.com\"\nexport CEPH_AUDIT_RATIONALE=\"draining osd.7 for disk replacement 2026-07-20\"\n```\n\n## If your model still struggles\n\nSome behaviours are model-capacity limits rather than prompt problems:\n\n- **Multi-tool workflows time out or drift.** Prefer `cluster_health` and\n  `fleet_overview` — they do the multi-step correlation inside one call, so the\n  model does not have to chain reads and keep OSD/PG ids straight.\n- **The model ignores later tool results in a long context.** Ask narrower\n  questions and use `limit` deliberately rather than dumping every PG in the\n  cluster; `pg_summary`'s histogram is usually the right level of detail.\n- **The model describes calls instead of making them.** This is usually a\n  runtime/tool-calling-format mismatch, not a prompt problem — check that your\n  client advertises the tools in the format your model was trained on.\n\nFeedback on running this with a specific local model is genuinely useful —\nopen an issue at\n[github.com/AIops-tools/Ceph-AIops](https://github.com/AIops-tools/Ceph-AIops/issues)\nwith the model, runtime, and what went wrong.\n\nFile v0.11.2:references/capabilities.md\n\n# ceph-aiops capabilities\n\n> 37 MCP tools (17 read, 18 write, 2 undo) over the **ceph-mgr\n> Dashboard REST API** (`https://<host>:8443`, JWT via `POST /api/auth`).\n> Multi-node rebalance behaviour and the write ops need live verification\n> (see `docs/VERIFICATION.md`).\n\n## Read tools (17)\n\n| Tool | API path | Returns |\n|------|----------|---------|\n| `cluster_health` | `GET /api/health/full` | **flagship RCA** — per active HEALTH_WARN/ERR check: code, plain-language meaning, likely cause, suggested action |\n| `cluster_status` | `GET /api/health/minimal` | `ceph -s` summary: health status, mon/mgr/osd/pg counts |\n| `osd_tree` | `GET /api/osd` | OSD tree: up/in, CRUSH weight, host, device class |\n| `osd_df` | `GET /api/osd` | per-OSD utilization %, **most-full first**, near-full / backfill-full flags |\n| `osd_perf` | `GET /api/osd` | commit/apply latency per OSD, **slowest first** |\n| `pg_summary` | `GET /api/pg` (+ `/api/health/full`) | PG **state histogram** + list of non-active+clean PGs |\n| `pg_dump_stuck` | `GET /api/pg` | stuck PGs (inactive/unclean/stale/undersized) + implicated OSDs |\n| `scrub_status` | `GET /api/health/full` | PGs overdue for scrub / deep-scrub |\n| `pool_ls` | `GET /api/pool` | pools: name, id, size, pg_num, autoscale mode, application |\n| `pool_df` | `GET /api/pool` | per-pool usage; **usable capacity = raw ÷ size** |\n| `rbd_ls` | `GET /api/block/image` | RBD images (optionally filtered by pool): name, size, pool |\n| `cephfs_status` | `GET /api/cephfs` | MDS ranks + **\"behind on trimming\"** + client count |\n| `rgw_status` | `GET /api/rgw/daemon` + `GET /api/rgw/bucket` | RGW daemons + buckets + **LARGE_OMAP / unsharded-index** findings |\n| `mon_status` | `GET /api/monitor` | monitors: in-quorum vs **out-of-quorum** |\n| `mgr_status` | `GET /api/health/full` | active mgr, standbys, enabled modules |\n| `slow_ops` | `GET /api/health/full` | blocked / slow requests grouped **by OSD** |\n| `capacity_forecast` | `GET /api/osd` (+ df) | raw/used/avail + **days-to-nearfull** projection |\n\n## Write tools (18)\n\n| Tool | Risk | API path | Undo / safety |\n|------|------|----------|---------------|\n| `cluster_flag_set` | medium | `GET`+`PUT /api/osd/flags` | set/unset noout/noscrub/nobackfill/norecover; captures prior flag set (undo) |\n| `osd_reweight` | medium | `POST /api/osd/{id}/reweight` | 0.0 = drain; captures prior weight (undo) |\n| `osd_mark_in` | medium | `POST /api/osd/{id}/mark` | captures prior up/in state (undo) |\n| `osd_mark_out` | **high** | `POST /api/osd/{id}/mark` | drains data; CLI double-confirm + dry-run; captures prior state |\n| `osd_purge` | **high** | `DELETE /api/osd/{id}` | destroy + crush rm + auth del; **irreversible**; dry-run + double-confirm |\n| `trigger_scrub` | medium | `POST /api/pg/{pgid}/scrub` | schedule a shallow scrub; no prior state |\n| `trigger_deep_scrub` | medium | `POST /api/pg/{pgid}/deep_scrub` | schedule a deep (data-integrity) scrub |\n| `set_pool_quota` | medium | `PUT /api/pool/{name}` | captures prior max_bytes/max_objects (undo) |\n| `set_pool_pg_num` | medium | `PUT /api/pool/{name}` | captures prior pg_num (undo) |\n| `set_pool_autoscale` | medium | `PUT /api/pool/{name}` | captures prior autoscale mode (undo) |\n| `pool_create` | medium | `POST /api/pool` | create a new pool |\n| `set_pool_size` | **high** | `PUT /api/pool/{name}` | replica change **forces data movement**; dry-run + double-confirm; captures prior size |\n| `pool_delete` | **high** | `DELETE /api/pool/{name}` | **destroys all data**; dry-run + double-confirm |\n| `rbd_image_create` | medium | `POST /api/block/image` | create an RBD image |\n| `rbd_snapshot_create` | medium | `POST /api/block/image/{spec}/snap` | reversible → delete the snapshot |\n| `rbd_image_delete` | **high** | `DELETE /api/block/image/{spec}` | **irreversible**; dry-run + double-confirm |\n| `rbd_snapshot_delete` | **high** | `DELETE /api/block/image/{spec}/snap/{snap}` | **irreversible**; dry-run + double-confirm |\n| `throttle_recovery` | medium | `GET`+`POST /api/cluster_conf` | tunes `osd_max_backfills` / `osd_recovery_max_active`; captures prior values (undo) |\n\n## Out of scope (by design)\n\n- RGW **multisite** replication and zone/zonegroup management\n- **NFS-Ganesha** exports\n- **cephadm orchestrator** host/daemon management (add/remove hosts, deploy daemons)\n- Per-daemon config sprawl beyond the recovery-tuning keys above\n\nThe ceph-mgr Dashboard API has no ETag / pagination, so this tool exposes none.\n\nWant one of these? Open an issue or PR — feedback and contributions welcome.\n\nFile v0.11.2:references/cli-reference.md\n\n# ceph-aiops CLI reference\n\n> The CLI is a convenience subset; the full 37-tool surface\n> is via the MCP server (`ceph-aiops mcp`). Talks to the ceph-mgr Dashboard REST\n> API (`https://<host>:8443`, JWT via `POST /api/auth`).\n\n## Setup & diagnostics\n\n```bash\nceph-aiops init                      # interactive onboarding wizard\nceph-aiops doctor [--skip-auth]      # config + secret store + JWT login + mgr-dashboard reachability\nceph-aiops mcp                       # start the MCP server (stdio transport)\n```\n\n## Secrets (encrypted store ~/.ceph-aiops/secrets.enc)\n\n```bash\nceph-aiops secret set <target> [--value <password>]  # store Dashboard password (hidden prompt if no --value)\nceph-aiops secret list                               # names only — values never shown\nceph-aiops secret rm <target>\nceph-aiops secret migrate                            # import legacy plaintext .env (CEPH_<T>_PASSWORD)\nceph-aiops secret rotate-password                    # re-encrypt under a new master password\n```\n\n## Read commands\n\n```bash\nceph-aiops overview [--target <t>]        # HEALTH status + active checks + OSD up/in\nceph-aiops health detail                  # decode active HEALTH_WARN/ERR checks → cause + action (RCA)\nceph-aiops health status                  # ceph -s summary\nceph-aiops osd tree                        # OSD tree: up/in, weight, host, device class\nceph-aiops osd df                          # per-OSD utilization, most-full first, near/backfill-full flags\n```\n\n## Write commands (governed; risk tier in parentheses)\n\n```bash\nceph-aiops osd reweight <osd_id> <weight> [--dry-run]   # (med) 0.0 = drain; reversible → prior weight\nceph-aiops osd out <osd_id> [--dry-run]                 # (high) mark out — drains data; double confirm\nceph-aiops osd purge <osd_id> [--dry-run]               # (high) purge — irreversible; double confirm\n```\n\nThe remaining writes (cluster flags, pool quota/pg_num/autoscale/size/create/delete,\nRBD image/snapshot create/delete, trigger scrubs, throttle recovery) are exposed\nthrough the **MCP server**, not the CLI.\n\n## Common options\n\n- `--target, -t <name>` — target name from `config.yaml` (omit to use the default/first target)\n- `--dry-run` — print the API call that would be made, change nothing\n- Destructive commands (`osd out`, `osd purge`) require `--dry-run` review + double confirmation\n- `doctor --skip-auth` — skip the JWT login / connectivity check (config + secret-store checks only)\n\n## Approver env vars (high-risk ops)\n\n```bash\nexport CEPH_AUDIT_APPROVED_BY='you@example.com'\nexport CEPH_AUDIT_RATIONALE='draining failed OSD 7 per ticket OPS-123'\n```\n\nFile v0.11.2:references/setup-guide.md\n\n# ceph-aiops setup & security guide\n\n> The cheapest live check is a single-node MicroCeph running `ceph-aiops doctor`.\n> See `docs/VERIFICATION.md` for the full live-verification checklist.\n\n## 1. Install\n\n```bash\nuv tool install ceph-aiops\n```\n\n## 2. Enable the ceph-mgr Dashboard module\n\nceph-aiops talks to the **ceph-mgr Dashboard REST API** (HTTPS, default port\n`8443`). The mgr **dashboard** module must be enabled and a Dashboard user must\nexist:\n\n```bash\nceph mgr module enable dashboard\nceph dashboard ac-user-create <username> -i <password-file> administrator\n# find the URL/port: ceph mgr services   → e.g. https://<host>:8443/\n```\n\nceph-aiops authenticates by exchanging the **username + password** for a\nshort-lived **JWT** at `POST /api/auth`; the token is cached in memory and used\nas a Bearer header for subsequent calls.\n\n## 3. Onboard\n\n```bash\nceph-aiops init\n```\n\nThe wizard collects (non-secret) connection details into\n`~/.ceph-aiops/config.yaml` and stores the Dashboard **password** encrypted into\n`~/.ceph-aiops/secrets.enc`. Example config:\n\n```yaml\ntargets:\n  - name: ceph1\n    host: 10.0.0.30\n    port: 8443\n    username: admin\n    verify_ssl: false          # self-signed lab certs only\n```\n\nThe `username` lives in the config file (it is not a secret); the password never\ndoes.\n\n## 4. Non-interactive use (MCP server / CI / cron)\n\nExport the master password so the encrypted store can be unlocked without a\nprompt:\n\n```bash\nexport CEPH_AIOPS_MASTER_PASSWORD='your-master-password'\n```\n\n## Credential security\n\n- The Dashboard password is **never** written to disk in plaintext. It lives only\n  in `~/.ceph-aiops/secrets.enc`, encrypted with Fernet (AES-128-CBC + HMAC), the\n  key derived from your master password via scrypt. Only a per-store random salt\n  and the ciphertext are on disk (chmod 600); the master password itself is never\n  stored.\n- A legacy plaintext env var `CEPH_<TARGET_NAME_UPPER>_PASSWORD` is still honoured\n  as a fallback with a deprecation warning — migrate with\n  `ceph-aiops secret migrate` (it imports then renames the old `.env`).\n- The password is held only in memory, exchanged for a JWT at request time, and\n  is never logged or echoed; exception text and tracebacks are scrubbed of\n  secret-shaped strings before being written to the audit log.\n\n## Audit-annotation env vars (optional)\n\nThe skill does not decide whether a write is permitted — that is the agent's\njudgement or the connecting Dashboard account's role. If you want the audit trail\nto record\n\nArchive v0.11.1: 7 files, 17386 bytes\n\nFiles: references/agent-guardrails.md (6913b), references/capabilities.md (4566b), references/cli-reference.md (2640b), references/setup-guide.md (4071b), skill-card.md (2996b), SKILL.md (14400b), _meta.json (130b)\n\nArchive v0.11.0: 7 files, 16957 bytes\n\nFiles: references/agent-guardrails.md (6913b), references/capabilities.md (4566b), references/cli-reference.md (2640b), references/setup-guide.md (4071b), skill-card.md (2290b), SKILL.md (14092b), _meta.json (130b)\n\nArchive v0.10.0: 7 files, 17088 bytes\n\nFiles: references/agent-guardrails.md (6913b), references/capabilities.md (4566b), references/cli-reference.md (2640b), references/setup-guide.md (4071b), skill-card.md (2587b), SKILL.md (14191b), _meta.json (130b)\n\nArchive v0.9.0: 7 files, 17065 bytes\n\nFiles: references/agent-guardrails.md (6913b), references/capabilities.md (4566b), references/cli-reference.md (2640b), references/setup-guide.md (4071b), skill-card.md (2572b), SKILL.md (14191b), _meta.json (129b)\n\nArchive v0.8.0: 7 files, 17110 bytes\n\nFiles: references/agent-guardrails.md (6913b), references/capabilities.md (4566b), references/cli-reference.md (2640b), references/setup-guide.md (4071b), skill-card.md (2747b), SKILL.md (14191b), _meta.json (129b)\n\nArchive v0.7.0: 7 files, 16467 bytes\n\nFiles: references/agent-guardrails.md (6913b), references/capabilities.md (4566b), references/cli-reference.md (2640b), references/setup-guide.md (4071b), skill-card.md (1294b), SKILL.md (14191b), _meta.json (129b)","readmeExcerpt":"Skill: ceph-aiops Owner: zw008 Summary: Use this skill whenever the user needs to operate or diagnose a Ceph cluster via its ceph-mgr Dashboard REST API — decode a HEALTH_WARN/ERR state into cause + action (cluster_health), read the cluster status, inspect OSDs (tree/df/perf), placement groups (summary/stuck/scrub), pools (list/usable capacity), RBD images and snapshots, CephFS/MDS and RGW status, monitors/managers, ","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"uv tool install ceph-aiops\nceph-aiops init       # interactive wizard: mgr host/port/username + encrypted Dashboard password\nceph-aiops doctor"},{"language":"bash","snippet":"openclaw plugins install clawhub:@zw008/ceph-aiops\nopenclaw skills info ceph-aiops          # expect: Visible to model: yes"},{"language":"text","snippet":"You operate a Ceph cluster through the ceph-aiops MCP tools, which talk to the\nceph-mgr Dashboard REST API.\n\nTOOL USE\n- Before answering any question about the current cluster, you MUST call a tool.\n  Never answer from memory or assumption.\n- Actually invoke the tool. Do not describe the call you would make, and do not\n  emit an example JSON response in place of calling it.\n- If a tool call fails, report the real error verbatim. Never fill the gap with\n  a plausible-sounding answer. A read that fails returns an \"error\" field rather\n  than raising — treat that as \"unknown\", not as \"healthy\".\n\nREADING RESULTS\n- Read the whole result before concluding. If a result contains a \"truncated\"\n  field that is true, say so and re-run with a higher limit instead of treating\n  the partial result as complete.\n- A null field means the Dashboard did not return that value. Report it as \"not\n  available\" — never infer it.\n- Report values exactly as returned. Do not normalise, translate, or prettify\n  status strings (HEALTH_WARN, active+undersized+degraded), PG ids, or OSD ids.\n- When cluster_health returns findings, quote each finding's \"cause\" and\n  \"suggestedAction\" rather than composing your own explanation of the check code.\n\n- `pool_delete`, `rbd_image_delete`, `rbd_snapshot_delete` and `set_pool_size` have no CLI\n  command, so nothing will ask you to confirm them. Call them with `dry_run=True` first,\n  show the operator what would change, and wait for an explicit go-ahead.\n\nSCOPE\n- Separate observation from interpretation. State what the tools returned, then\n  any interpretation, clearly marked as such.\n- Do not assert a capacity, performance, or data-loss problem unless a tool\n  result supports it. HEALTH_WARN is not automatically an emergency —\n  PG_NOT_DEEP_SCRUBBED on a small cluster is routine.\n- Do not confuse the identifier kinds: an OSD id is a number (3), a PG id is\n  pool.hex (\"2.1a\"), a pool name is a string, and an RBD image is addressed as\n  pool/name. Never pass o"},{"language":"bash","snippet":"# e.g. use a ceph-mgr Dashboard account with a read-only role. Then:\nceph-aiops doctor"},{"language":"bash","snippet":"export CEPH_AUDIT_APPROVED_BY=\"your.name@example.com\"\nexport CEPH_AUDIT_RATIONALE=\"draining osd.7 for disk replacement 2026-07-20\""},{"language":"bash","snippet":"ceph-aiops init                      # interactive onboarding wizard\nceph-aiops doctor [--skip-auth]      # config + secret store + JWT login + mgr-dashboard reachability\nceph-aiops mcp                       # start the MCP server (stdio transport)"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: ceph-aiops\nslug: ceph-aiops\ndisplayName: \"Ceph AIops\"\nsummary: \"Governed Ceph mgr ops: HEALTH_WARN RCA, OSD/PG/pool/RBD/CephFS/RGW, 37 tools.\"\nlicense: MIT\nhomepage: https://github.com/AIops-tools/Ceph-AIops\ntags: [aiops, mcp, governance, ceph]\ndescription: >\n  Use this skill whenever the user needs to operate or diagnose a Ceph cluster via its ceph-mgr Dashboard REST API — decode a HEALTH_WARN/ERR state into cause + action (cluster_health), read the cluster status, inspect OSDs (tree/df/perf), placement groups (summary/stuck/scrub), pools (list/usable capacity), RBD images and snapshots, CephFS/MDS and RGW status, monitors/managers, slow ops and capacity forecast — plus governed writes (set cluster flags, reweight/mark-in/mark-out/purge OSDs, trigger scrubs, set pool quota/pg_num/autoscale/size, create/delete pools, create/delete RBD images and snapshots, throttle recovery/backfill).\n  Always use this skill for \"ceph health\", \"what does this HEALTH_WARN mean\", \"PG_DEGRADED / OSD_NEARFULL / SLOW_OPS / MON_DOWN\", \"ceph -s\", \"which OSD is most full\", \"drain an OSD\", \"purge an OSD\", \"stuck PGs\", \"overdue scrub\", \"pool usable capacity\", \"set pool size / quota\", \"rebalance is too slow / throttle backfill\", \"RBD image or snapshot\", \"MDS behind on trimming\", \"RGW large omap\", \"mon quorum\", or \"days to nearfull\" when the context is a Ceph cluster (cephadm, hypervisor-bundled Ceph, or MicroCeph).\n  Do NOT use when the target is not Ceph — a hypervisor, a different storage appliance, a backup product, a Kubernetes cluster, or a network device. Route those to the appropriate other AIops-tools skill (negative routing hint only).\n  Common Ceph ops with a built-in governance harness (audit, policy, token budget, undo, risk-tiers).\ninstaller:\n  kind: uv\n  package: ceph-aiops\nargument-hint: \"[ceph question or describe your cluster task]\"\nallowed-tools:\n  - Bash\nmetadata: {\"openclaw\":{\"requires\":{\"anyBins\":[\"ceph-aiops\",\"uvx\"]},\"optional\":{\"env\":[\"CEPH_AIOPS_CONFIG\",\"CEPH_AIOPS_MASTER_PASSWORD\"]},\"homepage\":\"https://github.com/AIops-tools/Ceph-AIops\",\"emoji\":\"🐙\",\"os\":[\"macos\",\"linux\"]}}\ncompatibility: >\n  Standalone, self-governed Ceph operations. The governance harness (audit, policy, token/runaway budget, undo, risk-tiers) is bundled in the package — no external skill-family dependency. Works against vanilla ceph-mgr (cephadm / hypervisor-bundled Ceph / MicroCeph); no croit and no Kubernetes dependency.\n  All write operations are audited to a local SQLite DB under ~/.ceph-aiops/ (relocatable via CEPH_AIOPS_HOME).\n  Connection: the ceph-mgr Dashboard REST API over HTTPS (default port 8443). Authentication is username + password exchanged for a short-lived JWT at POST /api/auth; the mgr 'dashboard' module must be enabled. The username lives in config.yaml; the password is stored ENCRYPTED in ~/.ceph-aiops/secrets.enc (Fernet/AES-128 + scrypt-derived key) — never plaintext on disk. Run 'ceph-aiops init' to onboard, or 'ceph-aiops secret set <target>' to"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7b067awq2s97bn3d7p5qfhw5827pxc\",\n  \"slug\": \"ceph-aiops\",\n  \"version\": \"0.11.5\",\n  \"publishedAt\": 1789601119214\n}"},{"path":"references/agent-guardrails.md","content":"# Agent guardrails — running ceph-aiops with a smaller / local model\n\nIf you drive these tools with a local model (Llama, Qwen, Mistral … via Goose,\nOllama, LM Studio, or any OpenAI-compatible runtime), you will get noticeably\nbetter results with a short system prompt. This page gives you one, and — more\nimportantly — tells you which guardrails you **no longer need to write**, because\nthe tool now enforces them itself.\n\nThe distinction matters. A guardrail in a prompt is a request. A guardrail in the\nharness is a guarantee. Anything below that we could move into the harness, we did.\n\n## Authorization is not this tool's job — decide it where it belongs\n\nWhether a write should happen is your decision, or the account's. The tool does\nnot gate it — there is no read-only switch and no approval prompt to configure.\nThe two right places to control read vs write:\n\n- **The account you connect with.** Give it a ceph-mgr Dashboard account with a\n  read-only role. A write then fails at the mgr, which is the only place the\n  permission actually lives — no skill-side flag can be argued around by a\n  model, but a revoked permission cannot be.\n- **Your agent's system prompt.** If you want an observe-only session, tell the\n  model not to call the write tools (they are clearly tagged `[WRITE]`).\n\nWhat the tool *does* guarantee is that you can always see what happened:\n\n## What the tool enforces — do not waste prompt budget on these\n\n| You might be tempted to prompt | Why you don't need to |\n|---|---|\n| \"Log everything you do, over both MCP and the CLI\" | Every call is audited to `~/.ceph-aiops/audit.db` regardless of what the model says it did — and the CLI writes the same row the MCP path does, so there is no unaudited entry point. Reversible writes also record an undo token capturing the *prior* state. |\n| \"Don't invent a value when a field is missing\" | A field the Dashboard did not return comes back as `null`, never as `\"\"`. A missing `deviceClass`, `host`, MDS `state`, or `pg_autoscale_mode` is distinguishable from an empty one in the payload. |\n| \"Tell me if the output was cut off\" | `pg_dump_stuck` returns `{\"stuck\": [...], \"returned\": N, \"limit\": L, \"truncated\": true/false}` and `pg_summary` the same shape under `unhealthy` (plus `states`) — the list key differs, so look it up per tool rather than expecting `stuck` everywhere. Truncation is measured (one extra row is collected), not guessed. `pg_summary` also keeps `unhealthyCount` as the true total even when the list is capped. |\n| \"Explain what HEALTH_WARN means\" | `cluster_health` already folds each active check code (`PG_DEGRADED`, `OSD_NEARFULL`, `SLOW_OPS`, `LARGE_OMAP_OBJECTS`, …) into a plain-language `cause` and `suggestedAction`. The model should quote those, not compose its own. |\n| \"Confirm before anything destructive\" | Every destructive tool (`osd_purge`, `osd_mark_out`, `pool_delete`, `rbd_image_delete`, `rbd_snapshot_delete`, `set_pool_size`) takes `dry_run=True` for a preview and is `risk="},{"path":"references/capabilities.md","content":"# ceph-aiops capabilities\n\n> 37 MCP tools (17 read, 18 write, 2 undo) over the **ceph-mgr\n> Dashboard REST API** (`https://<host>:8443`, JWT via `POST /api/auth`).\n> Multi-node rebalance behaviour and the write ops need live verification\n> (see `docs/VERIFICATION.md`).\n\n## Read tools (17)\n\n| Tool | API path | Returns |\n|------|----------|---------|\n| `cluster_health` | `GET /api/health/full` | **flagship RCA** — per active HEALTH_WARN/ERR check: code, plain-language meaning, likely cause, suggested action |\n| `cluster_status` | `GET /api/health/minimal` | `ceph -s` summary: health status, mon/mgr/osd/pg counts |\n| `osd_tree` | `GET /api/osd` | OSD tree: up/in, CRUSH weight, host, device class |\n| `osd_df` | `GET /api/osd` | per-OSD utilization %, **most-full first**, near-full / backfill-full flags |\n| `osd_perf` | `GET /api/osd` | commit/apply latency per OSD, **slowest first** |\n| `pg_summary` | `GET /api/pg` (+ `/api/health/full`) | PG **state histogram** + list of non-active+clean PGs |\n| `pg_dump_stuck` | `GET /api/pg` | stuck PGs (inactive/unclean/stale/undersized) + implicated OSDs |\n| `scrub_status` | `GET /api/health/full` | PGs overdue for scrub / deep-scrub |\n| `pool_ls` | `GET /api/pool` | pools: name, id, size, pg_num, autoscale mode, application |\n| `pool_df` | `GET /api/pool` | per-pool usage; **usable capacity = raw ÷ size** |\n| `rbd_ls` | `GET /api/block/image` | RBD images (optionally filtered by pool): name, size, pool |\n| `cephfs_status` | `GET /api/cephfs` | MDS ranks + **\"behind on trimming\"** + client count |\n| `rgw_status` | `GET /api/rgw/daemon` + `GET /api/rgw/bucket` | RGW daemons + buckets + **LARGE_OMAP / unsharded-index** findings |\n| `mon_status` | `GET /api/monitor` | monitors: in-quorum vs **out-of-quorum** |\n| `mgr_status` | `GET /api/health/full` | active mgr, standbys, enabled modules |\n| `slow_ops` | `GET /api/health/full` | blocked / slow requests grouped **by OSD** |\n| `capacity_forecast` | `GET /api/osd` (+ df) | raw/used/avail + **days-to-nearfull** projection |\n\n## Write tools (18)\n\n| Tool | Risk | API path | Undo / safety |\n|------|------|----------|---------------|\n| `cluster_flag_set` | medium | `GET`+`PUT /api/osd/flags` | set/unset noout/noscrub/nobackfill/norecover; captures prior flag set (undo) |\n| `osd_reweight` | medium | `POST /api/osd/{id}/reweight` | 0.0 = drain; captures prior weight (undo) |\n| `osd_mark_in` | medium | `POST /api/osd/{id}/mark` | captures prior up/in state (undo) |\n| `osd_mark_out` | **high** | `POST /api/osd/{id}/mark` | drains data; CLI double-confirm + dry-run; captures prior state |\n| `osd_purge` | **high** | `DELETE /api/osd/{id}` | destroy + crush rm + auth del; **irreversible**; dry-run + double-confirm |\n| `trigger_scrub` | medium | `POST /api/pg/{pgid}/scrub` | schedule a shallow scrub; no prior state |\n| `trigger_deep_scrub` | medium | `POST /api/pg/{pgid}/deep_scrub` | schedule a deep (data-integrity) scrub |\n| `set_pool_quota` | medium | `PUT /api/pool/{name}` | "},{"path":"references/cli-reference.md","content":"# ceph-aiops CLI reference\n\n> The CLI is a convenience subset; the full 37-tool surface\n> is via the MCP server (`ceph-aiops mcp`). Talks to the ceph-mgr Dashboard REST\n> API (`https://<host>:8443`, JWT via `POST /api/auth`).\n\n## Setup & diagnostics\n\n```bash\nceph-aiops init                      # interactive onboarding wizard\nceph-aiops doctor [--skip-auth]      # config + secret store + JWT login + mgr-dashboard reachability\nceph-aiops mcp                       # start the MCP server (stdio transport)\n```\n\n## Secrets (encrypted store ~/.ceph-aiops/secrets.enc)\n\n```bash\nceph-aiops secret set <target> [--value <password>]  # store Dashboard password (hidden prompt if no --value)\nceph-aiops secret list                               # names only — values never shown\nceph-aiops secret rm <target>\nceph-aiops secret migrate                            # import legacy plaintext .env (CEPH_<T>_PASSWORD)\nceph-aiops secret rotate-password                    # re-encrypt under a new master password\n```\n\n## Read commands\n\n```bash\nceph-aiops overview [--target <t>]        # HEALTH status + active checks + OSD up/in\nceph-aiops health detail                  # decode active HEALTH_WARN/ERR checks → cause + action (RCA)\nceph-aiops health status                  # ceph -s summary\nceph-aiops osd tree                        # OSD tree: up/in, weight, host, device class\nceph-aiops osd df                          # per-OSD utilization, most-full first, near/backfill-full flags\n```\n\n## Write commands (governed; risk tier in parentheses)\n\n```bash\nceph-aiops osd reweight <osd_id> <weight> [--dry-run]   # (med) 0.0 = drain; reversible → prior weight\nceph-aiops osd out <osd_id> [--dry-run]                 # (high) mark out — drains data; double confirm\nceph-aiops osd purge <osd_id> [--dry-run]               # (high) purge — irreversible; double confirm\n```\n\nThe remaining writes (cluster flags, pool quota/pg_num/autoscale/size/create/delete,\nRBD image/snapshot create/delete, trigger scrubs, throttle recovery) are exposed\nthrough the **MCP server**, not the CLI.\n\n## Common options\n\n- `--target, -t <name>` — target name from `config.yaml` (omit to use the default/first target)\n- `--dry-run` — print the API call that would be made, change nothing\n- Destructive commands (`osd out`, `osd purge`) require `--dry-run` review + double confirmation\n- `doctor --skip-auth` — skip the JWT login / connectivity check (config + secret-store checks only)\n\n## Approver env vars (high-risk ops)\n\n```bash\nexport CEPH_AUDIT_APPROVED_BY='you@example.com'\nexport CEPH_AUDIT_RATIONALE='draining failed OSD 7 per ticket OPS-123'\n```"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":2319,"uniquenessScore":37,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T21:32:55.609Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T21:32:55.609Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T07:41:33.610Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}