{"id":"d9f37c1a-7946-4594-b6c4-4fe72c4df1ad","entityType":"agent","slug":"clawhub-zw008-vmware-aria","name":"vmware-aria","canonicalUrl":"https://www.xpersona.co/agent/clawhub-zw008-vmware-aria","canonicalPath":"/agent/clawhub-zw008-vmware-aria","generatedAt":"2026-10-09T16:27:51.275Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-09T04:21:42.936Z","emptyReason":null},"description":"Use this skill whenever the user needs VMware Aria Operations (VMware VCF Operations in VCF 9+) data — metrics, alerts, capacity, anomalies, reports. Directly handles: resource metrics plus metric key/property/relationship lookup, list/acknowledge/cancel alerts with notes and recommendations, alert definitions, capacity forecasts, anomalies, reports, resource maintenance mode, Aria's own node health and adapter collection state. Always use this skill for \"check vSphere capacity\", \"what Aria Operations alerts are active\", \"show VMware anomalies\", \"generate an Aria report\", \"rightsizing recommendations\", \"VCF Operations alerts\", \"put this host in Aria maintenance mode\", \"is Aria Operations still collecting from vCenter\", \"what does Aria recommend for this alert\", or any Aria Operations / VCF Operations / vRealize Operations task. Do NOT use for real-time vCenter alarms/events (use vmware-monitor), VM operations (use vmware-aiops), or NSX networking (use vmware-nsx). For load balancing/AVI/AKO use vmware-avi. Skill: vmware-aria Owner: zw008 Summary: Use this skill whenever the user needs VMware Aria Operations (VMware VCF Operations in VCF 9+) data — metrics, alerts, capacity, anomalies, reports. Directly handles: resource metrics plus metric key/property/relationship lookup, list/acknowledge/cancel alerts with notes and recommendations, alert definitions, capacity forecasts, anomalies, reports, resource maintenance mode,","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 5.1K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s171xgnmqse0nqvgqvqnaq5f9183kyre:vmware-aria","sourceUrl":"https://clawhub.ai/zw008/vmware-aria","homepage":"https://clawhub.ai/zw008/skills/vmware-aria","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/zw008/vmware-aria","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/zw008/skills/vmware-aria","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":48,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Use this skill whenever the user needs VMware Aria Operations (VMware VCF Operations in VCF 9+) data — metrics, alerts, capacity, anomalies, reports. Directly h"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T04:21:42.936Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T04:21:42.936Z","emptyReason":null},"stars":null,"forks":null,"downloads":5110,"packageName":null,"latestVersion":"1.17.0","tractionLabel":"5.1K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T04:21:42.935Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T04:21:42.936Z","lastCrawledAt":"2026-10-09T04:21:42.935Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T04:21:42.935Z","lastVerifiedAt":null,"highlights":[{"version":"1.17.0","createdAt":"2026-09-20T14:51:48.066Z","changelog":"MCP instructions now name the configured targets and how to choose one; a config that cannot be read says so instead of falling silent.","fileCount":10,"zipByteSize":51070},{"version":"1.16.0","createdAt":"2026-09-19T03:55:15.371Z","changelog":"Destructive MCP tools preview by default (confirm=False) and state their blast radius; confirm=True refuses on blockers or unreadable measurements. Requires vmware-policy>=1.17.0.","fileCount":10,"zipByteSize":51087},{"version":"1.15.0","createdAt":"2026-09-15T08:57:13.291Z","changelog":"Service state beside the badge, VC_APP alert symptoms name the down services, resource metrics --summary, pasteable alert ids and ISO times, UUID-validated lowercase ids, complete symptom paging, cheaper writes.","fileCount":10,"zipByteSize":50731},{"version":"1.14.1","createdAt":"2026-09-15T06:01:07.769Z","changelog":"CLI reads are audited under their MCP tool names; every CLI command declares what it reaches (needs vmware-policy 1.15.0)","fileCount":10,"zipByteSize":48793},{"version":"1.14.0","createdAt":"2026-09-15T03:13:17.222Z","changelog":"resource list shows which objects stopped reporting; rightsizing says whether a recommendation has settled","fileCount":10,"zipByteSize":48931},{"version":"1.13.0","createdAt":"2026-09-13T09:22:07.450Z","changelog":"11 new tools (44): metric keys/properties/relationships, Aria node self-check and adapter collection state, resource maintenance mode, alert notes and recommendations; ops playbooks; maintenance reads UNKNOWN as unknown.","fileCount":10,"zipByteSize":47266},{"version":"1.12.0","createdAt":"2026-09-13T07:37:49.683Z","changelog":"Live Aria Operations 8.18.7 fixes: alerts and symptoms named, rightsizing units/direction/actionable, health DEGRADED vs DOWN with version, metrics say why a key is missing (breaking: get_resource_metrics returns metrics + missing), top consumers report the ranked average.","fileCount":9,"zipByteSize":35213},{"version":"1.11.0","createdAt":"2026-09-12T00:15:46.291Z","changelog":"The TLS hint points at SSL_CERT_FILE with a PEM bundle of the public roots plus your own CA instead of telling you to disable verification. CLI writes are authorised and audited under their MCP tool names.","fileCount":9,"zipByteSize":29666}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s171xgnmqse0nqvgqvqnaq5f9183kyre:vmware-aria","setupComplexity":"low","setupSteps":["Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zw008-vmware-aria/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zw008-vmware-aria/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zw008-vmware-aria/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-zw008-vmware-aria/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-zw008-vmware-aria/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-zw008-vmware-aria/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-09T16:27:51.268Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zw008-vmware-aria/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zw008-vmware-aria/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zw008-vmware-aria/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zw008-vmware-aria/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-09T04:21:42.936Z","emptyReason":null},"readme":"Skill: vmware-aria\n\nOwner: zw008\n\nSummary: Use this skill whenever the user needs VMware Aria Operations (VMware VCF Operations in VCF 9+) data — metrics, alerts, capacity, anomalies, reports. Directly handles: resource metrics plus metric key/property/relationship lookup, list/acknowledge/cancel alerts with notes and recommendations, alert definitions, capacity forecasts, anomalies, reports, resource maintenance mode, Aria's own node health and adapter collection state. Always use this skill for \"check vSphere capacity\", \"what Aria Operations alerts are active\", \"show VMware anomalies\", \"generate an Aria report\", \"rightsizing recommendations\", \"VCF Operations alerts\", \"put this host in Aria maintenance mode\", \"is Aria Operations still collecting from vCenter\", \"what does Aria recommend for this alert\", or any Aria Operations / VCF Operations / vRealize Operations task. Do NOT use for real-time vCenter alarms/events (use vmware-monitor), VM operations (use vmware-aiops), or NSX networking (use vmware-nsx). For load balancing/AVI/AKO use vmware-avi.\n\nTags: latest:1.17.0\n\nVersion history:\n\nv1.17.0 | 2026-09-20T14:51:48.066Z | user\n\nMCP instructions now name the configured targets and how to choose one; a config that cannot be read says so instead of falling silent.\n\nv1.16.0 | 2026-09-19T03:55:15.371Z | user\n\nDestructive MCP tools preview by default (confirm=False) and state their blast radius; confirm=True refuses on blockers or unreadable measurements. Requires vmware-policy>=1.17.0.\n\nv1.15.0 | 2026-09-15T08:57:13.291Z | user\n\nService state beside the badge, VC_APP alert symptoms name the down services, resource metrics --summary, pasteable alert ids and ISO times, UUID-validated lowercase ids, complete symptom paging, cheaper writes.\n\nv1.14.1 | 2026-09-15T06:01:07.769Z | user\n\nCLI reads are audited under their MCP tool names; every CLI command declares what it reaches (needs vmware-policy 1.15.0)\n\nv1.14.0 | 2026-09-15T03:13:17.222Z | user\n\nresource list shows which objects stopped reporting; rightsizing says whether a recommendation has settled\n\nv1.13.0 | 2026-09-13T09:22:07.450Z | user\n\n11 new tools (44): metric keys/properties/relationships, Aria node self-check and adapter collection state, resource maintenance mode, alert notes and recommendations; ops playbooks; maintenance reads UNKNOWN as unknown.\n\nv1.12.0 | 2026-09-13T07:37:49.683Z | user\n\nLive Aria Operations 8.18.7 fixes: alerts and symptoms named, rightsizing units/direction/actionable, health DEGRADED vs DOWN with version, metrics say why a key is missing (breaking: get_resource_metrics returns metrics + missing), top consumers report the ranked average.\n\nv1.11.0 | 2026-09-12T00:15:46.291Z | user\n\nThe TLS hint points at SSL_CERT_FILE with a PEM bundle of the public roots plus your own CA instead of telling you to disable verification. CLI writes are authorised and audited under their MCP tool names.\n\nv1.10.0 | 2026-09-07T12:30:49.643Z | user\n\na zero from the capacity engine is not a size\n\nv1.9.1 | 2026-09-02T14:42:27.662Z | user\n\ntwo prompts before an irreversible report delete\n\nv1.9.0 | 2026-08-31T07:23:30.778Z | user\n\ntwo report fields that were not what their names said\n\nv1.8.15 | 2026-08-31T00:33:22.085Z | user\n\nfix: run the suite on a non-UTF-8 machine, and stop one skill answering for another\n\nv1.8.14 | 2026-08-30T15:18:20.747Z | user\n\nSecond-round fixes from the 2026-08-30 VCF 9.1 re-test; vmware-policy floor raised to 1.11.0 (the engine no longer fails open when rules.yaml cannot be read).\n\nv1.8.13 | 2026-08-30T09:34:34.556Z | user\n\nParameter descriptions now reach the MCP JSON schema (0% -> 100% coverage); additionalProperties closed; vmware-policy floor raised to 1.10.0.\n\nv1.8.12 | 2026-08-30T07:51:32.674Z | user\n\nfleet_domain_list and get_alert no longer render real data as empty; the 500 clamp is a page size with next_offset instead of a ceiling hiding 2,283 alerts; verify_ssl:false no longer needs an undeclared urllib3; doctor reads the config the tools read.\n\nv1.8.11 | 2026-08-28T02:53:57.017Z | user\n\nFixes the server's self-reported version and the advertised tool count; adds a Claude Code plugin manifest.\n\nv1.8.10 | 2026-08-06T09:43:43.112Z | user\n\nVCF Operations 9.1 fleet (certs/passwords/domains) + diagnostic findings + real-time PromQL — 5 read tools (28→33). + Fable5-review hardening. Beta: paths verified, field names defensive pending live appliance.\n\nv1.8.9 | 2026-08-01T03:10:14.566Z | user\n\nMoved to vmware-skills GitHub org; MCP Registry namespace → io.github.vmware-skills. Links updated.\n\nv1.8.8 | 2026-07-21T15:44:55.109Z | user\n\nCLI writes now route through the shared guard()+audit_call() core via @guarded, exactly like the MCP tools (HLD I-1/I-8). Requires vmware-policy>=1.8.8.\n\nv1.8.7 | 2026-07-21T11:41:33.762Z | user\n\nRemove read-only switch and approval tiers; read/write authz delegated to RBAC. Plus accumulated fixes since 1.8.5.\n\nv1.8.5 | 2026-07-20T13:05:38.816Z | user\n\nA failure that is returned is now audited as a failure, and certificate/URL detail no longer reaches the agent. Both fixes v1.8.4 announced were incomplete.\n\nv1.8.4 | 2026-07-20T08:26:38.203Z | user\n\nTeaching error messages, domain exceptions no longer redacted on the way to the agent, and tool descriptions that state when to use each tool and what to call next.\n\nv1.8.3 | 2026-07-20T03:44:45.580Z | user\n\nPer-target username can now come from an env var, resolved per access like the password; documented credential variables corrected against what each repo's code actually reads\n\nv1.8.2 | 2026-07-19T18:07:59.279Z | user\n\nMCP server moved into the package namespace — fixes two skills in one environment silently overwriting each other's server; agent-guardrails.md for local/small models now ships in every skill\n\nv1.8.1 | 2026-07-19T11:29:30.268Z | user\n\nRead-only mode now documented on every surface that teaches it (SKILL.md, setup-guide, capabilities) and reported by doctor\n\nv1.8.0 | 2026-07-19T09:44:19.753Z | user\n\nRead-only mode (7 write tools withheld), list-result envelope, declared environments; tool count corrected to 28\n\nv1.7.5 | 2026-07-13T07:17:14.467Z | user\n\ndead-code cleanup; family version alignment\n\nv1.7.4 | 2026-07-13T04:52:31.091Z | user\n\nFamily version alignment to 1.7.4 (substantive change this cycle is in vmware-monitor: host-check boundary read batching).\n\nv1.7.3 | 2026-07-03T00:48:50.086Z | user\n\nFamily version alignment (v1.7.3)\n\nv1.7.2 | 2026-07-02T14:31:21.455Z | user\n\nBulk stats query for anomaly/rightsizing + complete pagination\n\nv1.7.1 | 2026-07-02T10:52:43.707Z | user\n\nFamily version alignment with v1.7.1 (AIops/Monitor large-inventory scale fix, issue #31).\n\nv1.7.0 | 2026-06-27T01:01:29.467Z | user\n\nguided init wizard + .env.example + auth teaching\n\nv1.6.1 | 2026-06-24T00:01:09.108Z | user\n\nv1.6.1 .env password b64 obfuscation\n\nv1.6.0 | 2026-06-22T09:17:35.753Z | user\n\nv1.6.0 trust architecture: undo tokens + governance harness (budget/audit/risk-tiers)\n\nv1.5.39 | 2026-06-22T00:42:28.428Z | user\n\nv1.5.39: AIops snapshot-delete async + honest timeout (token-burn fix), Storage browse timeout fix; others version-aligned\n\nv1.5.38 | 2026-06-12T06:59:02.961Z | user\n\nbacklog finish: pagination bug fix, server split\n\nv1.5.37 | 2026-06-12T01:57:59.659Z | user\n\nbacklog: liveness caching, error-hint completeness\n\nv1.5.36 | 2026-06-11T23:21:29.536Z | user\n\nerror-translation completeness + parity tests\n\nv1.5.35 | 2026-06-10T00:44:38.459Z | user\n\nSecurity hardening: safe error handling, TLS/path/permission fixes\n\nv1.5.34 | 2026-06-09T15:02:38.533Z | user\n\nteaching errors instead of tracebacks for every suite-api call (#6 follow-up)\n\nv1.5.33 | 2026-06-09T14:25:21.972Z | user\n\nhealth status survives a 503 from an offline node (#6)\n\nv1.5.32 | 2026-06-08T02:47:42.299Z | user\n\nv1.5.32: 2nd-pass spec audit — auth header, Alert model fields, capacity statKeys\n\nv1.5.31 | 2026-06-07T23:28:32.285Z | user\n\nv1.5.31: API layer rewritten against official suite-api spec — 12 user-reported bugs + 6 invented endpoints fixed; spec-conformance regression added\n\nv1.5.30 | 2026-06-07T13:23:35.276Z | user\n\nv1.5.30: Glama TDQS tool description quality rewrite\n\nv1.5.29 | 2026-05-29T02:19:32.461Z | user\n\nFamily version alignment (no Aria-specific changes since v1.5.28)\n\nv1.5.28 | 2026-05-20T10:00:29.473Z | user\n\nFix subclass() arg 1 must be a class in goose/old-mcp environments. v1.5.25-1.5.27 only addressed PEP 604 X|None -> Optional[X] but kept 'from __future__ import annotations'; under mcp 1.10-1.13 FastMCP's issubclass() on string annotations crashed server load. This release removes the future import. CLAUDE.md pitfall #33 updated.\n\nv1.5.27 | 2026-05-20T06:56:45.228Z | user\n\nLoosen Python requirement to >= 3.10 (was >=3.11). v1.5.25/26 PEP 604 fix already enables 3.10 at runtime; this release lifts pip download/install block.\n\nv1.5.26 | 2026-05-20T06:17:02.228Z | user\n\nMCP server Python 3.10 compatibility (踩坑 #33): PEP 604 X|None → Optional[X] in tool signatures; mcp_cmd Python version guard; mcp[cli]>=1.10\n\nv1.5.23 | 2026-05-19T02:58:36.140Z | user\n\nVCF 9.0 / 9.1 compatibility declared. README version-compat tables updated. Added Official Broadcom References (VCF Python SDK, REST APIs, CLI tools).\n\nv1.5.22 | 2026-05-08T23:25:18.101Z | user\n\nv1.5.22 family alignment for Smithery rollout\n\nArchive index:\n\nArchive v1.17.0: 10 files, 51070 bytes\n\nFiles: evals/evals.json (1349b), references/agent-guardrails.md (3888b), references/capabilities.md (28727b), references/cli-reference.md (40401b), references/investigation-protocol.md (6850b), references/ops-playbooks.md (8762b), references/setup-guide.md (10238b), skill-card.md (2898b), SKILL.md (24468b), _meta.json (131b)\n\nFile v1.17.0:SKILL.md\n\n---\nname: vmware-aria\ndescription: >\n  Use this skill whenever the user needs VMware Aria Operations (VMware VCF Operations in VCF 9+) data — metrics, alerts, capacity, anomalies, reports.\n  Directly handles: resource metrics plus metric key/property/relationship lookup, list/acknowledge/cancel alerts with notes and recommendations, alert definitions, capacity forecasts, anomalies, reports, resource maintenance mode, Aria's own node health and adapter collection state.\n  Always use this skill for \"check vSphere capacity\", \"what Aria Operations alerts are active\", \"show VMware anomalies\", \"generate an Aria report\", \"rightsizing recommendations\", \"VCF Operations alerts\", \"put this host in Aria maintenance mode\", \"is Aria Operations still collecting from vCenter\", \"what does Aria recommend for this alert\", or any Aria Operations / VCF Operations / vRealize Operations task.\n  Do NOT use for real-time vCenter alarms/events (use vmware-monitor), VM operations (use vmware-aiops), or NSX networking (use vmware-nsx).\n  For load balancing/AVI/AKO use vmware-avi.\ninstaller:\n  kind: uv\n  package: vmware-aria\nallowed-tools:\n  - Bash\nmetadata: {\"openclaw\":{\"requires\":{\"anyBins\":[\"vmware-aria\",\"uvx\"]},\"optional\":{\"env\":[\"VMWARE_ARIA_CONFIG\",\"VMWARE_ARIA_<TARGET>_PASSWORD\",\"VMWARE_ARIA_<TARGET>_USERNAME\",\"VMWARE_AUDIT_APPROVED_BY\"],\"bins\":[\"vmware-policy\"]},\"homepage\":\"https://github.com/vmware-skills/VMware-Aria\",\"emoji\":\"📊\",\"os\":[\"macos\",\"linux\"]}}\ncompatibility: >\n  vmware-policy auto-installed as Python dependency (provides @vmware_tool decorator and audit logging). All write operations audited to ~/.vmware/audit.db.\n  Credentials: Each Aria Operations target requires a per-target password env var in ~/.vmware-aria/.env following the pattern VMWARE_ARIA_<TARGET_NAME_UPPER>_PASSWORD. Passwords are never logged or echoed.\n  Read-heavy: 34 of 44 tools are read-only. Write operations limited to alert acknowledge/cancel, alert notes, alert definition management, report management, and resource maintenance start/end.\n  No webhooks, no outbound network calls, no guest operations. Local only: stdio MCP + Aria Operations REST API (HTTPS 443).\n  Transitive dependencies: Only vmware-policy (audit/policy). No post-install scripts or background services.\n---\n\n# VMware Aria Operations\n\n> **Disclaimer**: This is a community-maintained open-source project and is **not affiliated with, endorsed by, or sponsored by VMware, Inc. or Broadcom Inc.** \"VMware\" and \"Aria\" are trademarks of Broadcom. Source code is publicly auditable at [github.com/vmware-skills/VMware-Aria](https://github.com/vmware-skills/VMware-Aria) under the MIT license.\n\nVMware Aria Operations (vRealize Operations 8.x, VCF Operations 9.x) AI-assisted monitoring — 44 MCP tools for resources (metric keys, properties, relationships), alerts (notes, recommendations), alert definitions, capacity planning, anomaly detection, report automation, resource maintenance mode, platform health (Aria node memory, adapter collection), and VCF 9.1 fleet certificates/passwords/domains, diagnostic findings, and real-time PromQL metrics.\n\n> **Know the version first**: `vmware-aria health status` (MCP `get_aria_health`) names the product line. Fleet, findings and PromQL exist only on VCF Operations 9.0+ (PromQL uses the 9.1 VODAP service); on 8.x they return a \"requires 9.0 or newer\" error, not data.\n> **Companion skills**: vmware-monitor (real-time vSphere), vmware-aiops (VM lifecycle), vmware-nsx (networking), vmware-avi (AVI/ALB/AKO), vmware-harden (compliance), vmware-pilot (approval workflows), vmware-policy (audit/policy).\n\n## What This Skill Does\n\n| Category | Tools | Count |\n|----------|-------|:-----:|\n| **Resources** | list, get details, metrics, health badge, top consumers, metric keys, properties, relationships | 8 |\n| **Alerts** | list, get details, investigate (alert→resource), acknowledge, cancel, list definitions, list/add notes, recommendations | 9 |\n| **Alert Definitions** | list symptoms, create definition, enable/disable, delete | 4 |\n| **Capacity** | cluster overview, remaining capacity, time remaining, rightsizing | 4 |\n| **Reports** | list templates, generate, list, get status+download URL, delete | 5 |\n| **Anomaly** | list anomalies, risk badge | 2 |\n| **Health** | Aria platform health, collector group status, Aria node memory/swap/heap, adapter collection state | 4 |\n| **Maintenance** | start / end resource maintenance, list maintenance schedules | 3 |\n| **Fleet / PromQL** (VCF Ops 9.1) | fleet certificates, password accounts, VCF domains, diagnostic findings, real-time PromQL query | 5 |\n\n**Total**: 44 tools (34 read-only + 10 write)\n\n## Quick Install\n\n```bash\nuv tool install vmware-aria==1.17.0\nvmware-aria init      # guided setup: writes config + .env (chmod 600, password grep-safe), then verifies\nvmware-aria doctor\n```\n\n## When to Use This Skill\n\n- **Lookup**: which metric keys a resource reports (name, unit) before querying them, its properties, its parents and children\n- **Performance**: VM contention (CPU Ready, balloon, swap), time-series metrics, top consumers, anomaly counts and risk badges\n- **Alerts**: list, investigate, acknowledge or cancel alerts; read or add notes (who is handling it); read the alert's prioritized recommendations; list, create, enable/disable or delete alert definitions (post-RCA)\n- **Capacity**: cluster headroom, time until full, VM rightsizing\n- **Reports**: generate, poll, download and delete reports\n- **Maintenance**: put a resource in maintenance before planned work (timed or until ended), end it, list maintenance schedules\n- **Platform**: is Aria Operations itself healthy (DEGRADED vs DOWN, which service), which version and product line, collector groups, whether the Aria node is short of memory, which adapter stopped collecting\n\nFor VM changes, NSX, vSphere alarms, storage or load balancing, route with the table below.\n\n## Related Skills — Skill Routing\n\n| User Intent | Recommended Skill |\n|-------------|-------------------|\n| Aria Operations monitoring, alerts, capacity | **vmware-aria** ← this skill |\n| VM lifecycle, deployment, guest ops | **vmware-aiops** |\n| NSX networking: segments, gateways, NAT, routing | **vmware-nsx** |\n| Read-only vSphere inventory, events, alarms | **vmware-monitor** |\n| Storage: iSCSI, vSAN, datastores | **vmware-storage** |\n| Multi-step workflows with approval | **vmware-pilot** |\n| Compliance baselines (CIS / 等保 / PCI-DSS), drift detection, LLM remediation advisor | **vmware-harden** (`uv tool install vmware-harden`) |\n| Load balancer, AVI, ALB, AKO, Ingress | **vmware-avi** (`uv tool install vmware-avi`) |\n| Audit log query | **vmware-policy** (`vmware-audit` CLI) |\n\n## Common Workflows\n\n> **Troubleshooting paths**: step-by-step playbooks for a DEGRADED platform, alert triage, VM contention, empty metrics, pre-resize checks and planned maintenance — [`references/ops-playbooks.md`](references/ops-playbooks.md).\n>\n> **Diagnostic investigations**: Before running any \"why is X slow / failing / down\" workflow, follow [`references/investigation-protocol.md`](references/investigation-protocol.md). It enforces the four root-cause completeness criteria (falsifiability / sufficiency / necessity / mechanism) and the up-to-three-rounds deepening loop. Stopping at a partial conclusion is an anti-pattern — always self-check against the criteria before outputting a report.\n\n### Daily VM Health Check (Proactive Ops)\n\n**Judgment**: don't chase the highest CPU consumer — chase the highest **contention** consumer. A VM at 90% CPU on a quiet host is healthy; a VM at 30% CPU but 15% Ready is starving. Key metrics: CPU Ready, Memory Balloon, Disk Latency.\n\n1. Find top CPU consumers → `vmware-aria resource top --metric 'cpu|usage_average' --top 20` (this is the **starting set**, not the answer)\n2. Check CPU Ready on hot VMs → `vmware-aria resource metrics <vm-id> --metrics 'cpu|readyPct' --hours 24`\n   - >5% = warning, >10% = problem, >20% = critical\n3. Check memory pressure → `vmware-aria resource metrics <vm-id> --metrics 'mem|balloonPct,mem|swapped_average' --hours 24`\n   - Balloon >0 = ESXi reclaiming memory; Swap >0 = severe — act immediately\n   - If a key comes back under `missing` instead of `metrics`, it is not a zero: `not_collected_for_resource` means a wrong key for this resource (use `similar_keys`), `no_data_in_window` means widen `--hours`\n4. List active CRITICAL/IMMEDIATE alerts → `vmware-aria alert list --criticality CRITICAL`\n5. Check anomaly counts → `vmware-aria anomaly list`\n6. Cross-validate against the [investigation protocol](references/investigation-protocol.md) before reporting any \"root cause\" — high consumption is rarely the root, usually a downstream symptom\n\n### Capacity Planning\n\n1. List clusters → `vmware-aria resource list --kind ClusterComputeResource`\n2. Get remaining capacity → `vmware-aria capacity remaining <cluster-id>`\n3. Predict time until full → `vmware-aria capacity time-remaining <cluster-id>`\n4. Get capacity overview → `vmware-aria capacity overview <cluster-id>`\n5. Find rightsizing candidates → `vmware-aria capacity rightsizing` — act only on rows with `Act. yes`; read each VM's caveats and the vendor minimum size before reducing\n   - If a yellow `properties_note` prints under the table, the VM property read failed: power state and current size are unknown and no row is actionable — retry, do not resize from it\n\n### Post-Incident: Create Detection Alert (RCA Follow-up)\n\nAfter resolving an incident, create an early-warning alert to prevent recurrence. Alert definition management is **MCP-only** (no CLI subcommands):\n\n1. Find matching symptom definitions → MCP `list_symptom_definitions` (filter by `name_filter` / `resource_kind`)\n2. Create alert definition referencing symptoms → MCP `create_alert_definition` with name, resource_kind, symptom_definition_ids, criticality (any one symptom firing triggers the alert)\n3. Verify it appears in definitions → `vmware-aria alert definitions --name \"Gold VM CPU\"` (criticality shown is the max severity across the definition's states)\n4. Enable or disable later → MCP `set_alert_definition_state`\n\n### Generate Capacity Report\n\n1. Find report template → `vmware-aria report definitions --name \"Capacity\"`\n2. Trigger report generation → `vmware-aria report generate <definition-id> --resources <resource-id>` (the Report API requires at least one resource UUID)\n3. Poll until completed → `vmware-aria report get <report-id>` (repeat until `status == COMPLETED`)\n4. Download via the returned `download_url` (PDF) or `csv_url`\n5. Clean up → `vmware-aria report delete <report-id>`\n\n## Usage Mode\n\n| Scenario | Recommended | Why |\n|----------|:-----------:|-----|\n| Local/small models (Ollama, Qwen) | **CLI** | ~2K tokens vs ~8K for MCP |\n| Cloud models (Claude, GPT-4o) | Either | MCP gives structured JSON I/O |\n| Automated pipelines | **MCP** | Type-safe parameters, structured output |\n\nRunning vmware-aria with a local or small model? See [`references/agent-guardrails.md`](references/agent-guardrails.md) for tool-calling guardrails (alert-to-resource correlation and Aria data fidelity).\n\nEvery command accepts `--target <name>` (every MCP tool `target`) to pick the Aria Operations instance.\n\n## MCP Tools (44 — 34 read, 10 write)\n\nAll MCP tools accept an optional `target`. The six destructive writes take `confirm`: without `confirm=true` they return a `blast_radius` and change nothing — show it to the user first.\n\n| Category | Tool | Type | Description |\n|----------|------|:----:|-------------|\n| Resource | `list_resources` | Read | List VMs, hosts, clusters by resource kind (or `all`); `collection_status` finds objects not receiving data |\n| | `get_resource` | Read | Get resource details with health, risk, efficiency badges |\n| | `get_resource_metrics` | Read | Time-series stats; `summary=true` gives n/min/max/avg/latest + change points; `missing` explains empty keys |\n| | `get_resource_health` | Read | Badge scores; service objects add `service.available` (a badge is not service state) |\n| | `get_top_consumers` | Read | Rank by last-hour average `value` (`latest_value` = newest point) |\n| | `list_metric_keys` | Read | Keys a resource reports with name/unit and `definition`, or a kind's defined keys — look up before querying |\n| | `get_resource_properties` | Read | Current property values (power state, parent host, extraConfig) |\n| | `get_resource_relationships` | Read | Related resources with `direction`; `relationship_type` ALL / PARENT / CHILD |\n| Alerts | `list_alerts` | Read | List active alerts with criticality, resource ID, name and kind (`resource_name: null` = unknown, see `resource_names_note`) |\n| | `get_alert` | Read | Alert details: named symptoms, the object each is on (e.g. the down service), ISO-8601 times |\n| | `investigate_alert` | Read | Resolve an alert to its confirmed affected resource in one call — returns both UUIDs explicitly labelled plus the vmware-monitor handoff |\n| | `acknowledge_alert` | **Write** | Mark an alert as acknowledged (does not close it) |\n| | `cancel_alert` | **Write** | Cancel (dismiss) an active alert |\n| | `list_alert_definitions` | Read | List alert templates configured in Aria Ops |\n| | `list_alert_notes` | Read | Notes on an alert (who is handling it, what was done) |\n| | `add_alert_note` | **Write** | Add a note; does not change the alert's status (low risk, not idempotent) |\n| | `get_alert_recommendations` | Read | Prioritized recommendations from the alert's definition; `status` found / partial / none_defined / unknown |\n| Alert Defs | `list_symptom_definitions` | Read | List symptom definitions — use IDs when creating alert defs |\n| | `create_alert_definition` | **Write** | Create new alert definition from symptom definition IDs |\n| | `set_alert_definition_state` | **Write** | Enable or disable an alert definition |\n| | `delete_alert_definition` | **Write** | Delete an alert definition permanently |\n| Capacity | `get_capacity_overview` | Read | Group-level remaining % + per-dimension headroom and days-until-full |\n| | `get_remaining_capacity` | Read | Remaining CPU, memory, disk before hitting limits |\n| | `get_time_remaining` | Read | Days until cluster capacity is exhausted |\n| | `list_rightsizing_recommendations` | Read | Per-VM recommended size (raw MHz/KB/GB; use `recommended_vcpus`), direction, power state, `recommendation_stable` / `recommendation_range` (7-day movement), `actionable`, `caveats`, `properties_note`, `history_note` |\n| Reports | `list_report_definitions` | Read | List available report definition templates |\n| | `generate_report` | **Write** | Trigger report generation (async; returns report_id) |\n| | `list_reports` | Read | List generated reports, optionally by definition |\n| | `get_report` | Read | Poll report status + get PDF/CSV download URLs |\n| | `delete_report` | **Write** | Delete a generated report |\n| Anomaly | `list_anomalies` | Read | Per-resource anomaly counts (System Attributes\\|total_alarms metric) |\n| | `get_resource_riskbadge` | Read | Risk score (0–100): likelihood of future problems |\n| Health | `get_aria_health` | Read | Platform `assessment` (HEALTHY/DEGRADED/DOWN/UNKNOWN), per-service health, product version |\n| | `list_collector_groups` | Read | Collector agents status and connectivity |\n| | `get_aria_node_resources` | Read | Aria node memory/swap/heap and watchdog restarts; memory pressure NORMAL / ELEVATED / HIGH / UNKNOWN |\n| | `list_adapters` | Read | Adapter instances, last collection age, `stale` |\n| Maintenance | `start_resource_maintenance` | **Write** | Timed (`duration_minutes` / `end_time_ms`) or manual maintenance; before/after state; undo = end |\n| | `end_resource_maintenance` | **Write** | End maintenance; refuses a resource not known to be in maintenance |\n| | `list_maintenance_schedules` | Read | Recurring maintenance schedules, optionally for one `resource_id` |\n| Fleet / PromQL (VCF Ops 9.1) | `fleet_certificate_list` | Read | Certificate status/expiry across the VCF fleet |\n| | `fleet_password_account_list` | Read | Managed password-account status (read-only; does not rotate) |\n| | `fleet_domain_list` | Read | SDDC/workload domains behind one registered VCF integration |\n| | `findings_list` | Read | Operations diagnostic findings (not compliance — see vmware-harden) |\n| | `promql_query` | Read | Real-time PromQL instant query via the VODAP service (base path INFERRED, unverified on real hardware) |\n\n**Read/write split**: 34 read-only, 10 write. All write operations are audit-logged to `~/.vmware/audit.db` (via vmware-policy).\n\n### List results are envelopes — read `truncated` before you summarise\n\nList tools return `{items, returned, limit, total, truncated, hint}`, not a bare array. Rows are under `items`; `truncated: true` means more rows exist — never call it the complete set; `total: null` means the API gave no size. Full rules, per-tool `total` sources and `list_anomalies`' scan fields: [`references/capabilities.md`](references/capabilities.md#list-result-envelope).\n\n## CLI Quick Reference\n\n```bash\n# Resources\nvmware-aria resource list [--kind VirtualMachine|HostSystem|ClusterComputeResource|all] [--name <filter>] [--collection-status NO_DATA_RECEIVING]\nvmware-aria resource get <resource-id>\nvmware-aria resource metrics <resource-id> --metrics 'cpu|usage_average,mem|usage_average' --hours 4\nvmware-aria resource metrics <vm-id> --metrics 'cpu|readyPct,mem|balloonPct' --hours 24 --summary\nvmware-aria resource health <resource-id>\nvmware-aria resource top --metric 'cpu|usage_average' --kind VirtualMachine --top 10\nvmware-aria resource keys <resource-id> [--filter 'mem|']   # or --kind VirtualMachine\nvmware-aria resource properties <resource-id> [--name 'summary|']\nvmware-aria resource relationships <resource-id> [--type PARENT]\n\n# Alerts\nvmware-aria alert list [--criticality CRITICAL|IMMEDIATE|WARNING|INFORMATION] [--json]\nvmware-aria alert get <alert-id>\nvmware-aria alert acknowledge <alert-id>\nvmware-aria alert cancel <alert-id>\nvmware-aria alert definitions [--name <filter>]\nvmware-aria alert notes <alert-id>\nvmware-aria alert note-add <alert-id> \"Taking this: rebooting esx-03\"\nvmware-aria alert recommendations <alert-id>\n\n# Alert Definitions: creation/enable/disable/delete and symptom-definition\n# lookup are MCP-only tools (list_symptom_definitions, create_alert_definition,\n# set_alert_definition_state, delete_alert_definition) — no CLI subcommands.\n\n# Capacity\nvmware-aria capacity overview <cluster-id>\nvmware-aria capacity remaining <resource-id>\nvmware-aria capacity time-remaining <resource-id>\nvmware-aria capacity rightsizing [--resource-id <vm-id>]\n\n# Reports (async: generate → poll get → download → delete)\nvmware-aria report definitions [--name <filter>]\nvmware-aria report generate <definition-id> --resources <id1,id2>   # at least one resource UUID required\nvmware-aria report list [--definition-id <id>]\nvmware-aria report get <report-id>        # poll until status == COMPLETED; shows download_url\nvmware-aria report delete <report-id>\n\n# Anomaly\nvmware-aria anomaly list [--resource-id <id>]\nvmware-aria anomaly risk <resource-id>\n\n# Health\nvmware-aria health status\nvmware-aria health collectors\nvmware-aria health node [--hours 24]        # Aria node memory pressure, watchdog restarts\nvmware-aria health adapters [--kind VMWARE] # stale = last collection older than max(3 x interval, 15 min)\n\n# Maintenance (writes ask once; --yes skips, --dry-run prints the API call without connecting)\nvmware-aria maintenance start <resource-id> --duration 60   # neither --duration nor --end = until `maintenance end`\nvmware-aria maintenance end <resource-id>\nvmware-aria maintenance schedules [--resource-id <id>]\n\n# Diagnostics\nvmware-aria doctor [--skip-auth]\n```\n\n### Key Metric Names (for `resource metrics` command)\n\n| Metric | API Key | Unit | What It Means |\n|--------|---------|------|--------------|\n| CPU Ready | `cpu\\|readyPct` | % | vCPU waiting for a physical core; >5% = warning |\n| CPU Usage | `cpu\\|usagemhz_average` | MHz | CPU actually used |\n| CPU Demand | `cpu\\|demandmhz` | MHz | CPU the VM requested |\n| Memory Consumed | `mem\\|consumed_average` | KB | Footprint on host (capacity) |\n| Memory Balloon | `mem\\|balloonPct` | % | **>0 = ESXi reclaiming memory** |\n| Memory Swapped | `mem\\|swapped_average` | KB | **>0 = severe pressure** |\n| Memory Contention | `mem\\|host_contentionPct` | % | Contention for host memory |\n| Disk Throughput | `virtualDisk\\|read_average`, `virtualDisk\\|write_average` | KBps | Read / write rate |\n| Disk Latency | `virtualDisk\\|peak_vDisk_readLatency`, `virtualDisk\\|peak_vDisk_writeLatency` | ms | Highest across the VM's virtual disks |\n| Network | `net\\|received_average`, `net\\|transmitted_average` | KBps | Receive / transmit rate |\n\nVirtualMachine keys and units as defined on Aria Operations 8.18.7; other resource kinds use different keys. Unreported keys come back under `missing`.\n\n> Full CLI reference with all options and output formats: see `references/cli-reference.md`\n\n## Troubleshooting\n\n### \"Token not found\" error after setup\n\nThe token acquisition request failed. Verify:\n1. Aria Ops is reachable: `vmware-aria doctor`\n2. The `auth_source` in config matches your environment (LOCAL, LDAP, AD)\n3. The password env var follows the naming convention: `VMWARE_ARIA_<TARGET>_PASSWORD`\n\n### Resources appear missing from list_resources\n\nThe collector agent may be offline. Check `list_collector_groups` for any collectors in a DOWN state. Restart the affected collector from the Aria Ops UI under Administration > Collector Groups.\n\n### Metrics return empty data\n\nRead `missing[].reason`: `not_collected_for_resource` (wrong key for this resource — try `similar_keys`), `no_data_in_window` (widen `--hours`, check collectors), `resource_reports_no_stat_keys`, or `undetermined`. Never report a missing key as zero.\n\n### `health status` says OFFLINE (HTTP 503) but data still flows\n\nThe node flag is OFFLINE whenever any one service is not running. Read `assessment`: DEGRADED means some services are OK and others are not — not an outage. Seen on Aria Operations 8.18.7 with only `LOCATOR` not OK; `services_not_ok` names the failed ones.\n\n### \"Password not found\" error\n\nVariable names follow the pattern `VMWARE_ARIA_<TARGET_NAME_UPPER>_PASSWORD` where hyphens become underscores. Example: target `prod` needs `VMWARE_ARIA_PROD_PASSWORD`. Check your `~/.vmware-aria/.env` file.\n\n### `invalid peer certificate: UnknownIssuer` when running uvx (corporate TLS proxy)\n\n`uvx` re-resolves dependencies from PyPI on every launch. Behind a corporate TLS-intercepting proxy whose CA is not in uv's bundled cert store, the handshake fails. Use the v1.5.15+ recommended single-command form `vmware-aria mcp` (after `uv tool install vmware-aria==1.17.0` — no network on launch), or set `UV_NATIVE_TLS=true` to make uv use the system cert store.\n\n## Audit & Safety\n\n1. **Source code**: [github.com/vmware-skills/VMware-Aria](https://github.com/vmware-skills/VMware-Aria) (MIT).\n2. **Config and credentials**: `config.yaml` holds hosts and usernames only; passwords live in `~/.vmware-aria/.env` (chmod 600) as `VMWARE_ARIA_<TARGET>_PASSWORD` and are never logged.\n3. **No webhooks**: no outbound calls besides the Aria Operations REST API over HTTPS 443; the MCP server is local stdio.\n4. **TLS**: verification on by default; for a private CA set `SSL_CERT_FILE` rather than `verify_ssl: false` (isolated labs only).\n5. **Prompt-injection defense**: API text is sanitized (control characters stripped, length capped) before it reaches the agent.\n6. **Least privilege**: use an Aria Operations account with read-only roles unless the write tools (alert acknowledge/cancel, alert notes, alert definitions, reports, resource maintenance) are needed.\n\nEvery tool call goes through vmware-policy (`@vmware_tool`): audited to `~/.vmware/audit.db`, subject to `~/.vmware/rules.yaml` deny rules and maintenance windows, each tool risk-tagged. View with `vmware-audit log --last 20` or `--status denied`. The suite-api token is re-acquired automatically before it expires. Setup, multiple targets, MCP clients and Docker: [`references/setup-guide.md`](references/setup-guide.md).\n\n## License\n\nMIT — [github.com/vmware-skills/VMware-Aria](https://github.com/vmware-skills/VMware-Aria)\n\nFile v1.17.0:_meta.json\n\n{\n  \"ownerId\": \"kn7b067awq2s97bn3d7p5qfhw5827pxc\",\n  \"slug\": \"vmware-aria\",\n  \"version\": \"1.17.0\",\n  \"publishedAt\": 1789915908066\n}\n\nFile v1.17.0:references/agent-guardrails.md\n\n# Operating vmware-aria with a local / small model\n\nClaude-class models drive this skill without special instruction. Smaller and\nlocally-hosted models — Llama 3.3 70B, Qwen, Mistral, and similar, served\nthrough Goose, Ollama, or OpenShift AI — need explicit operating rules to call\ntools reliably.\n\nThis page covers what goes wrong most often with vmware-aria specifically:\n**alert-to-resource correlation**. For the full cross-skill guardrail set, the\ncomplete system prompt, and the small-model failure-mode checklist, see the\ncanonical guide in\n[vmware-monitor's references](https://github.com/vmware-skills/VMware-Monitor/blob/main/skills/vmware-monitor/references/agent-guardrails.md).\n\nThese guardrails are adapted, with thanks, from the working configuration\n[@juanpf-ha](https://github.com/juanpf-ha) developed while running vmware-aria\nand vmware-monitor against a production vSphere estate\n([VMware-AIops#31](https://github.com/vmware-skills/VMware-AIops/issues/31)).\n\n> **Disclaimer**: This is a community-maintained open-source project and is\n> **not affiliated with, endorsed by, or sponsored by VMware, Inc. or Broadcom\n> Inc.** \"VMware\" and \"vSphere\" are trademarks of Broadcom.\n\n---\n\n## 1. Alert-to-resource correlation\n\nThis is the sequence small models get wrong most reliably. An Aria alert does\nnot carry the affected object's name — only a `resourceId`. Correlating an\nalert with vCenter therefore means: fetch the alert, read its `resourceId`,\nfetch that resource, confirm its name and kind, then query vmware-monitor. Two\nUUIDs are in play, and they get swapped.\n\n**Use `investigate_alert` instead of chaining the steps by hand.** It performs\nthe whole sequence server-side and returns:\n\n- `alert` — Aria's own criticality / status / impact / control-state values,\n  passed through verbatim.\n- `resource` — the affected object, or `null`.\n- `correlation` — both UUIDs **explicitly labelled** (`alert_id` vs\n  `resource_id`), plus `resource_name`, `resource_kind`, and a `confirmed`\n  flag. Every key is always present; unresolved values are explicit `null`.\n- `next_step` — the exact vmware-monitor tool and argument to call next, or\n  `null` when there is nothing confirmed to hand off.\n- `warnings` — why anything above is incomplete; empty on success.\n\nA resource that cannot be resolved degrades to a warning plus explicit nulls\nrather than an error, so the alert you already fetched is never lost.\n\nIf you must drive the steps manually, state these rules explicitly:\n\n```text\n- Use investigate_alert to go from an alert to its affected resource.\n- Do not pass an alert UUID where a resource UUID is expected. They are\n  different objects; investigate_alert labels both.\n- Only query vCenter for a resource once correlation.confirmed is true.\n  The next_step block names the exact tool and argument to use.\n```\n\n---\n\n## 2. Aria-specific data fidelity\n\nAria's enum values carry operational meaning and must survive the model\nuntouched:\n\n```text\n- Preserve the exact criticality (INFORMATION / WARNING / IMMEDIATE /\n  CRITICAL), status, impact, and control-state values the tools return.\n  Do not translate, normalise, or prettify them.\n- alertLevel is the criticality field. An alert definition has no top-level\n  criticality — it is the maximum severity across its states.\n- Recommendations hang off the alert definition, not the alert.\n- Do not claim a capacity, performance, or risk problem unless the tool output\n  contains explicit supporting evidence. Badge colour alone is not a diagnosis.\n```\n\n---\n\n## Reporting results\n\nLocal-model compatibility is an explicit design constraint for this family, and\nthe evidence base is small. If you evaluate a model against this skill, a\nreport of what worked and what did not is genuinely useful:\n[github.com/vmware-skills/VMware-Aria/issues](https://github.com/vmware-skills/VMware-Aria/issues).\n\nFile v1.17.0:references/capabilities.md\n\n# Capabilities\n\n## Automation Level Reference\n\nEach operation is classified by autonomy level per the Enterprise Harness Engineering framework. **vmware-aria is heavily L1/L2 (34 read / 10 write)** — primarily a monitoring and analysis skill.\n\n| Level | Meaning | Agent autonomy | Examples in this skill |\n|:-:|---|---|---|\n| **L1** | Read-only, raw data | Always auto-run | `list_resources`, `get_resource`, `get_resource_metrics`, `list_metric_keys`, `get_resource_properties`, `get_resource_relationships`, `list_alerts`, `get_alert`, `list_alert_definitions`, `list_alert_notes`, `list_maintenance_schedules`, `get_aria_node_resources`, `list_adapters`, capacity / badge queries |\n| **L2** | Read + analysis / recommendation | Always auto-run | `investigate_alert` (alert → confirmed affected resource), `get_alert_recommendations` (alert → definition state → prioritized recommendations), anomaly counts, top-N consumer ranking, capacity trend forecasting, rightsizing recommendations |\n| **L3** | Single write — user must approve | Only after explicit confirmation | `acknowledge_alert` (via takeownership action), `cancel_alert`, `create_alert_definition`, `set_alert_definition_state`, `delete_alert_definition`, `generate_report`, `delete_report`, `start_resource_maintenance`, `end_resource_maintenance`, `add_alert_note` *(the only writes; all auditable. On MCP, `acknowledge_alert`, `cancel_alert`, `delete_alert_definition`, `delete_report`, `start_resource_maintenance` and `end_resource_maintenance` take `confirm` (default false): without `confirm=true` they return a `blast_radius` — what the call would change — and change nothing; `confirm=true` is refused when a blocker is found or a field the radius depends on could not be read. `confirmed` is a deprecated alias. `add_alert_note` is low risk and has no gate — a note changes neither the alert nor monitoring; the CLI still asks once)* |\n| **L4** | Multi-step plan / apply workflow | *N/A currently* | — *(no multi-step orchestration; Aria is observe/analyze, not configure)* |\n| **L5** | Auto-remediation from learned pattern | Pattern library only; requires `risk:low` + `reversible:true` + `repeatable:true` | *(roadmap — candidates: auto-acknowledge known-noisy alerts, auto-cancel resolved-by-event alerts)* |\n\n**Notes**:\n- L1/L2 tools are always safe for agents to call without confirmation.\n- L3 alert-state writes pass through the `@vmware_tool` decorator: connection check → policy check → audit log. Cancel is irreversible by Aria API design and treated as a destructive operation.\n- For VM/host operations see [vmware-aiops](https://github.com/vmware-skills/VMware-AIops); Aria recommendations are advisory, not actuating.\n\n## What vmware-aria Can Do\n\n### Resource Monitoring\n\n- **List any resource type**: VirtualMachine, HostSystem, ClusterComputeResource, Datastore, Datacenter, ResourcePool\n- **Filter by name**: substring match across any resource list\n- **Find objects that stopped reporting**: every row carries `aria_state` (Aria's lifecycle state — `STARTED` for a powered-off VM too) and `collection_status` (`DATA_RECEIVING`, `NO_DATA_RECEIVING`, …; null when not reported). `collection_status=\"NO_DATA_RECEIVING\"` with `resource_kind=\"all\"` lists the objects behind \"Objects are not receiving data from adapter instance\"; a filter that matches nothing returns a `note` naming the statuses seen\n- **Get resource details**: health, risk, and efficiency badges plus all identifiers\n- **Fetch metric time series**: any metric key with configurable time window and rollup (AVG/MAX/MIN). Keys with no points are listed under `missing` with a reason (`not_collected_for_resource` with `similar_keys`, `no_data_in_window`, `resource_reports_no_stat_keys`, `undetermined` — also when stat-key rows are in an unrecognised form) instead of silently vanishing. `similar_keys` holds at most 10 keys, only ones `sanitize()` leaves unchanged and at most 200 characters long\n- **Look up metric keys**: `list_metric_keys` lists the keys one resource actually reports, joined with its kind's definitions for name and unit. Each row's `definition` is `found`, `found_by_instance` (an instanced key such as `guestfilesystem:/boot|usage` joined to `guestfilesystem|usage`), `not_defined_for_kind` (also listed in `unjoined_keys`) or `not_read` (the definitions could not be read: `definitions_status: undetermined`, name and unit unknown, not absent). With `resource_kind` instead, it lists the keys the kind defines — a defined key is not collected on every resource. A nonexistent resource id is an HTTP 404 error, not an empty list\n- **Read properties**: `get_resource_properties` — current `{name, value}` rows sorted by name (power state, `summary|parentHost`, configured CPU and memory, `config|extraConfig|*` flags), filtered by a name substring. Configuration facts, not time series\n- **Walk relationships**: `get_resource_relationships` — related resources with `id`, `name`, `kind`, `adapter_kind` and `direction` (`parent` / `child` / `both` / `other`). `relationship_type` is exactly ALL, PARENT or CHILD; ANCESTOR and DESCENDANT are refused (Aria Operations 8.18.7 answers HTTP 400 for them), so walk PARENT one level at a time. With ALL, `direction` is null and `direction_note` says why when the PARENT/CHILD lists could not be read\n- **Find top consumers**: rank VMs or hosts by a metric's last-hour average (`value`, 5-minute AVG rollup — what Aria ranks by, descending; `latest_value` is the most recent point). Resources with no data for the key are left out, not ranked at zero; `excluded_no_data` counts the ones the ranking listed with no points, and `hint` says when that shortened the list — when they took slots of a full `top_n`, the result is `truncated` and `hint` says to raise `top_n`\n\n### Alert Management\n\n- **List active or all alerts**: filter by criticality (INFORMATION/WARNING/IMMEDIATE/CRITICAL) or resource. The Alert model has no resource name, so each row's `resource_name` and `resource_kind` are resolved in one batched `/resources` lookup per page; `null` means unknown, and `resource_names_note` says how many could not be retrieved (a failed lookup, or an appliance that ignored the id filter), were not returned, or have no name in Aria Operations\n- **Inspect alert details**: contributing (triggered) symptoms from the dedicated contributingsymptoms endpoint, plus timeline. A symptom that carries no name or severity (every symptom on Aria Operations 8.18.7) takes both from its symptom definition, fetched in one batched `/symptomdefinitions` lookup; `definition_lookup` and `symptom_definitions_note` say when that did not work, including symptoms with no definition id to look up. Each symptom also carries the object it is on (`resource_id`, `resource_name`, `resource_kind`, `stat_key`, and `condition` from the instance message): on 8.18.7 the contributing-symptom payload has no resource id, so it is read from the symptom instance in `/symptoms` — which ignores its `id` filter there, so the tool walks it and keys rows by id. That is what names the down services behind \"vCenter app health is affected\" (e.g. `mem`, `system`). On 8.18.7 the walk runs for every alert, not only vCenter app alerts: one request per 1,000 symptoms in the appliance (capped at 20,000 symptoms), stopping once all are found, plus one batched `/resources` lookup; `acknowledge_alert` / `cancel_alert` skip it for their before-state. `resource_lookup` and `symptom_resources_note` say when an object could not be read — an empty `resource_id` is then unknown, not absent. Times come as epoch ms and ISO-8601 UTC (`*_time_utc`). Recommendations are attached to the alert definition, not the alert\n- **Investigate an alert end-to-end**: `investigate_alert` resolves an alert to its affected resource in one call — fetches the alert, reads `resourceId`, fetches that resource, confirms name and kind, and returns both UUIDs *explicitly labelled* plus a ready-to-use handoff naming the exact vmware-monitor tool and argument. Use it instead of chaining `get_alert` → `get_resource` by hand: the two UUIDs are different objects and a small model will otherwise swap them. An unresolvable resource degrades to a warning plus explicit nulls rather than losing the alert\n- **Acknowledge alerts**: mark as seen without closing (control state → ACKNOWLEDGED)\n- **Alert notes**: `list_alert_notes` returns an alert's notes (`note`, `type` USER / SYSTEM, `user_name`, `created_time_ms`); an unknown alert id is HTTP 404, and a set `notes_note` means the notes are unknown, not absent. `add_alert_note` records who is handling the alert or what was done — it does not change status or ownership (use `acknowledge_alert`), is not idempotent (two calls add two notes) and has no undo. `created: null` with `confirmation_note` means Aria did not confirm the note: run `list_alert_notes` before retrying\n- **Alert recommendations**: `get_alert_recommendations` resolves the alert to its alert definition, uses the state whose severity matches the alert's criticality (else the definition's only state, else every state merged at each recommendation's highest priority, with a `note`), and returns the recommendations sorted by priority (lower is more important) with description and action. `status`: `found`; `partial` (some text could not be read — `description: null` is unknown, not blank); `none_defined` (`recommendations: []`); `unknown` (the definition could not be read — `recommendations: null`, never \"no recommendations\"). On Aria Operations 8.18.7 a CRITICAL vCenter-app alert returned 7 prioritized recommendations\n- **Cancel alerts**: permanently dismiss (status → CANCELLED)\n- **Browse alert definitions**: the templates that define when alerts fire\n\n### Capacity Planning\n\n- **Cluster capacity overview**: group-level remaining-capacity percentage plus per-dimension (cpu/mem/diskspace) headroom and days-until-full (the percentage metric only exists at group level)\n- **Remaining capacity**: how much more CPU, memory, disk can be added before hitting limits\n- **Time remaining**: predicted days until each capacity dimension is exhausted (based on trend)\n- **Rightsizing recommendations**: identify over-provisioned VMs (reclaim resources) and under-provisioned VMs (prevent degradation). Raw recommendations are MHz / KB / GB (`recommended_units`); `recommended_vcpus` converts CPU with the VM's own MHz per vCPU. `cpu_direction` / `memory_direction` compare against the current configuration (memory within 1% is `right_sized`). Powered-off VMs and templates are listed but never `actionable`, nor is a VM whose power state or template flag is unknown. If the VM property read fails, the property-derived fields are null, no row is actionable, and `properties_note` names the failure. `caveats` flag engine disagreement and vendor minimum sizes before any reduction. `recommendation_range` gives the daily low and high of each recommendation over the last 7 days with `days_with_data`; a CPU or memory range wider than 5% of its high makes `recommendation_stable` false and the row not actionable. Null when no history came back; `history_note` names a failed history read\n\n### Resource Maintenance\n\n- **Start maintenance**: `start_resource_maintenance` stops Aria alerting on one resource and collecting its data. With `duration_minutes` (1–525600) or `end_time_ms` (epoch milliseconds, in the future) the resource is `MAINTAINED` for that window and returns to its prior state when it expires; pass one, not both. With neither it is `MAINTAINED_MANUAL` until ended. Returns `requested`, `before` / `after` state, `confirmed` (true / false / null — null means the after-state could not be read, which is unknown, not failure) and `note`. On MCP a call without `confirm=true` reads the resource and returns `blast_radius` (name, kind, adapter states, whether it is in maintenance, the requested window) without changing anything; `confirm=true` is refused when the resource is already in maintenance (starting again would replace that window) or its state is unknown. An impossible window is refused before connecting. Risk medium; audited with the before and after state; undo is `end_resource_maintenance`, recorded only when the resource was known not to be in maintenance before (so an undo never closes a window this call did not open)\n- **End maintenance**: `end_resource_maintenance` refuses a resource known not to be in maintenance (an adapter reports a state such as `STARTED` or `STOPPED`). When the state is unknown — it cannot be read, or an adapter reports `UNKNOWN` / `NONE` — the MCP tool refuses too (its blast radius cannot be measured); the CLI proceeds and `before` says unknown. Without `confirm=true` the MCP tool returns the `blast_radius` and changes nothing. Risk medium, audited. Its undo, recorded only when the resource was known to be in maintenance, re-enters manual maintenance — the end of a timed window is not restored\n- **Maintenance schedules**: `list_maintenance_schedules` — name, schedule type (ONCE / DAILY / WEEKLY / MONTHLY / YEARLY), recurrence, start hour and minute, duration in minutes, time zone, start and expiry. A schedule does not list its resources; pass `resource_id` for the schedules of one resource. A set `schedules_note` means the schedules are unknown, not absent\n\n### Anomaly Detection\n\n- **List anomalies**: per-resource Total Anomalies counts (`System Attributes|total_alarms` metric — active symptoms, events, and DT violations on the object and its children), optionally scoped to one resource. The UI's anomalous-metrics list is not part of the public API\n- **Risk badge**: composite risk score (0–100) from the resource's `badges[]` array — predicts likelihood of future problems; for contributing causes inspect the resource's active alerts\n\n### Platform Health\n\n- **Aria health check**: `assessment` HEALTHY / DEGRADED / DOWN / UNKNOWN from the node status plus the per-service breakdown, and the product version and line (8.x / 9.x). The node reports OFFLINE whenever any one service is not running, so OFFLINE with some services OK is DEGRADED, not down. An ONLINE node whose per-service breakdown could not be read is HEALTHY with `details` saying no service was checked individually; a service in a state other than OK or ERROR makes it UNKNOWN unless DEGRADED. An unreadable version or service list is reported (`version_error` / `services_error`), not raised\n- **Collector group status**: list collector groups (member IDs) enriched with each collector's name, UP/DOWN state, and local flag\n- **Aria node resources**: `get_aria_node_resources` reads Aria's own self-monitoring objects (`vC-Ops-Node`, `vC-Ops-Watchdog`): per node memory (`mem|total`, `mem|used`, `mem|free`, `mem|actualFree`, `mem|actualUsed`), swap, heap overall and per component, and watchdog restarts per service. Each value has `latest`, `unit` (from Aria's statkey definitions, never assumed) and `window` min / avg / max / points over `window_hours` (1–720, default 24) of 5-minute averages. `memory_pressure.level` is an indicator: HIGH when actual free memory is below 10% of total, ELEVATED below 20%, NORMAL at 20% or more, UNKNOWN when the readings do not settle it. A key with no value is in the node's `missing` list (`not_reported` / `no_data` / `undetermined`), never zeroed; `watchdog_restarts: null` is unknown, not zero. On Aria Operations 8.18.7 after a memory upgrade the node read `mem|total` 15.61 GB (was 7.75), actual free 41%, pressure NORMAL, and 9 watchdog services with 0 restarts\n- **Adapter collection state**: `list_adapters` — each adapter instance's kind, collector, monitoring interval, resources and metrics collected, last collected and last heartbeat with ages in seconds, its own message, and `stale`: true when the last collection is older than max(3 × monitoring interval, 15 minutes), false within that, null when the fields cannot support a verdict (`stale_basis` shows the arithmetic). Ages use the appliance clock from node status when it is available (`reference_clock`). An empty or unrecognised answer is an error, never \"no adapters\". A recent collection does not prove every object behind the adapter receives data. On Aria Operations 8.18.7 all 6 adapters were not stale\n\n---\n\n## What vmware-aria Cannot Do\n\n| Capability | Use Instead |\n|-----------|-------------|\n| Create / delete / power VMs | `vmware-aiops` |\n| Configure NSX segments, gateways, NAT | `vmware-nsx` |\n| NSX DFW / firewall rules | `vmware-nsx-security` |\n| vSphere inventory (VMs, hosts, clusters) read-only | `vmware-monitor` |\n| Storage: iSCSI, vSAN, datastores | `vmware-storage` |\n| Tanzu Kubernetes cluster management | `vmware-vks` |\n| Create alert definitions | Supported via `create_alert_definition` (from symptom definition IDs) |\n| Configure dashboards | Not supported (UI required) |\n| Manage Aria adapter instances | Not supported (UI required); `list_adapters` reads their collection state |\n\n---\n\n## Aria Operations API Coverage\n\nAll requests carry `Authorization: vRealizeOpsToken <token>`.\n\n| Endpoint | Used For |\n|----------|---------|\n| `POST /suite-api/api/auth/token/acquire` | Token authentication |\n| `POST /suite-api/api/auth/token/release` | Token release on close (no body; token identified by the Authorization header) |\n| `GET /suite-api/api/resources` | list_resources (also candidate listing for topn / anomaly / rightsizing scans); list_alerts and get_alert symptom resource names (`resourceId` repeated, 100 per request); get_aria_node_resources (self-monitoring `vC-Ops-Node` / `vC-Ops-Watchdog` objects) |\n| `GET /suite-api/api/resources/{id}` | get_resource, get_resource_health, get_resource_riskbadge (badges come from the `badges[]` array — there are no `/badge/*` endpoints), investigate_alert (resource-side leg), list_metric_keys (the resource's adapter and resource kind), start_resource_maintenance / end_resource_maintenance (blast radius, and state before and after, from `resourceStatusStates`), acknowledge_alert / cancel_alert (the alert's resource name, context only) |\n| `POST /suite-api/api/resources/{id}/stats/query` | get_resource_metrics |\n| `GET /suite-api/api/resources/{id}/statkeys` | get_resource_metrics (only when a requested key returned no points, to explain why); list_metric_keys (the keys a resource reports); get_aria_node_resources (the keys each node reports) |\n| `GET /suite-api/api/adapterkinds/{adapterKind}/resourcekinds/{resourceKind}/statkeys` | list_metric_keys (key names and units); get_aria_node_resources (units for the self-monitoring kinds) |\n| `GET /suite-api/api/resources/{id}/properties` | get_resource_properties; get_resource_health (`SERVICE\\|STATUS`, service-kind objects only) |\n| `GET /suite-api/api/resources/{id}/relationships` + `/relationships/{PARENT\\|CHILD}` | get_resource_relationships (paged walk; with ALL the PARENT and CHILD lists label `direction`) |\n| `GET /suite-api/api/resources/stats/latest` | get_aria_node_resources (latest values) |\n| `POST /suite-api/api/resources/properties/latest/query` | list_rightsizing_recommendations (current vCPUs, memory, CPU speed, power state, template flag, product name) |\n| `GET /suite-api/api/resources/stats/topn` | get_top_consumers (resourceId list capped at 100) |\n| `GET /suite-api/api/resources/{id}/stats/latest` | get_capacity_overview, get_remaining_capacity, get_time_remaining (OnlineCapacityAnalytics keys) |\n| `POST /suite-api/api/resources/stats/query` | list_rightsizing_recommendations (OnlineCapacityAnalytics recommendedSize keys), list_anomalies (`System Attributes\\|total_alarms`) — one request for a resourceId array; get_aria_node_resources (window min / avg / max); get_resource_health (latest `SERVICE\\|AVAILABILITY`, service-kind objects only) |\n| `PUT /suite-api/api/resources/{id}/maintained` | start_resource_maintenance (window as the `duration` or `end` query parameter; neither = manual maintenance) |\n| `DELETE /suite-api/api/resources/{id}/maintained` | end_resource_maintenance |\n| `GET /suite-api/api/maintenanceschedules` | list_maintenance_schedules (paged; `resourceId` filter) |\n| `POST /suite-api/api/alerts/query` | list_alerts (server-side status/criticality/resource filtering) |\n| `GET /suite-api/api/alerts/{id}` | get_alert, investigate_alert (alert-side leg; no dedicated endpoint — the tool composes the two existing reads), get_alert_recommendations (the alert's definition id and criticality), acknowledge_alert / cancel_alert (blast radius) |\n| `GET /suite-api/api/alerts/{id}/notes` | list_alert_notes (paged) |\n| `POST /suite-api/api/alerts/{id}/notes` | add_alert_note (body `{\"content\": …}`) |\n| `GET /suite-api/api/alertdefinitions/{id}` | get_alert_recommendations (states and their recommendation priorities), delete_alert_definition (blast radius) |\n| `GET /suite-api/api/recommendations` | get_alert_recommendations (recommendation text; `id` repeated, 50 per request) |\n| `GET /suite-api/api/alerts/contributingsymptoms?id={alertId}` | get_alert (triggered symptoms) |\n| `GET /suite-api/api/symptoms` | get_alert / investigate_alert (the object each symptom is on — `resourceId`, `statKey`, `message` — when the contributing-symptom leaf carries none; 8.18.7 ignores the `id` filter, so pages are walked until every symptom is found or `pageInfo.totalCount` is reached — up to 20 requests; not used by acknowledge_alert / cancel_alert) |\n| `POST /suite-api/api/alerts?action=takeownership` | acknowledge_alert |\n| `POST /suite-api/api/alerts?action=cancel` | cancel_alert |\n| `GET /suite-api/api/alertdefinitions` | list_alert_definitions |\n| `POST /suite-api/api/alertdefinitions` | create_alert_definition |\n| `PUT /suite-api/api/alertdefinitions/{id}/enable` (or `/disable`) | set_alert_definition_state |\n| `DELETE /suite-api/api/alertdefinitions/{id}` | delete_alert_definition |\n| `GET /suite-api/api/symptomdefinitions` | list_symptom_definitions (filter param is `resourceKind`); get_alert / investigate_alert symptom names and severities (`id` repeated, 50 per request) |\n| `GET /suite-api/api/reportdefinitions` | list_report_definitions (`subject` is an array of resource-kind strings) |\n| `POST /suite-api/api/reports` | generate_report (requires at least one resource UUID) |\n| `GET /suite-api/api/reports` / `GET /suite-api/api/reports/{id}` | list_reports / get_report (timestamp field is `completionTime`; definition filter and limit applied client-side); delete_report (blast radius) |\n| `DELETE /suite-api/api/reports/{id}` | delete_report |\n| `GET /suite-api/api/deployment/node/status` | get_aria_health (a 503 is read as a status, not an error), is_alive, list_adapters (the appliance clock, `systemTime`, for collection ages) |\n| `GET /suite-api/api/adapters` | list_adapters (unpaged) |\n| `GET /suite-api/api/deployment/node/services/info` | get_aria_health, doctor (per-service health) |\n| `GET /suite-api/api/versions/current` | get_aria_health, doctor \"Aria version\" row, and the version shown when a 9.0+ tool (fleet_*, findings_list, promql_query) is called on an older appliance |\n| `GET /suite-api/api/collectorgroups` + `GET /suite-api/api/collectors` | list_collector_groups (groups carry member IDs; details enriched from /collectors) |\n\n---\n\n## List Result Envelope\n\nEvery list-returning tool wraps its rows in the family envelope\n(`vmware_policy.paginated`) rather than returning a bare array, so an agent can\ntell a complete answer from page one instead of guessing (VMware-AIops issue\n#31). Keys: `items`, `returned`, `limit`, `total`, `truncated`, `hint` — always\nall six, with explicit `null` where a value is unknown.\n\n`total` is only populated where the suite-api genuinely reports a collection\nsize. It is never inferred:\n\n| Tool | `total` source | Notes |\n|------|---------------|-------|\n| `list_resources` | `pageInfo.totalCount` on `GET /resources` | Suppressed under `name_filter` — that filter is client-side, so the server's count describes the unfiltered kind |\n| `list_alert_definitions` | `pageInfo.totalCount` on `GET /alertdefinitions` | Suppressed under `name_filter` |\n| `list_symptom_definitions` | `pageInfo.totalCount` on `GET /symptomdefinitions` | `resource_kind` is a server-side param, so it is reflected in the count; suppressed under `name_filter` |\n| `list_report_definitions` | `pageInfo.totalCount` on `GET /reportdefinitions` | Suppressed under `name_filter` |\n| `list_reports` | Count of matches from the unpaged `GET /reports` | The whole matching set is in hand, so the count is exact |\n| `list_rightsizing_recommendations` | VM `pageInfo.totalCount` from the candidate `GET /resources` | One row per VM evaluated, so the count describes the same collection |\n| `list_anomalies` | Count of flagged objects on a complete scan; VM `pageInfo.totalCount` when the scan hit its cap | Also carries `scanned`, `vm_total`, `scan_complete`. `limit` bounds the answer, not the scan. Only flagged VMs are returned, so a short list is not evidence of a clean environment |\n| `list_alerts` | — | `POST /alerts/query` reports no count; a full page is conservatively flagged truncated |\n| `get_top_consumers` | — | A top-N ranking is a slice of an unbounded set |\n| `list_collector_groups` | — | `GET /collectorgroups` is unpaged and takes no limit, so `truncated` is always `false` |\n| `list_adapters` | Count of matches from the unpaged `GET /adapters` | Exact; the `adapter_kind` filter is applied client-side and reflected in the count |\n| `list_metric_keys` | Count of rows after `key_filter` from the unpaged statkeys read | Exact |\n| `get_resource_properties` | Count of rows after `name_filter` from the unpaged `GET /resources/{id}/properties` | Exact |\n| `get_resource_relationships` | Count of related resources from a complete walk of the paged relationships endpoint | `null` only when the walk hit its safety cap (`direction_note` says so) |\n| `list_maintenance_schedules` | `pageInfo.totalCount` on `GET /maintenanceschedules` | `null` when the API reports no size; a set `schedules_note` means `items` is unknown, not empty |\n| `list_alert_notes` | `pageInfo.totalCount` on `GET /alerts/{id}/notes` | `null` when the API reports no size; a set `notes_note` means `items` is unknown, not empty |\n\nThe six tools added above also return `next_offset`: pass it back as `offset`\nand stop when it is `null`.\n\nCLI commands unwrap `items` and print the rows; the envelope is the MCP/library\ncontract.\n\nReading rules for an agent:\n\n- **Rows live under `items`.** An empty `items` with `returned: 0` means the query genuinely matched nothing — report that, not a tool failure.\n- **`truncated: true` means more rows exist.** Never describe it as the complete set; say it is partial or re-query with a higher `limit` or a narrower filter, as `hint` says.\n- **`truncated: false` means the answer is complete.**\n- **`total: null` means the API reported no collection size**, so a page filled exactly to the limit is flagged truncated conservatively; a follow-up with a larger limit settles it.\n- **`list_anomalies`**: `limit` bounds the answer, not the scan — the environment is ranked in full and the worst `limit` objects returned. Only VMs with a non-zero count are returned, so a short list is not evidence of a clean environment. With `scan_complete: true`, `total` is the number of anomalous objects; with `scan_complete: false` the scan hit its cap, `total` is the VM count, and a `note` says the ranking is partial.\n\n---\n\n## Aria Operations / VCF Operations Version Compatibility\n\n| Feature | Minimum Version |\n|---------|----------------|\n| VCF Operations 9.1 (VCF 9.1) | ✅ All tools; PromQL uses the 9.1 VODAP service (base path inferred, not yet verified on real hardware). Aria Operations was rebranded VCF Operations in VCF 9. |\n| VCF Operations 9.0 (VCF 9.0) | ✅ suite-api tools plus fleet certificates / passwords / domains and diagnostic findings |\n| Aria Operations 8.x | ✅ suite-api tools. Fleet, findings and PromQL are 9.0+: on 8.x they return a \"requires VCF Operations 9.0 or newer\" error naming the version the appliance reports |\n| Token authentication | 6.6+ |\n| Resource metrics stats query | 6.7+ |\n| Rightsizing recommendations | 7.0+ |\n| Anomaly detection | 7.5+ |\n| Suite API v2 paths used | 8.0+ |\n\nEndpoints are checked against the vROps 8.6 and VCF Operations 9.1 API indexes in `tests/eval/spec/`. Live-verified on Aria Operations 8.18.7 (2026-09): resources, metrics, alerts and symptoms, rightsizing, health, doctor, node resources, adapters, alert recommendations, and reads of maintenance schedules and alert notes (none existed).\n\nFile v1.17.0:references/cli-reference.md\n\n# CLI Reference\n\nComplete reference for `vmware-aria` command-line interface.\n\n## Global Options\n\nAll commands accept:\n- `--target / -t <name>` — Target name from config (uses default if omitted)\n- `--config / -c <path>` — Custom config file path\n\n---\n\n## `vmware-aria doctor`\n\nRun pre-flight diagnostics.\n\n```\nvmware-aria doctor [OPTIONS]\n\nOptions:\n  --skip-auth    Skip authentication check (only tests config + network)\n  --config -c    Path to config file\n```\n\n**Checks performed**:\n1. Config file exists at `~/.vmware-aria/config.yaml`\n2. `.env` file permissions (warns if wider than 600)\n3. Config parse succeeds (validates YAML and target structure)\n4. Password env vars are set for each target\n5. Network TCP connectivity to port 443 for each target\n6. Aria Operations token acquisition (unless `--skip-auth`)\n7. Aria version from `GET /versions/current`, e.g. `VMware Aria Operations 8.18.7 (8.x line, build 25423534)` (**WARN** `Not read: …` when it cannot be read — an unreadable version says nothing about whether the target works), and Aria platform health: PASS when HEALTHY, **WARN** when DEGRADED or UNKNOWN (the failed services are named), FAIL when DOWN. Only a failure to connect is an auth FAIL: an error after the token was acquired is a FAIL on the \"Aria platform\" row (`Checks did not complete: …`). The doctor disconnects from each target either way\n8. MCP server module importable\n\nError details in the doctor table do not tell you to run the doctor again.\n\n---\n\n## Resource Commands\n\n### `vmware-aria resource list`\n\nList resources by kind.\n\n```\nvmware-aria resource list [OPTIONS]\n\nOptions:\n  --kind -k TEXT             Resource kind [default: VirtualMachine]\n                             Values: VirtualMachine, HostSystem, ClusterComputeResource,\n                                     Datastore, Datacenter, ResourcePool, or all\n  --limit -n INT             Max results [default: 50]\n  --name TEXT                Filter by name substring (case-insensitive)\n  --collection-status TEXT   Keep objects with this data-collection status,\n                             e.g. NO_DATA_RECEIVING (case-insensitive)\n  --target -t TEXT           Target name\n```\n\n**Output**: Table with Name (and Kind with `--kind all`), ID, Health (color + score), Aria state, and Collection.\n**Aria state** is Aria's lifecycle state for the object — `STARTED` for a powered-off VM too, so it is not a power state.\n**Collection** is whether data is arriving (`DATA_RECEIVING`, `NO_DATA_RECEIVING`, …; `—` when Aria reports none).\nTo find the objects behind \"Objects are not receiving data from adapter instance\", run\n`vmware-aria resource list --kind all --collection-status NO_DATA_RECEIVING`.\nIf the filter matches nothing, a yellow line names the statuses the listing did contain.\n\n### `vmware-aria resource get`\n\nGet full resource details.\n\n```\nvmware-aria resource get <resource-id> [OPTIONS]\n\nArguments:\n  resource-id  Resource UUID (required)\n\nOptions:\n  --target -t TEXT  Target name\n```\n\n**Output**: JSON with all resource fields including health, risk, efficiency badges and identifiers.\n\n### `vmware-aria resource metrics`\n\nFetch time-series metrics for a resource.\n\n```\nvmware-aria resource metrics <resource-id> [OPTIONS]\n\nArguments:\n  resource-id  Resource UUID (required)\n\nOptions:\n  --metrics -m TEXT    Comma-separated metric keys\n                       [default: cpu|usage_average,mem|usage_average]\n  --hours INT          History window in hours [default: 1]\n  --summary            Per metric n/min/max/avg/latest and change points, instead of every point\n  --target -t TEXT     Target name\n```\n\n**Common metric keys** (VirtualMachine names and units as Aria Operations 8.18.7 defines them; other resource kinds differ):\n- `cpu|usage_average` — CPU|Usage (%)\n- `mem|usage_average` — Memory|Usage (%)\n- `cpu|demandmhz` — CPU|Demand (MHz)\n- `mem|workload` — Memory|Workload (%)\n- `disk|usage_average` — Physical Disk|Total Throughput (KBps) — a throughput, not a utilization percentage\n- `net|usage_average` — Network|Usage Rate (KBps)\n\n**Output**: JSON object:\n\n```json\n{\n  \"resource_id\": \"<uuid>\",\n  \"window_begin_ms\": 1757700000000,\n  \"window_end_ms\": 1757703600000,\n  \"mode\": \"raw\",\n  \"metrics\": {\"cpu|usage_average\": [{\"timestamp_ms\": 1757700300000, \"value\": 3.2}]},\n  \"missing\": [\n    {\"metric_key\": \"mem|usage_avg\", \"reason\": \"not_collected_for_resource\",\n     \"detail\": \"...\", \"similar_keys\": [\"mem|usage_average\", \"...\"]}\n  ],\n  \"stat_keys_on_resource\": 170\n}\n```\n\n`metrics` holds only keys that returned at least one point. Every other requested key is in `missing`, with `reason`:\n`not_collected_for_resource` (the resource never reports it; `similar_keys` lists up to 10 of its keys in the same group,\nonly keys that `sanitize()` leaves unchanged and that are at most 200 characters — when some were dropped, `detail` says\n`N key(s) in the same group were omitted: they contain control characters or exceed 200 characters.`),\n`no_data_in_window` (reported, but no points in the window), `resource_reports_no_stat_keys`, or `undetermined`\n(the resource's stat-key list could not be read, or some of its rows are in an unrecognised form — `detail` then says\n`N of M 'stat-key' rows in an unrecognised form`). `metric_key` is sanitized. `stat_keys_on_resource` is `null` unless something is missing.\nA missing key is not a zero. (Before this change the output was a bare object keyed by metric.)\n\nWith `--summary` (MCP `summary=true`) `mode` is `summary` and `metrics` is replaced by `summary`: per key `n`, `min`,\n`max`, `avg`, `latest`, `first_timestamp_ms`, `latest_timestamp_ms`, `change_count`, `change_points` (each\n`{timestamp_ms, from, to}` where the value differed from the point before — e.g. `badge|health` 100 → 25; at most 50,\nthe most recent kept, `change_points_truncated` says when more were dropped) and `non_numeric_points`. `missing` is\nunchanged. A continuously varying metric such as CPU usage changes at almost every point, so read `change_count` there.\n\n### `vmware-aria resource health`\n\nGet the health, risk and efficiency badges for a resource — and, for a service object, its service state.\n\n```\nvmware-aria resource health <resource-id> [OPTIONS]\n```\n\n**Output**: JSON with `name`, `kind`, and each badge's score (0–100) and color. The badges score the alerts attached to\nthat object, not a service's own state: on Aria Operations 8.18.7 the `mem` and `system` children of a vCenter app\nobject showed HEALTH GREEN 100 while `SERVICE|STATUS` was `orange` and `SERVICE|AVAILABILITY` was 0, because the alert is\nraised on the parent. For a kind containing `SERVICE` (e.g. `VCENTER_APPLIANCE_HEALTH_SERVICES`) the output adds `service`:\n`status` (`SERVICE|STATUS`, e.g. `green` / `orange`), `availability` (latest `SERVICE|AVAILABILITY` in the last hour),\n`available` (`true` for 1, `false` for 0, `null` for anything else or nothing read), `read_errors` (why a value is\nunknown) and `note`. `service` is `null` for other kinds.\n\n### `vmware-aria resource top`\n\nList top resource consumers by metric.\n\n```\nvmware-aria resource top [OPTIONS]\n\nOptions:\n  --metric TEXT     Metric key to rank by [default: cpu|usage_average]\n  --kind -k TEXT    Resource kind [default: VirtualMachine]\n  --top -n INT      Number of top consumers [default: 10]\n  --target -t TEXT  Target name\n```\n\n**Output**: Table with rank, name, value, unit. `value` is the average of the metric's points over the last hour (5-minute AVG rollup) — the number Aria Operations ranks by — and rows keep Aria's order, descending by `value`. The MCP tool `get_top_consumers` also returns `latest_value`, the most recent point. Resources with no data for the metric in the last hour are left out, not ranked at zero; `excluded_no_data` (MCP) counts the ones the ranking listed with no points. A yellow hint under the table says when that made the list shorter (or when no resources of that kind exist). When no-data rows took slots of a full `--top`, the result is marked truncated and the hint says to raise top_n.\n\n### `vmware-aria resource keys`\n\nList the metric keys a resource reports (with name and unit), or the keys a resource kind defines. Look keys up here before `resource metrics` or `resource top` instead of guessing.\n\n```\nvmware-aria resource keys [RESOURCE_ID] [OPTIONS]\nvmware-aria resource keys <vm-id> --filter 'mem|'\nvmware-aria resource keys --kind HostSystem --filter cpu\n\nArguments:\n  RESOURCE_ID           Resource UUID; omit and pass --kind for a kind's definitions\n\nOptions:\n  --kind -k TEXT        Resource kind, e.g. VirtualMachine\n  --adapter-kind TEXT   Adapter kind for --kind [default: VMWARE]\n  --filter -f TEXT      Substring of key or name, e.g. 'mem|'\n  --limit -n INT        Page size, 1-500 [default: 100]\n  --offset INT          Rows to skip; the next-page offset is printed below the table [default: 0]\n  --target -t TEXT      Target name\n  --config -c PATH      Config file path\n```\n\nPass exactly one of RESOURCE_ID or `--kind`; both or neither is a usage error (exit 2).\n\n**Output**: Table titled with the resource kind — Key, Name, Unit, and for a resource a Definition column: `found`, `found_by_instance` (an instanced key such as `guestfilesystem:/boot|usage` matched to `guestfilesystem|usage`), `not_defined_for_kind`, or `not_read` (the kind's definitions could not be read, so name and unit are unknown, not absent; a yellow note says why). A yellow line counts keys not defined for the kind. With `--kind` the table lists what the kind defines — a defined key is not collected on every resource. A nonexistent resource ID is an HTTP 404 error, not an empty table. The MCP tool `list_metric_keys` returns the envelope plus `source`, `resource_kind`, `adapter_kind`, `definitions_status` (`read` / `undetermined`), `definitions_note`, `unjoined_count`, `unjoined_keys` and `next_offset`.\n\n### `vmware-aria resource properties`\n\nList a resource's current properties (power state, parent host, extraConfig, ...).\n\n```\nvmware-aria resource properties <resource-id> [OPTIONS]\nvmware-aria resource properties <vm-id> --name 'summary|'\n\nOptions:\n  --name TEXT       Substring of the property name, e.g. 'summary|'\n  --limit -n INT    Page size, 1-500 [default: 100]\n  --offset INT      Rows to skip; the next-page offset is printed below the table [default: 0]\n  --target -t TEXT  Target name\n  --config -c PATH  Config file path\n```\n\n**Output**: Table of Name and Value, sorted by name — e.g. `summary|parentHost`, `summary|parentVcenter`, configured CPU and memory, `config|extraConfig|mem_hotadd`. These are current values; use `resource metrics` for time series. A failed or unrecognised read is an error, never an empty table; a missing resource is HTTP 404.\n\n### `vmware-aria resource relationships`\n\nList resources related to a resource (parents and children).\n\n```\nvmware-aria resource relationships <resource-id> [OPTIONS]\nvmware-aria resource relationships <vm-id> --type PARENT\n\nOptions:\n  --type TEXT       ALL, PARENT or CHILD [default: ALL]\n  --limit -n INT    Page size, 1-500 [default: 100]\n  --offset INT      Rows to skip; the next-page offset is printed below the table [default: 0]\n  --target -t TEXT  Target name\n  --config -c PATH  Config file path\n```\n\n**Output**: Table of Direction (`parent`, `child`, `both`, `other`, or `unknown`), Kind, Name, ID, sorted by kind and name. `--type` must be exactly `ALL`, `PARENT` or `CHILD` in upper case; `ANCESTOR` and `DESCENDANT` are refused (exit 2) — Aria Operations 8.18.7 answers HTTP 400 for them — so walk further by running the command again on a returned ID. With `ALL`, direction is `unknown` and a yellow note says why when the PARENT/CHILD lists could not be read. The MCP tool is `get_resource_relationships`.\n\n---\n\n## Alert Commands\n\n### `vmware-aria alert list`\n\nList alerts.\n\n```\nvmware-aria alert list [OPTIONS]\n\nOptions:\n  --active / --all          Active alerts only vs all [default: active]\n  --criticality TEXT        Filter: INFORMATION, WARNING, IMMEDIATE, CRITICAL\n  --limit -n INT            Max results [default: 50]\n  --json                    Print the result envelope as JSON instead of a table\n  --target -t TEXT          Target name\n```\n\n**Output**: Table with ID, Name, Criticality, Status, Started (UTC), Resource (name), Resource ID. IDs are never\nshortened. A terminal narrower than 160 columns cannot hold both UUID columns, so there each alert prints as a short\nblock instead (ID, criticality, status and start time; name; resource name and ID). `--json` prints every row field,\nincluding the start and update times in ISO-8601 UTC (the fields ending `_time_utc`) beside the millisecond times.\nThe start-time field there is `start_time_utc`. Names and kinds come from one batched `GET /resources` lookup per page. A resource whose name could not be resolved prints as `?` — unknown, not \"no resource\" — and a yellow note under the table says how many could not be retrieved, were not returned (deleted or stale), or were returned with no name in Aria Operations. If the lookup answers with rows that were not requested (the appliance ignored the id filter), requested ids it left out count as could not be retrieved — retry — not as deleted.\n\n### `vmware-aria alert get`\n\nGet full alert details with contributing (triggered) symptoms. Recommendations are attached to the alert definition, not the alert. `get_alert` carries the resource ID only — resolve the name via `vmware-aria resource get <id>` or `alert list`.\n\nSymptoms that carry no name or severity themselves (all of them on Aria Operations 8.18.7) take both from their symptom definition, fetched in one batched `GET /symptomdefinitions` lookup. Each symptom has `definition_lookup`: `resolved`, `not_needed`, `not_found`, `failed`, or `no_definition_id`. A `symptom_definitions_note` key appears when some did not resolve, or when symptoms carry no definition id to look a missing name or severity up by; an empty name there means unknown.\n\nEach symptom also names the object it is on: `resource_id`, `resource_name`, `resource_kind`, `stat_key`, and\n`condition` filled from the symptom instance's message (e.g. `HT not equal 0 != 1`). On 8.18.7 the contributing-symptom\npayload carries no resource id, so it is read from the symptom instance in `GET /symptoms`; that endpoint ignores its\n`id` filter there, so the tool walks the collection and keys rows by id. For \"vCenter app health is affected\" this names\nthe services that are down (e.g. `mem`, `system`).\n\nCost and scope: the walk runs for every alert whose symptoms carry no resource id — on 8.18.7 that is every alert,\nnot only vCenter app alerts. It costs one `GET /symptoms` per page of the appliance's whole symptom list (1,000 per\npage; 78 symptoms on the lab = one request), stops as soon as every symptom is found, and is capped at 20,000\nsymptoms; plus one batched `GET /resources`. `alert acknowledge` and `alert cancel` read the alert for their\nbefore-state without the walk. A page that fails keeps the symptoms found before it; a server that returns no total\ncount leaves the unfound ones `failed` (unknown), not `not_found`. `resource_lookup` is `not_needed`, `resolved`, `not_found` (every\npage was read and the instance was not there), `failed` (the read failed or stopped early), `no_symptom_id`, or\n`instance_names_no_resource`. A `symptom_resources_note` key appears when some could not be read — an empty\n`resource_id` there is unknown, not absent.\n\nThe start, update and cancel times are each returned twice: as epoch milliseconds (the fields ending `_time_ms`) and\nas ISO-8601 UTC (the fields ending `_time_utc`). The cancel time is `0` in milliseconds and `null` in UTC when the\nalert was never cancelled.\n\n```\nvmware-aria alert get <alert-id> [OPTIONS]\n```\n\n### `vmware-aria alert acknowledge`\n\nAcknowledge an alert (marks as seen, does not close it).\n\n```\nvmware-aria alert acknowledge <alert-id> [OPTIONS]\n\nOptions:\n  --yes -y          Skip confirmation prompt\n  --target -t TEXT  Target name\n```\n\n**Audit logged**: yes.\n\n### `vmware-aria alert cancel`\n\nCancel (dismiss) an alert. **Asks twice** unless `--yes` is given.\n\n```\nvmware-aria alert cancel <alert-id> [OPTIONS]\n\nOptions:\n  --yes -y          Skip both confirmation prompts\n  --target -t TEXT  Target name\n```\n\n**Audit logged**: yes. Cancelled alerts will not re-trigger unless the underlying condition recurs.\n\n### `vmware-aria alert definitions`\n\nList alert definition templates.\n\n```\nvmware-aria alert definitions [OPTIONS]\n\nOptions:\n  --name TEXT       Filter by name substring\n  --limit -n INT    Max results [default: 50]\n  --target -t TEXT  Target name\n```\n\n**Output**: Table with Name, Criticality (max severity across the definition's states), Resource Kind, Impact. Creating, enabling/disabling, and deleting alert definitions are MCP-only tools.\n\n### `vmware-aria alert notes`\n\nList the notes on an alert — who is handling it and what was done.\n\n```\nvmware-aria alert notes <alert-id> [OPTIONS]\n\nOptions:\n  --limit -n INT    Page size, 1-500 [default: 50]\n  --offset INT      Rows to skip; the next-page offset is printed below the table [default: 0]\n  --target -t TEXT  Target name\n  --config -c PATH  Config file path\n```\n\n**Output**: Table of Created (ms), Type (USER / SYSTEM), User, Note. An unknown alert ID is HTTP 404, not an empty table. A yellow note means Aria answered without a readable notes list: whether the alert has notes is unknown, not \"no notes\". The MCP tool `list_alert_notes` defaults to 100 rows.\n\n### `vmware-aria alert note-add`\n\nAdd a note to an alert. It does not change the alert's status or ownership — use `alert acknowledge` for that. **Asks once** unless `--yes` is given.\n\n```\nvmware-aria alert note-add <alert-id> <text> [OPTIONS]\nvmware-aria alert note-add <alert-id> \"Taking this: rebooting esx-03\" --dry-run\n\nOptions:\n  --dry-run         Print the API call without executing it\n  --yes -y          Skip the confirmation prompt\n  --target -t TEXT  Target name\n  --config -c PATH  Config file path\n```\n\n`--dry-run` prints the target and `POST /suite-api/api/alerts/<alert-id>/notes` with its body `{\"content\": \"<text>\"}`, and makes no connection. Empty text is refused (exit 2).\n\n**Output**: JSON with `alert_id`, `action`, `created` (the stored note) and `confirmation_note`. When `created` is null Aria did not confirm the note — run `alert notes` before adding it again, because every call adds a note.\n\n**Audit logged**: yes. Risk low; there is no undo for a note.\n\n### `vmware-aria alert recommendations`\n\nShow the prioritized recommendations for an alert, from its alert definition.\n\n```\nvmware-aria alert recommendations <alert-id> [OPTIONS]\n\nOptions:\n  --target -t TEXT  Target name\n  --config -c PATH  Config file path\n```\n\n**Output**: JSON with `alert_id`, `alert_name`, `criticality`, `alert_definition_id`, `alert_definition_name`, `state_severity`, `status`, `recommendations` (each: `id`, `priority` — lower is more important — `description`, `action`, `lookup`) and `note`. The alert's definition state whose severity matches the alert's criticality is used; otherwise the definition's only state; otherwise the recommendations of every state are merged at their highest priority, `state_severity` is null and `note` says so.\n\n| `status` | Meaning |\n|----------|---------|\n| `found` | Every recommendation was read |\n| `partial` | Some texts could not be read — ids and priorities are listed; `description: null` is unknown, not blank |\n| `none_defined` | The definition defines none — `recommendations` is `[]` |\n| `unknown` | The definition could not be read — `recommendations` is `null`; never report it as \"no recommendations\" |\n\nAn alert that cannot be read is an error. On Aria Operations 8.18.7 a CRITICAL vCenter-app alert returned 7 prioritized recommendations. The MCP tool is `get_alert_recommendations`.\n\n---\n\n## Capacity Commands\n\n### `vmware-aria capacity overview`\n\nGet a capacity overview for a cluster.\n\n```\nvmware-aria capacity overview <cluster-id> [OPTIONS]\n```\n\n**Output**: JSON with group-level `capacity_remaining_pct` plus per-dimension (cpu/mem/diskspace) `capacity_remaining` and `time_remaining_days`. The percentage metric exists only at group level. Values are None while capacity analytics warm up.\n\n### `vmware-aria capacity remaining`\n\nGet remaining capacity headroom for a cluster or host.\n\n```\nvmware-aria capacity remaining <resource-id> [OPTIONS]\n```\n\n**Output**: JSON with group-level `capacity_remaining_pct` and per-dimension `remaining_value` (absolute, unit per dimension e.g. MHz/KB).\n\n### `vmware-aria capacity time-remaining`\n\nPredict how many days until capacity is exhausted.\n\n```\nvmware-aria capacity time-remaining <resource-id> [OPTIONS]\n```\n\n**Output**: JSON with projected days per capacity dimension (None while capacity analytics have no data).\n\n### `vmware-aria capacity rightsizing`\n\nList VM rightsizing recommendations.\n\n```\nvmware-aria capacity rightsizing [OPTIONS]\n\nOptions:\n  --resource-id TEXT   Scope to a specific VM UUID\n  --limit -n INT       Max results [default: 20]\n  --target -t TEXT     Target name\n```\n\n**Output**: Table with VM name, power state (`template` for templates), sizing status, `vCPU now→rec`, `Mem GiB now→rec`, `Disk GB`, and `Act.` (actionable), from the `OnlineCapacityAnalytics|{cpu,mem,diskspace}|recommendedSize` metrics. Status `reclaimable` means the engine publishes 0 for the VM — that is not a recommendation of zero; `none published` means the VM either needs no resizing or was never scored, which the appliance does not distinguish.\n\nThe raw recommendations are MHz (cpu), KB (memory) and GB (disk) — verified on Aria Operations 8.18.7. The table converts them: CPU MHz is divided by the VM's own MHz per vCPU (`cpu|speed` / `numCpu`) and rounded up — except that a result at most 0.01 vCPU (1% of one core's MHz) above a whole number of cores counts as that number, absorbing float noise between the engine's MHz and `cpu|speed`; a real fraction above that still rounds up (1.5 → 2). Memory is shown in GiB. Arrows mark direction against the current configuration: `↓` oversized, `↑` undersized, `=` right-sized. Memory within 1% of the configured size is right-sized. Disk has no direction.\n\n`Act.` is `yes` only when the power state was read as `Powered On`, the template flag was read as false, and CPU or memory is off its recommendation; an unknown power state or template flag is never taken as running. Per-VM caveats print under the table: powered off, template, power state / template flag not published, a power state other than `Powered On`, current size not published, disagreement between `recommendedSize` and the engine's own `summary|oversized|*` / `summary|undersized|*` statistics, and — for every reduction — check the vendor minimum size first (appliances cannot be identified reliably from the API).\n\nThe MCP tool returns the same rows as JSON: `recommended_cpu` / `recommended_memory` / `recommended_diskspace` (raw), `recommended_units`, `sizing_status`, `current_vcpus`, `cpu_mhz_per_vcpu`, `recommended_vcpus`, `cpu_direction`, `current_memory_kb`, `memory_direction`, `power_state`, `is_template`, `product_name` (only when the VM publishes a vApp product), `aria_verdict`, `recommendation_range`, `recommendation_stable`, `actionable`, `caveats`, plus the top-level `properties_note` and `history_note`.\n\nWhether a recommendation has settled: two more bulk queries read each VM's daily low and high of the three `recommendedSize` keys over the last 7 days (`recommendation_range`: `window_days`, `days_with_data`, and `cpu_mhz` / `memory_kb` / `diskspace_gb` as `[low, high]`). If CPU or memory ranged by more than 5% of its high, `recommendation_stable` is false, `Act.` is `no`, and a caveat prints the range — for example `recommendation not settled: memory ranged 8.0–32.0 GiB over the last 3 day(s) of history`. The appliance may hold fewer days than the window, which `days_with_data` says. With no history returned, `recommendation_stable` is null. If the history read fails, `history_note` names the failure (the CLI prints it in yellow), both fields are null, and `Act.` is decided without them.\n\nIf the bulk property read (`POST /resources/properties/latest/query`) fails, the rows still come back from the stats, but `power_state`, `is_template`, `current_vcpus`, `current_memory_kb`, `recommended_vcpus`, both directions and `product_name` are null — unknown, not unpublished — no row is actionable, and each row carries one caveat starting `VM properties could not be read`. `properties_note` (null when the read succeeded) names the failure — the HTTP status, or no HTTP response — and the CLI prints it in yellow under the table.\n\n---\n\n## Anomaly Commands\n\n### `vmware-aria anomaly list`\n\nList per-resource anomaly counts (`System Attributes|total_alarms` Total Anomalies metric — the public API does not expose the UI's anomalous-metrics list).\n\n```\nvmware-aria anomaly list [OPTIONS]\n\nOptions:\n  --resource-id TEXT   Scope to a specific resource\n  --limit -n INT       Max VMs to scan when listing [default: 20]\n  --target -t TEXT     Target name\n```\n\n**Output**: Table with resource name (or ID) and anomaly count; without `--resource-id`, only resources with non-zero counts are shown, sorted descending.\nAria's key catalogue names `System Attributes|total_alarms` \"Total Anomalies\". It is not the alert count, which is a separate key, `System Attributes|total_alert_count` (verified on 8.18.7: vcsa read 5 anomalies with no alerts).\n\n### `vmware-aria anomaly risk`\n\nGet risk badge score for a resource.\n\n```\nvmware-aria anomaly risk <resource-id> [OPTIONS]\n```\n\n**Output**: JSON with risk score (0–100) and color (from the resource's `badges[]` array). For contributing causes, inspect the resource's active alerts.\n\n---\n\n## Health Commands\n\n### `vmware-aria health status`\n\nCheck Aria Operations platform health: node, each service, and the release.\n\n```\nvmware-aria health status [OPTIONS]\n```\n\n**Output**: Console summary with the assessment, the node status, the version (e.g. `VMware Aria Operations 8.18.7 — 8.x line`, or `unknown (<version_error>)`), a per-service table (Service / Health / Details, from `GET /deployment/node/services/info`) or `Services: not read (<reason>)`, and details (always printed).\n\n| Assessment | Meaning |\n|------------|---------|\n| `HEALTHY` | Node reports ONLINE and every listed service reports OK — or node reports ONLINE and the per-service breakdown was not read, in which case details say no service was checked individually |\n| `DEGRADED` | At least one service OK and at least one ERROR — the platform still answers |\n| `DOWN` | Every service reports ERROR |\n| `UNKNOWN` | The observations do not settle it: node not ONLINE and the breakdown not read; a service reports a state other than OK or ERROR (and it is not DEGRADED); or node not ONLINE while every service reports OK |\n\nA services reply with no service objects in it counts as not read.\n\nThe node status is OFFLINE (and `/deployment/node/status` answers HTTP 503) whenever any one service is not running, so OFFLINE alone is not an outage. On Aria Operations 8.18.7 a node with only `LOCATOR` not OK reads DEGRADED. The MCP tool `get_aria_health` returns `assessment`, `overall_status`, `healthy` (assessment is HEALTHY), `system_time_ms`, `services` (null when unreadable, with `services_error`), `services_not_ok`, `services_unrecognized`, `release_name`, `product_name`, `product_version`, `product_line`, `build_number`, `version_error`, and `details`. `details` is composed: the node status (with `(HTTP 503 at /deployment/node/status)` and the node's own details, up to 300 characters), the assessment in words, `Not OK: …` naming services in ERROR, and `services_error`. A version or service list that cannot be read — including a 2xx body that is not JSON — is reported in `version_error` / `services_error`, not raised; `release_name` and `product_name` are sanitized. A node-status failure other than HTTP 503 still raises.\n\n### `vmware-aria health collectors`\n\nList collector groups and member status.\n\n```\nvmware-aria health collectors [OPTIONS]\n```\n\n**Output**: Per-group tables listing collector ID, name, state (UP/DOWN), and local flag (marks the built-in collector on the Aria node).\n\n### `vmware-aria health node`\n\nAria node memory, swap, heap and watchdog restarts, with a memory-pressure indicator. Use it when `health status` shows a service in ERROR or the Aria UI/API is slow.\n\n```\nvmware-aria health node [OPTIONS]\nvmware-aria health node --hours 72 --json\n\nOptions:\n  --hours INT       Window for min/avg/max, 1-720 hours [default: 24]\n  --json            Print the full result as JSON\n  --target -t TEXT  Target name\n  --config -c PATH  Config file path\n```\n\nReads Aria's own self-monitoring objects (`vC-Ops-Node`, `vC-Ops-Watchdog`).\n\n**Output**: Per node — name and collection status; `Memory pressure: <level> — <basis>`; a table (Metric, Latest, Min, Avg, Max) of memory (`mem|total`, `mem|used`, `mem|free`, `mem|actualFree`, `mem|actualUsed`), swap (`swap|total`, `swap|used`, `swap|free`), heap (`heap|MaxHeapSize`, `heap|CurrentHeapSize`, `heap|CommittedMemory`, `heap|NodeHeapMemoryRemaining`) and committed heap per component, with units from Aria's own statkey definitions (on 8.18.7: GB for memory and swap, MB for the heap sizes, % for `heap|NodeHeapMemoryRemaining`); then the latest watchdog restarts per service, and a yellow `Missing <key>: <reason> — <detail>` line for each key with no value (`not_reported`, `no_data` or `undetermined` — never shown as zero). Min/avg/max are over 5-minute averages, so a shorter spike is smoothed. Watchdog restarts print `unknown — <why>` when they could not be read; yellow `units_error` / `latest_error` / `window_error` / `watchdog_error` lines name reads that failed.\n\nMemory pressure is an indicator, not a diagnosis: HIGH when actual free memory (`mem|actualFree`) is below 10% of `mem|total`, ELEVATED below 20%, NORMAL at 20% or more, UNKNOWN when the readings do not settle it. On Aria Operations 8.18.7 after a memory upgrade the node read `mem|total` 15.61 GB (was 7.75), actual free 41%, NORMAL, and 9 watchdog services with 0 restarts. The MCP tool is `get_aria_node_resources` (`window_hours`).\n\n### `vmware-aria health adapters`\n\nAdapter instances, when each last collected, and whether that is stale. Use it for \"Objects are not receiving data\" or metrics that stopped updating.\n\n```\nvmware-aria health adapters [OPTIONS]\nvmware-aria health adapters --kind VMWARE\n\nOptions:\n  --kind TEXT       Adapter kind key, e.g. VMWARE (case-insensitive)\n  --limit -n INT    Page size, 1-500 [default: 100]\n  --offset INT      Rows to skip; the next-page offset is printed below the table [default: 0]\n  --json            Print the full result as JSON\n  --target -t TEXT  Target name\n  --config -c PATH  Config file path\n```\n\n**Output**: Table titled with the clock the ages are measured on (`appliance` — the node status `systemTime` — or `local` when that cannot be read): Name, Kind, Last collected (seconds ago), Interval (minutes), Stale (yes / no / unknown), Resources, Metrics; then each adapter's own message. Stale means the last collection is older than max(3 × the adapter's monitoring interval, 15 minutes); unknown means the timestamp or interval is missing. When `--kind` matches nothing, the kinds present are printed. An empty or unrecognised `GET /adapters` answer is an error, not \"no adapters\" — every deployment runs a self-monitoring adapter. A recent last collection does not prove every object behind the adapter receives data. `--json` adds `id`, `resource_kind`, `collector_id`, `collector_group_id`, `last_heartbeat_ms` and its age, `stale_basis`, and envelope-level `stale_adapters`, `staleness_unknown` and `adapter_kinds_present`. On Aria Operations 8.18.7 all 6 adapters were not stale. The MCP tool is `list_adapters`.\n\n---\n\n## Report Commands\n\n### `vmware-aria report definitions`\n\nList available report definition templates.\n\n```\nvmware-aria report definitions [OPTIONS]\n\nOptions:\n  --name TEXT       Filter by name substring\n  --limit -n INT    Max results [default: 50]\n  --target -t TEXT  Target name\n```\n\n**Output**: Table with Name, ID, Subject Type (resource kinds the template applies to), Owner.\n\n### `vmware-aria report generate`\n\nTrigger report generation from a definition template (async).\n\n```\nvmware-aria report generate <definition-id> --resources <id1,id2> [OPTIONS]\n\nOptions:\n  --resources TEXT  Comma-separated resource UUIDs — at least one is required\n                    (the Report API generates against a resource)\n  --target -t TEXT  Target name\n```\n\n**Audit logged**: yes. Returns the queued `report_id`; poll with `report get`.\n\n### `vmware-aria report list`\n\nList generated reports.\n\n```\nvmware-aria report list [OPTIONS]\n\nOptions:\n  --definition-id TEXT  Filter by definition UUID (applied client-side)\n  --limit -n INT        Max results [default: 20]\n  --target -t TEXT      Target name\n```\n\n### `vmware-aria report get`\n\nGet status and download URLs for a generated report.\n\n```\nvmware-aria report get <report-id> [OPTIONS]\n```\n\n**Output**: Status; when `COMPLETED`, the PDF `download_url` and `csv_url`.\n\n### `vmware-aria report delete`\n\nDelete a generated report (the definition and schedules remain intact).\n**Irreversible — asks twice** unless `--yes` is given.\n\n```\nvmware-aria report delete <report-id> [OPTIONS]\n\nOptions:\n  --yes -y          Skip both confirmation prompts\n  --target -t TEXT  Target name\n```\n\n**Audit logged**: yes.\n\n---\n\n## Maintenance Commands\n\nMaintenance stops Aria alerting on a resource and collecting its data. `start` and `end` **ask once** unless `--yes` is given; `--dry-run` prints the API call and makes none — no connection, no request. Both are audited with the state before and after, and governed by vmware-policy under the MCP tool names `start_resource_maintenance` / `end_resource_maintenance` (risk medium).\n\n### `vmware-aria maintenance start`\n\nPut a resource in maintenance.\n\n```\nvmware-aria maintenance start <resource-id> [OPTIONS]\nvmware-aria maintenance start <host-id> --duration 120\nvmware-aria maintenance start <host-id> --duration 60 --dry-run\nvmware-aria maintenance start <host-id> --end $(( ($(date +%s) + 7200) * 1000 ))\n\nArguments:\n  resource-id  Resource UUID (from `vmware-aria resource list`) (required)\n\nOptions:\n  --duration INT    Window length in minutes\n  --end INT         Window end, epoch milliseconds\n  --dry-run         Print the API call without executing it\n  --yes -y          Skip the confirmation prompt\n  --target -t TEXT  Target name\n  --config -c PATH  Config file path\n```\n\nWith `--duration` (1–525600 minutes) or `--end` (epoch **milliseconds**, in the future — the example above computes two hours from now in any POSIX shell; a fixed timestamp stops working once it passes) the resource is `MAINTAINED` for that window and returns to its prior state when it expires; give one, not both. With neither it enters manual maintenance (`MAINTAINED_MANUAL`) that lasts until `maintenance end` — easy to forget, so prefer a window. An invalid window is refused (exit 2) before anything is sent. `--dry-run` prints `PUT /suite-api/api/resources/<resource-id>/maintained` with the `duration` or `end` query parameter.\n\n**Output**: JSON with `requested` (`mode` timed / manual, `duration_minutes`, `end_time_ms`), `before` and `after` (each: `name`, `kind`, `adapter_states`, `in_maintenance`, `maintenance_mode`, `note`, `read_error`), `confirmed` (true / false / null — null means the state could not be read afterwards, which is unknown, not failure) and `note`.\n\n**Audit logged**: yes, with before and after state. Undo: `maintenance end`.\n\n### `vmware-aria maintenance end`\n\nTake a resource out of maintenance: Aria resumes alerting and collection.\n\n```\nvmware-aria maintenance end <resource-id> [OPTIONS]\n\nOptions:\n  --dry-run         Print the API call without executing it\n  --yes -y          Skip the confirmation prompt\n  --target -t TEXT  Target name\n  --config -c PATH  Config file path\n```\n\nRefused (exit 2) only when the resource is known not to be in maintenance — an adapter reports a state such as `STARTED` or `STOPPED`, so there is nothing to end. When the state is unknown (it cannot be read, or an adapter reports `UNKNOWN` / `NONE`) the call proceeds and `before` says unknown. `--dry-run` prints `DELETE /suite-api/api/resources/<resource-id>/maintained`.\n\n**Output**: JSON with `before`, `after`, `confirmed` (true once the resource no longer reports maintenance, false when it still does, null when unknown) and `note`.\n\n**Audit logged**: yes, with before and after state. The MCP undo (`start_resource_maintenance`) is recorded only when the resource was known to be in maintenance before; it re-enters manual maintenance — the end of a timed window is not restored.\n\n### `vmware-aria maintenance schedules`\n\nList maintenance schedules.\n\n```\nvmware-aria maintenance schedules [OPTIONS]\n\nOptions:\n  --resource-id TEXT  Only schedules for this resource\n  --limit -n INT      Page size, 1-500 [default: 50]\n  --offset INT        Rows to skip; the next-page offset is printed below the table [default: 0]\n  --target -t TEXT    Target name\n  --config -c PATH    Config file path\n```\n\n**Output**: Table of Name, Type (ONCE / DAILY / WEEKLY / MONTHLY / YEARLY), Start (hour:minute and time zone), Duration (min), Recurrence, Expires (date, or `after N runs`), ID. A schedule does not list the resources it applies to — use `--resource-id`. A yellow note means the list could not be read: the schedules are unknown, not absent. The MCP tool `list_maintenance_schedules` defaults to 100 rows.\n\n---\n\n## Fleet Commands (VCF Operations 9.1)\n\nRead-only fleet / diagnostics queries added for VCF Operations 9.1. The suite-api\n*paths* are verified against the VCF 9.1 OpenAPI; the response *schemas* are read\ndefensively, and an unrecognised shape returns an empty result carrying a `note`\nthat the empty result is unconfirmed (never a silent \"none\"). The `promql`\nsub-command reaches the real-time metrics (VODAP) service on a separate base\n(`/data-query-service`) whose prefix is **INFERRED and not yet confirmed on real\nhardware** — every result carries `base_path_confirmed: false`.\n\n### `vmware-aria fleet certificates`\n\nList certificate status/expiry across the VCF fleet.\n\n```\nvmware-aria fleet certificates [OPTIONS]\n\nOptions:\n  --limit -n INT    Max rows (default 50)\n  --target -t TEXT  Target name\n```\n\n### `vmware-aria fleet passwords`\n\nList managed password-account status across the VCF fleet (read-only; does not rotate).\n\n```\nvmware-aria fleet passwords [OPTIONS]\n\nOptions:\n  --limit -n INT    Max rows (default 50)\n  --target -t TEXT  Target name\n```\n\n### `vmware-aria fleet domains`\n\nList SDDC/workload domains behind one registered VCF integration. The integration\nUUID comes from the Operations Integrations page.\n\n```\nvmware-aria fleet domains <integration-id> [OPTIONS]\n\nOptions:\n  --limit -n INT    Max rows (default 50)\n  --target -t TEXT  Target name\n```\n\n### `vmware-aria fleet findings`\n\nList Operations diagnostic findings (not compliance — use vmware-harden for that).\n\n```\nvmware-aria fleet findings [OPTIONS]\n\nOptions:\n  --severities TEXT  Comma-separated, e.g. CRITICAL,WARNING\n  --categories TEXT  Comma-separated category filter\n  --types TEXT       Comma-separated findingType filter\n  --limit -n INT     Max rows (default 50)\n  --target -t TEXT   Target name\n```\n\n### `vmware-aria fleet promql`\n\nRun a real-time PromQL instant query against the VCF 9.1 VODAP service. Base path\nINFERRED — confirm against a live appliance (`base_path_confirmed: false`).\n\n```\nvmware-aria fleet promql <query> [OPTIONS]\n\nOptions:\n  --time TEXT       Evaluation timestamp (RFC3339 or Unix seconds)\n  --source TEXT     Data-source id to scope the query\n  --limit -n INT    Max result series (default 50)\n  --target -t TEXT  Target name\n```\n\n---\n\n## Exit Codes\n\n| Code | Meaning |\n|------|---------|\n| 0 | Success |\n| 1 | Error (API failure, auth failure, validation error) |\n| 2 | Usage error (invalid arguments) |\n\n`vmware-aria doctor` exits 0 if all checks pass, 1 if any fail. WARN rows do not fail.\n\nFile v1.17.0:references/investigation-protocol.md\n\n# Investigation Protocol — Causal Chain Root Cause Analysis\n\nA protocol for AI agents performing diagnostic investigations on VMware infrastructure (alarms, performance regressions, availability incidents). Adopted from Enterprise Harness Engineering, drawing on 5 Whys, Google SRE, ITIL, and NASA Fault Tree Analysis.\n\n## When to Apply\n\nUse this protocol whenever the user asks:\n\n- \"Why is X slow / failing / down?\"\n- \"What caused this alarm / alert / incident?\"\n- \"Investigate / diagnose / debug …\"\n- Any open-ended question that requires identifying a root cause rather than just reading state.\n\nDo NOT apply for:\n\n- Simple state lookups (\"is the VM on?\", \"list datastores\")\n- Operational requests (\"clone this VM\", \"create this rule\")\n- Configuration questions (\"what's the default for X?\")\n\n## The Four Criteria for Root Cause Completeness\n\nA diagnostic conclusion is **incomplete** unless ALL four criteria are satisfied. The agent must self-check against each one before outputting a report.\n\n### 1. Falsifiability (可证伪性)\n\nThe root cause must be independently measurable and verifiable. If you cannot test it, it is a hypothesis, not a root cause.\n\n- ✅ \"Datastore latency exceeded 50ms because IOPS hit the SAN cap of 10,000\" — directly testable via `get_resource_metrics` on the host (`datastore|numberReadAveraged_average`, `datastore|numberWriteAveraged_average`)\n- ❌ \"Network was congested\" — too vague to verify\n\n### 2. Sufficiency (充分性)\n\nRemoving the root cause must make the symptom disappear. If the symptom persists after the supposed fix, the cause was wrong or partial.\n\n- ✅ \"Deleting the orphaned snapshot freed 200 GB and the alarm cleared within 60 seconds\"\n- ❌ \"Restarted the VM and the issue went away\" — correlation, not causation\n\n### 3. Necessity (必要性)\n\nThe symptom must occur whenever the root cause is present. If the same condition exists elsewhere without the symptom, you have not found the true root cause.\n\n- ✅ \"Every cluster with 80%+ memory overcommit shows the same vMotion stall\"\n- ❌ \"Only this one VM has the issue\" — without explaining why this VM specifically\n\n### 4. Mechanism (机制性)\n\nYou must explain the propagation chain: root cause → propagation → amplification → impact. A single point claim with no mechanism is a guess.\n\n- ✅ \"Snapshot delta files filled the datastore (root) → VM I/O blocked on write (propagation) → guest filesystem went read-only (amplification) → application timeout (impact)\"\n- ❌ \"The datastore was full\" — describes a state, not a chain\n\n## Investigation Workflow — Up to Three Depth Rounds\n\n### Round 1 — Initial Hypothesis\n\n1. Gather symptoms via L1/L2 read tools (alarms, metrics, events, logs)\n2. Form an initial causal chain hypothesis\n3. Apply the four criteria\n\nIf all four pass → output report.\nIf any criterion fails → proceed to Round 2 with that criterion as the focus.\n\n### Round 2 — Targeted Deepening\n\n1. Identify which criterion failed\n2. Gather additional evidence aimed specifically at that criterion (e.g. failed Necessity → compare against unaffected peers; failed Mechanism → trace next propagation step)\n3. Refine the causal chain\n4. Re-apply the four criteria\n\nIf all four pass → output report.\nIf any still fails → proceed to Round 3.\n\n### Round 3 — Final Deepen or Escalate\n\n1. If a deeper cause is reachable, gather final evidence and finalize the chain\n2. If evidence is unavailable, system-bounded, or beyond the agent's tool surface, **escalate to a human** and explicitly label the conclusion as `⚠️ INCOMPLETE — <criterion> unsatisfied`\n3. **Never** silently output a partial conclusion as if it were complete\n\n## Output Format\n\nEvery investigation report must structure findings exactly as:\n\n```\n🔴 [ROOT CAUSE]   <falsifiable, mechanism-explained statement>\n  → [PROPAGATION] <how the root cause spread to neighboring systems>\n    → [AMPLIFICATION] <what made the impact worse, if applicable>\n      → [IMPACT]    <observable user / business / SLA effect>\n\n✅ Falsifiability:  <evidence — metric name, log query, command output>\n✅ Sufficiency:     <evidence or stated counterfactual>\n✅ Necessity:       <evidence or peer comparison>\n✅ Mechanism:       <see propagation chain above>\n```\n\nIf any criterion is unmet, mark it `⚠️ INCOMPLETE — <reason>` and state explicitly what additional evidence would be required to satisfy it.\n\n## Anti-Patterns\n\n| ❌ Pattern | Why it fails |\n|---|---|\n| \"thanos-cn unreachable\" alone | Describes symptom; does not answer **why** unreachable |\n| \"Datastore full\" alone | No propagation, no impact chain |\n| \"Try restarting it\" | Skips diagnosis entirely |\n| \"Probably the network\" | Not falsifiable |\n| Stopping at the first plausible cause | Skips Necessity check |\n| Silent partial conclusion | Hides incompleteness from the user |\n\n## Worked Examples\n\n### Bad — Incomplete Diagnosis\n\n> \"VM is slow because the host is busy.\"\n\nMissing:\n- **Falsifiability**: which metric, what threshold?\n- **Necessity**: why this VM only?\n- **Mechanism**: how does host load translate into VM slowness?\n\n### Good — Complete Diagnosis\n\n> 🔴 [ROOT] Host `esx-03` CPU ready time exceeds 15% (validated via `get_resource_metrics` on the host, `cpu|max_cpu_ready`)\n>   → [PROPAGATION] vCPU contention from 4-VM reservation collision in resource pool `prod-rp`\n>     → [AMPLIFICATION] DRS is in manual mode, so VMs are not rebalanced\n>       → [IMPACT] Application p99 latency doubled from 200 ms to 400 ms\n>\n> ✅ Falsifiability: `cpu|max_cpu_ready` (worst VM CPU Ready %) directly observable; threshold defined in vSphere docs\n> ✅ Sufficiency: vMotion `vm-A` off `esx-03` reduced ready time to 3% and p99 latency back to 200 ms\n> ✅ Necessity: only VMs in `prod-rp` with active reservations are affected; identical workloads in `staging-rp` are healthy\n> ✅ Mechanism: CPU Ready = vCPU waiting for pCPU → guest perceives as CPU starvation → app threadpool exhaustion → tail latency\n\n## Related Skills\n\nA complete investigation often chains across skills:\n\n- **vmware-aria** (this skill): metrics, alerts, anomaly detection — primary L1/L2 data source\n- [vmware-monitor](https://github.com/vmware-skills/VMware-Monitor): inventory, alarms, events — additional read-only data source, code-level safe\n- [vmware-aiops](https://github.com/vmware-skills/VMware-AIops): VM/host state, deployment history; can also remediate at L3+\n- [vmware-pilot](https://github.com/vmware-skills/VMware-Pilot): orchestrate the investigation itself as a multi-step Dispatcher → Subagent workflow\n\nThe agent should treat investigation as **read-heavy first**: gather across skills, reason centrally, only invoke L3+ write tools after the four criteria are satisfied AND the user has approved a remediation plan.\n\nFile v1.17.0:references/ops-playbooks.md\n\n# Ops Playbooks — Troubleshooting Paths with vmware-aria\n\nEach path below uses only commands this skill ships, in the order an operator would run them. Every read step was run against a live Aria Operations 8.18.7 appliance (2026-09-13); the examples quote what came back. The write steps — `alert note-add`, `alert acknowledge`, `maintenance start` and `maintenance end` — were not sent to it: they were run only with `--dry-run` or not at all, so what they return is described from the code and its tests, not observed. MCP tool names are in brackets where they differ from the command.\n\nQuote any argument that contains `|` (`'cpu|readyPct'`) — a shell reads a bare `|` as a pipe.\n\nFor a root cause (not just a symptom), finish with [`investigation-protocol.md`](investigation-protocol.md).\n\n---\n\n## 1. Aria Operations itself reports DEGRADED or OFFLINE\n\n**Question**: is the platform down, or is one service out while data still flows?\n\n1. Assessment and the failed service → `vmware-aria health status` [`get_aria_health`]\n   - `DEGRADED` = some services OK, some ERROR — not an outage. `DOWN` = no service OK. The node flag reads OFFLINE (HTTP 503) whenever any one service is not running, so OFFLINE alone proves nothing.\n   - The version line tells you 8.x or 9.x; fleet, findings and PromQL only exist on 9.0+.\n2. Is the node starved? → `vmware-aria health node --hours 24` [`get_aria_node_resources`]\n   - `Memory pressure` is HIGH below 10% actual free, ELEVATED below 20%. Also read swap used, heap per component, and watchdog restarts per service.\n   - Restarts of 0 with a service still failing means it is not crash-looping — it is stuck or misconfigured.\n3. Is collection still working? → `vmware-aria health adapters` [`list_adapters`]\n   - `Stale: yes` means the adapter's last collection is older than 3 intervals (at least 15 minutes).\n4. Service-specific follow-up happens on the appliance (console or SSH), not through the API.\n\n**Example**: `LOCATOR ERROR`, everything else OK → DEGRADED. `health node` showed 7.61 of 7.75 GB used and swap in use, so the VM was resized to 16 GB. Afterwards `health node` read 15.61 GB total, 41% actually free, pressure NORMAL, all watchdog restarts 0 — and LOCATOR was **still** in ERROR. Memory pressure was real but not the cause. The locator was in fact running (port 6061 listening, reachable by IP and by hostname, `locators=` matching the node IP, no \"Could not contact any of the locators\" in `analytics-*.log`), so the IP-change failure in Broadcom KB 404805 did not apply. What failed was the status check itself: `/storage/vcops/log/api.log` logged `GemfireLocatorStatusCheckVerifier` with `javax.net.ssl.SSLHandshakeException: No subject alternative DNS name matching aria-ops-01 found`. The check connects to the locator's JMX manager (port 1099, TLS 1.3 with client authentication) by the node's hostname, and the appliance certificate (`CN=vROps-slice-1`) lists only `localhost`, `127.0.0.1` and the node IP as Subject Alternative Names — so the handshake fails and the API sets the node OFFLINE while collection carries on. Check a node the same way: `grep -h -A1 \"nested exception is:\" /storage/vcops/log/api.log | sort | uniq -c`, then compare the certificate's SANs (`openssl s_client -connect 127.0.0.1:1099 </dev/null | openssl x509 -noout -ext subjectAltName`) with `hostname`. GemFire KB 439258 describes this handshake failure and its options (advertise a name that is on the certificate with `jmx-manager-hostname-for-clients`, or reissue the certificate); both change the appliance's configuration.\n\n## 2. Triage a critical alert\n\n**Question**: what is affected, why, and what does VMware recommend?\n\n1. Open alerts → `vmware-aria alert list --criticality CRITICAL` [`list_alerts`] — rows carry the affected resource's name and ID (the MCP tool also returns its `resource_kind`; the CLI table has no kind column)\n2. Symptoms → `vmware-aria alert get <alert-id>` [`get_alert`] — each symptom is named from its definition, with severity\n3. Recommended actions → `vmware-aria alert recommendations <alert-id>` [`get_alert_recommendations`]\n   - `status: found` lists them by priority; `none_defined` means the definition has none; `unknown` means they could not be read — not the same thing\n4. The affected object → `vmware-aria resource get <resource-id>` [`get_resource`] (MCP [`investigate_alert`] does steps 2 and 4 in one call and labels the two UUIDs)\n5. Where it sits → `vmware-aria resource relationships <resource-id> --type PARENT` [`get_resource_relationships`]\n6. Record who is on it → `vmware-aria alert notes <alert-id>` [`list_alert_notes`], then `vmware-aria alert note-add <alert-id> \"Investigating: …\" --dry-run`, and again without `--dry-run` [`add_alert_note`, write]\n7. When handled → `vmware-aria alert acknowledge <alert-id>` [`acknowledge_alert`, write]\n\n**Example**: \"vCenter app health is affected\" (CRITICAL) — symptom \"vCenter appliance health service is down\" (CRITICAL); resource `vCenter-192.168.60.16`, kind `VC_APP`, health 25 (RED); parents \"vCenter Health\" and adapter instance \"new VC\"; 7 recommendations, priority 1 \"Please check the Health Status of app\"; no notes yet.\n\n## 3. A VM is slow — find contention, not just consumption\n\n1. Candidates → `vmware-aria resource top --metric 'cpu|usage_average' --top 10` [`get_top_consumers`]\n   - `value` is the last-hour average Aria ranks by; resources with no data are left out, not ranked at zero\n2. Contention on a candidate → `vmware-aria resource metrics <vm-id> --metrics 'cpu|readyPct,mem|balloonPct,mem|swapped_average' --hours 24` [`get_resource_metrics`]\n   - CPU Ready >5% warning, >10% problem; balloon >0 = ESXi reclaiming memory; swapped >0 = severe\n3. A busy VM with low Ready is healthy; a quiet VM with high Ready is starving. Confirm with the host's view before blaming the VM.\n\n## 4. A metric comes back empty\n\n1. Ask → `vmware-aria resource metrics <id> --metrics '<key>' --hours 1`\n2. Read `missing[].reason` — never report it as zero:\n   - `not_collected_for_resource` — this resource never reports the key; `similar_keys` lists keys from the same group\n   - `no_data_in_window` — reported before, nothing in this window: widen `--hours`, then check `health adapters`\n   - `undetermined` — the key list could not be read\n3. Find the right key → `vmware-aria resource keys <id> --filter 'cpu|demand'` [`list_metric_keys`] — name and unit come from the kind's definitions\n\n**Example**: `cpu|demand_average` → `not_collected_for_resource` on a VM. `resource keys --filter 'cpu|demand'` returns `cpu|demandPct` (CPU|Demand, %) and `cpu|demandmhz` (CPU|Demand, MHz).\n\n## 5. Before resizing a VM\n\n1. Candidates → `vmware-aria capacity rightsizing --limit 20` [`list_rightsizing_recommendations`]\n   - Act only where `Act.` is yes; read every caveat under the table\n   - A yellow `properties_note` means the property read failed — power state and current size are unknown, so nothing is actionable\n2. Current facts → `vmware-aria resource properties <vm-id> --name powerState` and `--name hotadd` [`get_resource_properties`]\n   - `config|extraConfig|mem_hotadd: false` means the change needs a power-off window\n3. Check the vendor's minimum size before any reduction — appliances cannot be identified reliably from the API.\n4. Make the change with vmware-aiops (this skill does not change VMs); wrap the window in maintenance (path 6).\n5. Afterwards, re-read `capacity rightsizing` only once the analysis window reflects the new size.\n\n**Example**: the Aria appliance VM had just been resized from 8 GB to 16 GB after memory pressure. The next `capacity rightsizing` still recommended 16.0 → 9.0 GiB with `Act.` yes — the engine's window was mostly pre-change data, and the VM is a vendor appliance. Following it would have undone the fix.\n\n## 6. Planned maintenance on a monitored object\n\n1. Existing windows → `vmware-aria maintenance schedules` [`list_maintenance_schedules`]\n2. Preview → `vmware-aria maintenance start <resource-id> --duration 60 --dry-run` — prints `PUT /suite-api/api/resources/<id>/maintained?duration=60` without connecting\n3. Start → the same without `--dry-run` [`start_resource_maintenance`, write, medium] — omit `--duration` and `--end` for indefinite maintenance\n4. Do the work (for VM changes, vmware-aiops)\n5. End → `vmware-aria maintenance end <resource-id>` [`end_resource_maintenance`, write, medium] — refused only when the resource is known not to be in maintenance; when its state is unknown (unreadable, or an adapter reports `UNKNOWN` / `NONE`) it proceeds\n\nBoth write commands confirm once (`--yes` skips) and are audited to `~/.vmware/audit.db`.\n\nFile v1.17.0:references/setup-guide.md\n\n# Setup Guide\n\nComplete setup and security guide for `vmware-aria`.\n\n## Prerequisites\n\n- Python 3.10+\n- VMware Aria Operations 8.x (or vRealize Operations 8.x)\n- Network access to Aria Ops on port 443 (HTTPS)\n- Aria Operations credentials:\n  - Read operations: `ReadOnly` role minimum\n  - Alert acknowledge/cancel: `PowerUser` role or higher\n\n## Installation\n\n### Via uv (recommended)\n\n```bash\nuv tool install vmware-aria==1.17.0\n```\n\n### Via pip\n\n```bash\npip install vmware-aria==1.17.0\n```\n\n### From source\n\n```bash\ngit clone --branch v1.17.0 https://github.com/vmware-skills/VMware-Aria.git\ncd VMware-Aria\npip install -e .\n```\n\n## Configuration\n\n### 1. Create config directory\n\n```bash\nmkdir -p ~/.vmware-aria\n```\n\n### 2. Create config.yaml\n\n```bash\ncp config.example.yaml ~/.vmware-aria/config.yaml\n```\n\nEdit `~/.vmware-aria/config.yaml`:\n\n```yaml\ntargets:\n  prod:\n    host: aria-ops.example.com    # Aria Ops FQDN or IP\n    username: admin\n    port: 443\n    verify_ssl: true\n    auth_source: LOCAL            # LOCAL | LDAP | AD\n\ndefault_target: prod\n```\n\n### 3. Set password\n\n**Option A — .env file (recommended)**:\n```bash\ncat > ~/.vmware-aria/.env << 'EOF'\nVMWARE_ARIA_PROD_PASSWORD=your_password_here\nEOF\nchmod 600 ~/.vmware-aria/.env\n```\n\n**Option B — shell environment**:\n```bash\nexport VMWARE_ARIA_PROD_PASSWORD=your_password_here\n```\n\nPassword variable naming convention: `VMWARE_ARIA_<TARGET_UPPER>_PASSWORD`\n- Target `prod` → `VMWARE_ARIA_PROD_PASSWORD`\n- Target `aria-lab` → `VMWARE_ARIA_ARIA_LAB_PASSWORD`\n\n### 4. Verify setup\n\n```bash\nvmware-aria doctor\n```\n\nExpected output: All checks PASS. WARN rows (an unreadable version, a DEGRADED or UNKNOWN platform) do not fail the doctor; FAIL rows do.\n\n---\n\n## Multiple Targets\n\n```yaml\ntargets:\n  prod:\n    host: aria-prod.example.com\n    username: admin\n    port: 443\n    verify_ssl: true\n    auth_source: LOCAL\n\n  lab:\n    host: aria-lab.example.com\n    username: admin\n    port: 443\n    verify_ssl: true     # private CA? see \"TLS Certificate Verification\" below\n    auth_source: LOCAL\n\ndefault_target: prod\n```\n\nSet passwords for each target:\n```bash\nVMWARE_ARIA_PROD_PASSWORD=prod_pw\nVMWARE_ARIA_LAB_PASSWORD=lab_pw\n```\n\nUse `--target` to select:\n```bash\nvmware-aria resource list --target lab\nvmware-aria alert list --target prod\n```\n\n---\n\n## Authentication (LDAP / Active Directory)\n\nFor LDAP or AD authentication, update `auth_source` in config:\n\n```yaml\ntargets:\n  corp:\n    host: aria-ops.corp.example.com\n    username: jsmith@corp.example.com\n    port: 443\n    verify_ssl: true\n    auth_source: LDAP    # or AD\n```\n\nThe `auth_source` value must match the configured authentication source name in Aria Ops (Administration > Authentication Sources).\n\n---\n\n## TLS Certificate Verification\n\n`verify_ssl` defaults to `true`: a target that omits the key is verified. The\nclient trusts the public CA bundle shipped with Python's `certifi` package, **not**\nthe operating-system trust store, so adding your CA to macOS Keychain or\n`/etc/pki` does not change what `vmware-aria` accepts.\n\n**Appliance signed by a private or enterprise CA** — keep `verify_ssl: true` and\npoint `SSL_CERT_FILE` at the PEM certificate of the CA that signed the Aria\nOperations certificate:\n\n```bash\nexport SSL_CERT_FILE=/path/to/aria-ca.pem\nvmware-aria doctor\n```\n\nFor the MCP server, add `\"SSL_CERT_FILE\": \"/path/to/aria-ca.pem\"` to the\nserver's `env` block next to `VMWARE_ARIA_CONFIG`. `SSL_CERT_FILE` replaces the\ndefault bundle for the whole process; if that process must also reach public\nHTTPS (for example `uvx` resolving packages from PyPI), point it at a bundle that\ncontains both your CA and the public roots.\n\n**Isolated lab only** — `verify_ssl: false` disables certificate *and* hostname\nchecking for that one target. The username and password are then sent to\nwhatever answers at that address, so use it only on a lab network you control,\nnever for production or for a target reached over a network you do not trust.\n\n---\n\n## MCP Server Setup\n\n### With Claude Code\n\nAdd to `~/.claude.json` (or use `claude mcp add`):\n\n```json\n{\n  \"mcpServers\": {\n    \"vmware-aria\": {\n      \"command\": \"vmware-aria\",\n      \"args\": [\"mcp\"],\n      \"env\": {\n        \"VMWARE_ARIA_CONFIG\": \"/Users/<username>/.vmware-aria/config.yaml\"\n      }\n    }\n  }\n}\n```\n\n> v1.5.15+ recommends `vmware-aria mcp`. Pre-1.5.15 used the legacy\n> `vmware-aria-mcp` console script (still kept for backward compatibility).\n> If using `uvx --from vmware-aria==1.17.0 vmware-aria mcp` and you hit\n> `invalid peer certificate: UnknownIssuer` behind a corporate TLS proxy,\n> set `UV_NATIVE_TLS=true` or use the recommended form above.\n\n### With Cursor\n\nAdd to `.cursor/mcp.json` in your project (see `examples/mcp-configs/cursor.json`).\n\n### With Goose\n\nAdd to `~/.config/goose/config.yaml` (see `examples/mcp-configs/goose.json`).\n\n---\n\n## Docker Deployment\n\n### Build and run\n\n```bash\ndocker build -t vmware-aria .\ndocker run -i \\\n  -v ~/.vmware-aria:/root/.vmware-aria:ro \\\n  -e VMWARE_ARIA_CONFIG=/root/.vmware-aria/config.yaml \\\n  vmware-aria\n```\n\n### Using docker-compose\n\n```bash\n# Set password in your shell first\nexport VMWARE_ARIA_PROD_PASSWORD=your_password\n\ndocker-compose up\n```\n\nThe docker-compose.yml mounts `~/.vmware-aria` read-only into the container.\n\n---\n\n### Password obfuscation at rest\n\nOn first load, any plaintext `*_PASSWORD` value in `.env` is automatically\nrewritten to a grep-safe `b64:<encoded>` form and decoded transparently at\nruntime, so a casual `grep` of the file no longer reveals the password. Values\nare read and written through python-dotenv's own parser, so the stored secret\nnever drifts from what you configured (quotes, inline comments, and trailing\nwhitespace are handled correctly).\n\n> **This is obfuscation, not encryption.** Anyone who can read the file can\n> still decode it. For real secrecy at rest, do not store the password in `.env`\n> at all — inject it from a secret manager (HashiCorp Vault, CyberArk, AWS\n> Secrets Manager, or a Kubernetes Secret) into the `*_PASSWORD` environment\n> variable at process start. The code reads the env var either way.\n\n## Read-Only Operation\n\nTo run the agent read-only, give it a read-only Aria service account (RBAC).\n\n## Architecture\n\n```\nUser (natural language)\n  |\nAI Agent (Claude Code / Goose / Cursor)\n  | reads SKILL.md\nvmware-aria CLI or MCP server (stdio transport)\n  | Aria Operations Suite API (REST/JSON over HTTPS)\n  | POST /suite-api/api/auth/token/acquire → vRealizeOpsToken\nAria Operations Manager\n  |\nVMs / Hosts / Clusters / Datastores / Alerts / Capacity\n```\n\nThe MCP server uses stdio transport (local only, no network listener). Connections to Aria Ops use HTTPS on port 443 with vRealizeOpsToken authentication (6-hour sliding token validity, auto-refreshed).\n\n## Security Notes\n\n> **Disclaimer**: This is a community-maintained open-source project and is **not affiliated with, endorsed by, or sponsored by VMware, Inc. or Broadcom Inc.** \"VMware\" and \"Aria\" are trademarks of Broadcom.\n\n- **Source Code**: Fully open source at [github.com/vmware-skills/VMware-Aria](https://github.com/vmware-skills/VMware-Aria) (MIT). The `uv` installer fetches the `vmware-aria` package from PyPI, which is built from this GitHub repository. We recommend reviewing the source code and commit history before deploying in production.\n- **Never** store passwords in `config.yaml` — use env vars or `.env` file\n- `.env` file must be `chmod 600` (owner read/write only)\n- TLS verification is on by default (`verify_ssl: true`); trust a private CA with `SSL_CERT_FILE` rather than disabling verification (see [TLS Certificate Verification](#tls-certificate-verification))\n- The MCP server uses stdio transport — it runs locally only, no network listener\n- Audit log at `~/.vmware/audit.db` (SQLite WAL, via vmware-policy) records all write operations (alert acknowledge/cancel, alert definition management, report generate/delete)\n- Tokens have a 6-hour sliding validity (extended on each call); the client re-acquires automatically 60 seconds before expiry. Requests carry `Authorization: vRealizeOpsToken <token>`\n\n---\n\n## Troubleshooting\n\n### Connection refused on port 443\n\n```bash\n# Test connectivity\nvmware-aria doctor\n# Or manually, verifying the certificate against your CA\ncurl --cacert /path/to/aria-ca.pem https://aria-ops.example.com/suite-api/api/versions/current\n```\n\nCheck: firewall rules, VPN connectivity, Aria Ops service status. Any HTTP\nresponse (even 401) means the network path and certificate are fine; a curl\ncertificate error means the CA file is not the one that signed the appliance\ncertificate.\n\n### \"401 Unauthorized\" on token acquire\n\nVerify:\n1. Username and password are correct\n2. `auth_source` matches the authentication source name in Aria Ops\n3. The user account is not locked\n\n### \"the body is not JSON\" or \"response carried no 'token' field\"\n\nAria Operations answered with a success status but not with suite-api JSON — a\nlogin page, an SSO redirect, or a proxy in front of the node answers like this.\nEvery tool reports it as an error naming the method, path, HTTP status and\ncontent type (`NonJsonBodyError`) instead of a JSON decode traceback; during\ntoken acquisition it is reported as \"most likely not an Aria Operations suite-api\nendpoint\" (`NotSuiteApiError`). Check that `host` and `port` for the target reach\nthe Aria Operations node itself.\n\n### Self-signed certificate error\n\nExport the CA that signed the Aria Ops certificate as PEM and set\n`SSL_CERT_FILE=/path/to/aria-ca.pem` (see\n[TLS Certificate Verification](#tls-certificate-verification)). Installing it\ninto the system trust store has no effect: the client uses the `certifi` bundle.\n`verify_ssl: false` is for isolated labs only.\n\n### Metrics return empty list\n\nThe metric key may not apply to this resource kind, or collection has not started yet. The `missing` list in the output says which: `not_collected_for_resource` (try one of its `similar_keys`), `no_data_in_window`, `resource_reports_no_stat_keys`, or `undetermined` (the stat-key list could not be read or is in an unrecognised form). You can also browse available metric keys in the Aria Ops UI: navigate to the resource → Metrics tab.\n\nFile v1.17.0:skill-card.md\n\n## Description:\n\nUse this skill when an agent needs VMware Aria Operations or VMware VCF Operations data for metrics, alerts, capacity, anomalies, reports, maintenance state, platform health, adapter collection state, and operational recommendations.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[zw008](https://clawhub.ai/user/zw008)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and infrastructure operators use this skill to investigate VMware Aria Operations and VCF Operations environments, including resource metrics, active alerts, capacity forecasts, anomalies, reports, maintenance mode, platform health, and adapter collection state.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill uses Aria Operations credentials and can expose or act on production operational data.\n\nMitigation: Use a least-privilege Aria Operations account, preferably read-only unless alert, report, or maintenance writes are required.\n\nRisk: Production passwords may be stored in local configuration files.\n\nMitigation: Prefer a secret manager or session-only environment injection; if a local .env file is used, restrict permissions and avoid echoing credentials.\n\nRisk: Write-capable actions can acknowledge or cancel alerts, alter alert definitions, generate or delete reports, or start and end maintenance mode.\n\nMitigation: Review the blast radius and require explicit approval before write actions, especially when using CLI --yes or ending maintenance.\n\nRisk: The package is a third-party VMware Aria integration intended for operational environments.\n\nMitigation: Review the vmware-aria and vmware-policy package source before production deployment.\n\n## Reference(s):\n\n- [Setup Guide](artifact/references/setup-guide.md)\n- [Capabilities](artifact/references/capabilities.md)\n- [CLI Reference](artifact/references/cli-reference.md)\n- [Ops Playbooks](artifact/references/ops-playbooks.md)\n- [Investigation Protocol](artifact/references/investigation-protocol.md)\n- [Agent Guardrails](artifact/references/agent-guardrails.md)\n- [Project homepage](https://github.com/vmware-skills/VMware-Aria)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown with inline shell commands and structured operational summaries]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May include command suggestions, configuration guidance, investigation steps, and risk notes for write-capable Aria Operations actions.]\n\n## Skill Version(s):\n\n1.17.0 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.17.0:evals/evals.json\n\n{\n  \"skill_name\": \"vmware-aria\",\n  \"evals\": [\n    {\n      \"id\": 1,\n      \"prompt\": \"Which VMs are over-provisioned? Show me rightsizing recommendations and remaining capacity\",\n      \"expected_output\": \"Rightsizing list with CPU/memory recommendations\",\n      \"files\": [],\n      \"expectations\": [\n        \"Uses list_rightsizing_recommendations\",\n        \"Uses get_remaining_capacity for context\",\n        \"Presents actionable resize suggestions\"\n      ]\n    },\n    {\n      \"id\": 2,\n      \"prompt\": \"Generate a capacity report for the production cluster - how many days until we run out of resources?\",\n      \"expected_output\": \"Capacity forecast with time-remaining estimates\",\n      \"files\": [],\n      \"expectations\": [\n        \"Uses get_time_remaining for forecast\",\n        \"Uses get_capacity_overview for current state\",\n        \"Reports days remaining by resource type (CPU/memory/disk)\"\n      ]\n    },\n    {\n      \"id\": 3,\n      \"prompt\": \"Are there any anomalies in the last 24 hours? Create an alert definition for high CPU ready time\",\n      \"expected_output\": \"Anomaly list + new alert definition created\",\n      \"files\": [],\n      \"expectations\": [\n        \"Uses list_anomalies for recent anomalies\",\n        \"Uses create_alert_definition with CPU ready symptom\",\n        \"Confirms the new alert definition is active\"\n      ]\n    }\n  ]\n}\n\nArchive v1.16.0: 10 files, 51087 bytes\n\nFiles: evals/evals.json (1349b), references/agent-guardrails.md (3888b), references/capabilities.md (28727b), references/cli-reference.md (40401b), references/investigation-protocol.md (6850b), references/ops-playbooks.md (8762b), references/setup-guide.md (10238b), skill-card.md (2778b), SKILL.md (24468b), _meta.json (131b)\n\nFile v1.16.0:SKILL.md\n\n---\nname: vmware-aria\ndescription: >\n  Use this skill whenever the user needs VMware Aria Operations (VMware VCF Operations in VCF 9+) data — metrics, alerts, capacity, anomalies, reports.\n  Directly handles: resource metrics plus metric key/property/relationship lookup, list/acknowledge/cancel alerts with notes and recommendations, alert definitions, capacity forecasts, anomalies, reports, resource maintenance mode, Aria's own node health and adapter collection state.\n  Always use this skill for \"check vSphere capacity\", \"what Aria Operations alerts are active\", \"show VMware anomalies\", \"generate an Aria report\", \"rightsizing recommendations\", \"VCF Operations alerts\", \"put this host in Aria maintenance mode\", \"is Aria Operations still collecting from vCenter\", \"what does Aria recommend for this alert\", or any Aria Operations / VCF Operations / vRealize Operations task.\n  Do NOT use for real-time vCenter alarms/events (use vmware-monitor), VM operations (use vmware-aiops), or NSX networking (use vmware-nsx).\n  For load balancing/AVI/AKO use vmware-avi.\ninstaller:\n  kind: uv\n  package: vmware-aria\nallowed-tools:\n  - Bash\nmetadata: {\"openclaw\":{\"requires\":{\"anyBins\":[\"vmware-aria\",\"uvx\"]},\"optional\":{\"env\":[\"VMWARE_ARIA_CONFIG\",\"VMWARE_ARIA_<TARGET>_PASSWORD\",\"VMWARE_ARIA_<TARGET>_USERNAME\",\"VMWARE_AUDIT_APPROVED_BY\"],\"bins\":[\"vmware-policy\"]},\"homepage\":\"https://github.com/vmware-skills/VMware-Aria\",\"emoji\":\"📊\",\"os\":[\"macos\",\"linux\"]}}\ncompatibility: >\n  vmware-policy auto-installed as Python dependency (provides @vmware_tool decorator and audit logging). All write operations audited to ~/.vmware/audit.db.\n  Credentials: Each Aria Operations target requires a per-target password env var in ~/.vmware-aria/.env following the pattern VMWARE_ARIA_<TARGET_NAME_UPPER>_PASSWORD. Passwords are never logged or echoed.\n  Read-heavy: 34 of 44 tools are read-only. Write operations limited to alert acknowledge/cancel, alert notes, alert definition management, report management, and resource maintenance start/end.\n  No webhooks, no outbound network calls, no guest operations. Local only: stdio MCP + Aria Operations REST API (HTTPS 443).\n  Transitive dependencies: Only vmware-policy (audit/policy). No post-install scripts or background services.\n---\n\n# VMware Aria Operations\n\n> **Disclaimer**: This is a community-maintained open-source project and is **not affiliated with, endorsed by, or sponsored by VMware, Inc. or Broadcom Inc.** \"VMware\" and \"Aria\" are trademarks of Broadcom. Source code is publicly auditable at [github.com/vmware-skills/VMware-Aria](https://github.com/vmware-skills/VMware-Aria) under the MIT license.\n\nVMware Aria Operations (vRealize Operations 8.x, VCF Operations 9.x) AI-assisted monitoring — 44 MCP tools for resources (metric keys, properties, relationships), alerts (notes, recommendations), alert definitions, capacity planning, anomaly detection, report automation, resource maintenance mode, platform health (Aria node memory, adapter collection), and VCF 9.1 fleet certificates/passwords/domains, diagnostic findings, and real-time PromQL metrics.\n\n> **Know the version first**: `vmware-aria health status` (MCP `get_aria_health`) names the product line. Fleet, findings and PromQL exist only on VCF Operations 9.0+ (PromQL uses the 9.1 VODAP service); on 8.x they return a \"requires 9.0 or newer\" error, not data.\n> **Companion skills**: vmware-monitor (real-time vSphere), vmware-aiops (VM lifecycle), vmware-nsx (networking), vmware-avi (AVI/ALB/AKO), vmware-harden (compliance), vmware-pilot (approval workflows), vmware-policy (audit/policy).\n\n## What This Skill Does\n\n| Category | Tools | Count |\n|----------|-------|:-----:|\n| **Resources** | list, get details, metrics, health badge, top consumers, metric keys, properties, relationships | 8 |\n| **Alerts** | list, get details, investigate (alert→resource), acknowledge, cancel, list definitions, list/add notes, recommendations | 9 |\n| **Alert Definitions** | list symptoms, create definition, enable/disable, delete | 4 |\n| **Capacity** | cluster overview, remaining capacity, time remaining, rightsizing | 4 |\n| **Reports** | list templates, generate, list, get status+download URL, delete | 5 |\n| **Anomaly** | list anomalies, risk badge | 2 |\n| **Health** | Aria platform health, collector group status, Aria node memory/swap/heap, adapter collection state | 4 |\n| **Maintenance** | start / end resource maintenance, list maintenance schedules | 3 |\n| **Fleet / PromQL** (VCF Ops 9.1) | fleet certificates, password accounts, VCF domains, diagnostic findings, real-time PromQL query | 5 |\n\n**Total**: 44 tools (34 read-only + 10 write)\n\n## Quick Install\n\n```bash\nuv tool install vmware-aria==1.16.0\nvmware-aria init      # guided setup: writes config + .env (chmod 600, password grep-safe), then verifies\nvmware-aria doctor\n```\n\n## When to Use This Skill\n\n- **Lookup**: which metric keys a resource reports (name, unit) before querying them, its properties, its parents and children\n- **Performance**: VM contention (CPU Ready, balloon, swap), time-series metrics, top consumers, anomaly counts and risk badges\n- **Alerts**: list, investigate, acknowledge or cancel alerts; read or add notes (who is handling it); read the alert's prioritized recommendations; list, create, enable/disable or delete alert definitions (post-RCA)\n- **Capacity**: cluster headroom, time until full, VM rightsizing\n- **Reports**: generate, poll, download and delete reports\n- **Maintenance**: put a resource in maintenance before planned work (timed or until ended), end it, list maintenance schedules\n- **Platform**: is Aria Operations itself healthy (DEGRADED vs DOWN, which service), which version and product line, collector groups, whether the Aria node is short of memory, which adapter stopped collecting\n\nFor VM changes, NSX, vSphere alarms, storage or load balancing, route with the table below.\n\n## Related Skills — Skill Routing\n\n| User Intent | Recommended Skill |\n|-------------|-------------------|\n| Aria Operations monitoring, alerts, capacity | **vmware-aria** ← this skill |\n| VM lifecycle, deployment, guest ops | **vmware-aiops** |\n| NSX networking: segments, gateways, NAT, routing | **vmware-nsx** |\n| Read-only vSphere inventory, events, alarms | **vmware-monitor** |\n| Storage: iSCSI, vSAN, datastores | **vmware-storage** |\n| Multi-step workflows with approval | **vmware-pilot** |\n| Compliance baselines (CIS / 等保 / PCI-DSS), drift detection, LLM remediation advisor | **vmware-harden** (`uv tool install vmware-harden`) |\n| Load balancer, AVI, ALB, AKO, Ingress | **vmware-avi** (`uv tool install vmware-avi`) |\n| Audit log query | **vmware-policy** (`vmware-audit` CLI) |\n\n## Common Workflows\n\n> **Troubleshooting paths**: step-by-step playbooks for a DEGRADED platform, alert triage, VM contention, empty metrics, pre-resize checks and planned maintenance — [`references/ops-playbooks.md`](references/ops-playbooks.md).\n>\n> **Diagnostic investigations**: Before running any \"why is X slow / failing / down\" workflow, follow [`references/investigation-protocol.md`](references/investigation-protocol.md). It enforces the four root-cause completeness criteria (falsifiability / sufficiency / necessity / mechanism) and the up-to-three-rounds deepening loop. Stopping at a partial conclusion is an anti-pattern — always self-check against the criteria before outputting a report.\n\n### Daily VM Health Check (Proactive Ops)\n\n**Judgment**: don't chase the highest CPU consumer — chase the highest **contention** consumer. A VM at 90% CPU on a quiet host is healthy; a VM at 30% CPU but 15% Ready is starving. Key metrics: CPU Ready, Memory Balloon, Disk Latency.\n\n1. Find top CPU consumers → `vmware-aria resource top --metric 'cpu|usage_average' --top 20` (this is the **starting set**, not the answer)\n2. Check CPU Ready on hot VMs → `vmware-aria resource metrics <vm-id> --metrics 'cpu|readyPct' --hours 24`\n   - >5% = warning, >10% = problem, >20% = critical\n3. Check memory pressure → `vmware-aria resource metrics <vm-id> --metrics 'mem|balloonPct,mem|swapped_average' --hours 24`\n   - Balloon >0 = ESXi reclaiming memory; Swap >0 = severe — act immediately\n   - If a key comes back under `missing` instead of `metrics`, it is not a zero: `not_collected_for_resource` means a wrong key for this resource (use `similar_keys`), `no_data_in_window` means widen `--hours`\n4. List active CRITICAL/IMMEDIATE alerts → `vmware-aria alert list --criticality CRITICAL`\n5. Check anomaly counts → `vmware-aria anomaly list`\n6. Cross-validate against the [investigation protocol](references/investigation-protocol.md) before reporting any \"root cause\" — high consumption is rarely the root, usually a downstream symptom\n\n### Capacity Planning\n\n1. List clusters → `vmware-aria resource list --kind ClusterComputeResource`\n2. Get remaining capacity → `vmware-aria capacity remaining <cluster-id>`\n3. Predict time until full → `vmware-aria capacity time-remaining <cluster-id>`\n4. Get capacity overview → `vmware-aria capacity overview <cluster-id>`\n5. Find rightsizing candidates → `vmware-aria capacity rightsizing` — act only on rows with `Act. yes`; read each VM's caveats and the vendor minimum size before reducing\n   - If a yellow `properties_note` prints under the table, the VM property read failed: power state and current size are unknown and no row is actionable — retry, do not resize from it\n\n### Post-Incident: Create Detection Alert (RCA Follow-up)\n\nAfter resolving an incident, create an early-warning alert to prevent recurrence. Alert definition management is **MCP-only** (no CLI subcommands):\n\n1. Find matching symptom definitions → MCP `list_symptom_definitions` (filter by `name_filter` / `resource_kind`)\n2. Create alert definition referencing symptoms → MCP `create_alert_definition` with name, resource_kind, symptom_definition_ids, criticality (any one symptom firing triggers the alert)\n3. Verify it appears in definitions → `vmware-aria alert definitions --name \"Gold VM CPU\"` (criticality shown is the max severity across the definition's states)\n4. Enable or disable later → MCP `set_alert_definition_state`\n\n### Generate Capacity Report\n\n1. Find report template → `vmware-aria report definitions --name \"Capacity\"`\n2. Trigger report generation → `vmware-aria report generate <definition-id> --resources <resource-id>` (the Report API requires at least one resource UUID)\n3. Poll until completed → `vmware-aria report get <report-id>` (repeat until `status == COMPLETED`)\n4. Download via the returned `download_url` (PDF) or `csv_url`\n5. Clean up → `vmware-aria report delete <report-id>`\n\n## Usage Mode\n\n| Scenario | Recommended | Why |\n|----------|:-----------:|-----|\n| Local/small models (Ollama, Qwen) | **CLI** | ~2K tokens vs ~8K for MCP |\n| Cloud models (Claude, GPT-4o) | Either | MCP gives structured JSON I/O |\n| Automated pipelines | **MCP** | Type-safe parameters, structured output |\n\nRunning vmware-aria with a local or small model? See [`references/agent-guardrails.md`](references/agent-guardrails.md) for tool-calling guardrails (alert-to-resource correlation and Aria data fidelity).\n\nEvery command accepts `--target <name>` (every MCP tool `target`) to pick the Aria Operations instance.\n\n## MCP Tools (44 — 34 read, 10 write)\n\nAll MCP tools accept an optional `target`. The six destructive writes take `confirm`: without `confirm=true` they return a `blast_radius` and change nothing — show it to the user first.\n\n| Category | Tool | Type | Description |\n|----------|------|:----:|-------------|\n| Resource | `list_resources` | Read | List VMs, hosts, clusters by resource kind (or `all`); `collection_status` finds objects not receiving data |\n| | `get_resource` | Read | Get resource details with health, risk, efficiency badges |\n| | `get_resource_metrics` | Read | Time-series stats; `summary=true` gives n/min/max/avg/latest + change points; `missing` explains empty keys |\n| | `get_resource_health` | Read | Badge scores; service objects add `service.available` (a badge is not service state) |\n| | `get_top_consumers` | Read | Rank by last-hour average `value` (`latest_value` = newest point) |\n| | `list_metric_keys` | Read | Keys a resource reports with name/unit and `definition`, or a kind's defined keys — look up before querying |\n| | `get_resource_properties` | Read | Current property values (power state, parent host, extraConfig) |\n| | `get_resource_relationships` | Read | Related resources with `direction`; `relationship_type` ALL / PARENT / CHILD |\n| Alerts | `list_alerts` | Read | List active alerts with criticality, resource ID, name and kind (`resource_name: null` = unknown, see `resource_names_note`) |\n| | `get_alert` | Read | Alert details: named symptoms, the object each is on (e.g. the down service), ISO-8601 times |\n| | `investigate_alert` | Read | Resolve an alert to its confirmed affected resource in one call — returns both UUIDs explicitly labelled plus the vmware-mo\n\nArchive v1.15.0: 10 files, 50731 bytes\n\nFiles: evals/evals.json (1349b), references/agent-guardrails.md (3888b), references/capabilities.md (27655b), references/cli-reference.md (40401b), references/investigation-protocol.md (6850b), references/ops-playbooks.md (8762b), references/setup-guide.md (10238b), skill-card.md (2823b), SKILL.md (24386b), _meta.json (131b)\n\nArchive v1.14.1: 10 files, 48793 bytes\n\nFiles: evals/evals.json (1349b), references/agent-guardrails.md (3888b), references/capabilities.md (26176b), references/cli-reference.md (36663b), references/investigation-protocol.md (6850b), references/ops-playbooks.md (8762b), references/setup-guide.md (10238b), skill-card.md (2821b), SKILL.md (24315b), _meta.json (131b)\n\nArchive v1.14.0: 10 files, 48931 bytes\n\nFiles: evals/evals.json (1349b), references/agent-guardrails.md (3888b), references/capabilities.md (26176b), references/cli-reference.md (36663b), references/investigation-protocol.md (6850b), references/ops-playbooks.md (8762b), references/setup-guide.md (10238b), skill-card.md (3119b), SKILL.md (24315b), _meta.json (131b)\n\nArchive v1.13.0: 10 files, 47266 bytes\n\nFiles: evals/evals.json (1349b), references/agent-guardrails.md (3888b), references/capabilities.md (25417b), references/cli-reference.md (34806b), references/investigation-protocol.md (6850b), references/ops-playbooks.md (7733b), references/setup-guide.md (10238b), skill-card.md (2808b), SKILL.md (24123b), _meta.json (131b)\n\nArchive v1.12.0: 9 files, 35213 bytes\n\nFiles: evals/evals.json (1349b), references/agent-guardrails.md (3888b), references/capabilities.md (14311b), references/cli-reference.md (20231b), references/investigation-protocol.md (6850b), references/setup-guide.md (9672b), skill-card.md (2544b), SKILL.md (24532b), _meta.json (131b)\n\nArchive v1.11.0: 9 files, 29666 bytes\n\nFiles: evals/evals.json (1349b), references/agent-guardrails.md (3888b), references/capabilities.md (10899b), references/cli-reference.md (12044b), references/investigation-protocol.md (6716b), references/setup-guide.md (8729b), skill-card.md (2815b), SKILL.md (23667b), _meta.json (131b)\n\nArchive v1.10.0: 9 files, 28827 bytes\n\nFiles: evals/evals.json (1349b), references/agent-guardrails.md (3888b), references/capabilities.md (10899b), references/cli-reference.md (12044b), references/investigation-protocol.md (6716b), references/setup-guide.md (6822b), skill-card.md (2779b), SKILL.md (23514b), _meta.json (131b)\n\nArchive v1.9.1: 9 files, 28645 bytes\n\nFiles: evals/evals.json (1349b), references/agent-guardrails.md (3888b), references/capabilities.md (10899b), references/cli-reference.md (11802b), references/investigation-protocol.md (6716b), references/setup-guide.md (6822b), skill-card.md (2600b), SKILL.md (23514b), _meta.json (130b)","readmeExcerpt":"Skill: vmware-aria Owner: zw008 Summary: Use this skill whenever the user needs VMware Aria Operations (VMware VCF Operations in VCF 9+) data — metrics, alerts, capacity, anomalies, reports. Directly handles: resource metrics plus metric key/property/relationship lookup, list/acknowledge/cancel alerts with notes and recommendations, alert definitions, capacity forecasts, anomalies, reports, resource maintenance mode,","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"uv tool install vmware-aria==1.17.0\nvmware-aria init      # guided setup: writes config + .env (chmod 600, password grep-safe), then verifies\nvmware-aria doctor"},{"language":"bash","snippet":"# Resources\nvmware-aria resource list [--kind VirtualMachine|HostSystem|ClusterComputeResource|all] [--name <filter>] [--collection-status NO_DATA_RECEIVING]\nvmware-aria resource get <resource-id>\nvmware-aria resource metrics <resource-id> --metrics 'cpu|usage_average,mem|usage_average' --hours 4\nvmware-aria resource metrics <vm-id> --metrics 'cpu|readyPct,mem|balloonPct' --hours 24 --summary\nvmware-aria resource health <resource-id>\nvmware-aria resource top --metric 'cpu|usage_average' --kind VirtualMachine --top 10\nvmware-aria resource keys <resource-id> [--filter 'mem|']   # or --kind VirtualMachine\nvmware-aria resource properties <resource-id> [--name 'summary|']\nvmware-aria resource relationships <resource-id> [--type PARENT]\n\n# Alerts\nvmware-aria alert list [--criticality CRITICAL|IMMEDIATE|WARNING|INFORMATION] [--json]\nvmware-aria alert get <alert-id>\nvmware-aria alert acknowledge <alert-id>\nvmware-aria alert cancel <alert-id>\nvmware-aria alert definitions [--name <filter>]\nvmware-aria alert notes <alert-id>\nvmware-aria alert note-add <alert-id> \"Taking this: rebooting esx-03\"\nvmware-aria alert recommendations <alert-id>\n\n# Alert Definitions: creation/enable/disable/delete and symptom-definition\n# lookup are MCP-only tools (list_symptom_definitions, create_alert_definition,\n# set_alert_definition_state, delete_alert_definition) — no CLI subcommands.\n\n# Capacity\nvmware-aria capacity overview <cluster-id>\nvmware-aria capacity remaining <resource-id>\nvmware-aria capacity time-remaining <resource-id>\nvmware-aria capacity rightsizing [--resource-id <vm-id>]\n\n# Reports (async: generate → poll get → download → delete)\nvmware-aria report definitions [--name <filter>]\nvmware-aria report generate <definition-id> --resources <id1,id2>   # at least one resource UUID required\nvmware-aria report list [--definition-id <id>]\nvmware-aria report get <report-id>        # poll until status == COMPLETED; shows download_url\nvmware-aria report delete <report-id>\n\n# Anomaly\nvmware-a"},{"language":"text","snippet":"- Use investigate_alert to go from an alert to its affected resource.\n- Do not pass an alert UUID where a resource UUID is expected. They are\n  different objects; investigate_alert labels both.\n- Only query vCenter for a resource once correlation.confirmed is true.\n  The next_step block names the exact tool and argument to use."},{"language":"text","snippet":"- Preserve the exact criticality (INFORMATION / WARNING / IMMEDIATE /\n  CRITICAL), status, impact, and control-state values the tools return.\n  Do not translate, normalise, or prettify them.\n- alertLevel is the criticality field. An alert definition has no top-level\n  criticality — it is the maximum severity across its states.\n- Recommendations hang off the alert definition, not the alert.\n- Do not claim a capacity, performance, or risk problem unless the tool output\n  contains explicit supporting evidence. Badge colour alone is not a diagnosis."},{"language":"text","snippet":"vmware-aria doctor [OPTIONS]\n\nOptions:\n  --skip-auth    Skip authentication check (only tests config + network)\n  --config -c    Path to config file"},{"language":"text","snippet":"vmware-aria resource list [OPTIONS]\n\nOptions:\n  --kind -k TEXT             Resource kind [default: VirtualMachine]\n                             Values: VirtualMachine, HostSystem, ClusterComputeResource,\n                                     Datastore, Datacenter, ResourcePool, or all\n  --limit -n INT             Max results [default: 50]\n  --name TEXT                Filter by name substring (case-insensitive)\n  --collection-status TEXT   Keep objects with this data-collection status,\n                             e.g. NO_DATA_RECEIVING (case-insensitive)\n  --target -t TEXT           Target name"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: vmware-aria\ndescription: >\n  Use this skill whenever the user needs VMware Aria Operations (VMware VCF Operations in VCF 9+) data — metrics, alerts, capacity, anomalies, reports.\n  Directly handles: resource metrics plus metric key/property/relationship lookup, list/acknowledge/cancel alerts with notes and recommendations, alert definitions, capacity forecasts, anomalies, reports, resource maintenance mode, Aria's own node health and adapter collection state.\n  Always use this skill for \"check vSphere capacity\", \"what Aria Operations alerts are active\", \"show VMware anomalies\", \"generate an Aria report\", \"rightsizing recommendations\", \"VCF Operations alerts\", \"put this host in Aria maintenance mode\", \"is Aria Operations still collecting from vCenter\", \"what does Aria recommend for this alert\", or any Aria Operations / VCF Operations / vRealize Operations task.\n  Do NOT use for real-time vCenter alarms/events (use vmware-monitor), VM operations (use vmware-aiops), or NSX networking (use vmware-nsx).\n  For load balancing/AVI/AKO use vmware-avi.\ninstaller:\n  kind: uv\n  package: vmware-aria\nallowed-tools:\n  - Bash\nmetadata: {\"openclaw\":{\"requires\":{\"anyBins\":[\"vmware-aria\",\"uvx\"]},\"optional\":{\"env\":[\"VMWARE_ARIA_CONFIG\",\"VMWARE_ARIA_<TARGET>_PASSWORD\",\"VMWARE_ARIA_<TARGET>_USERNAME\",\"VMWARE_AUDIT_APPROVED_BY\"],\"bins\":[\"vmware-policy\"]},\"homepage\":\"https://github.com/vmware-skills/VMware-Aria\",\"emoji\":\"📊\",\"os\":[\"macos\",\"linux\"]}}\ncompatibility: >\n  vmware-policy auto-installed as Python dependency (provides @vmware_tool decorator and audit logging). All write operations audited to ~/.vmware/audit.db.\n  Credentials: Each Aria Operations target requires a per-target password env var in ~/.vmware-aria/.env following the pattern VMWARE_ARIA_<TARGET_NAME_UPPER>_PASSWORD. Passwords are never logged or echoed.\n  Read-heavy: 34 of 44 tools are read-only. Write operations limited to alert acknowledge/cancel, alert notes, alert definition management, report management, and resource maintenance start/end.\n  No webhooks, no outbound network calls, no guest operations. Local only: stdio MCP + Aria Operations REST API (HTTPS 443).\n  Transitive dependencies: Only vmware-policy (audit/policy). No post-install scripts or background services.\n---\n\n# VMware Aria Operations\n\n> **Disclaimer**: This is a community-maintained open-source project and is **not affiliated with, endorsed by, or sponsored by VMware, Inc. or Broadcom Inc.** \"VMware\" and \"Aria\" are trademarks of Broadcom. Source code is publicly auditable at [github.com/vmware-skills/VMware-Aria](https://github.com/vmware-skills/VMware-Aria) under the MIT license.\n\nVMware Aria Operations (vRealize Operations 8.x, VCF Operations 9.x) AI-assisted monitoring — 44 MCP tools for resources (metric keys, properties, relationships), alerts (notes, recommendations), alert definitions, capacity planning, anomaly detection, report automation, resource maintenance mode, platform health (Aria node memory, adapter co"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7b067awq2s97bn3d7p5qfhw5827pxc\",\n  \"slug\": \"vmware-aria\",\n  \"version\": \"1.17.0\",\n  \"publishedAt\": 1789915908066\n}"},{"path":"references/agent-guardrails.md","content":"# Operating vmware-aria with a local / small model\n\nClaude-class models drive this skill without special instruction. Smaller and\nlocally-hosted models — Llama 3.3 70B, Qwen, Mistral, and similar, served\nthrough Goose, Ollama, or OpenShift AI — need explicit operating rules to call\ntools reliably.\n\nThis page covers what goes wrong most often with vmware-aria specifically:\n**alert-to-resource correlation**. For the full cross-skill guardrail set, the\ncomplete system prompt, and the small-model failure-mode checklist, see the\ncanonical guide in\n[vmware-monitor's references](https://github.com/vmware-skills/VMware-Monitor/blob/main/skills/vmware-monitor/references/agent-guardrails.md).\n\nThese guardrails are adapted, with thanks, from the working configuration\n[@juanpf-ha](https://github.com/juanpf-ha) developed while running vmware-aria\nand vmware-monitor against a production vSphere estate\n([VMware-AIops#31](https://github.com/vmware-skills/VMware-AIops/issues/31)).\n\n> **Disclaimer**: This is a community-maintained open-source project and is\n> **not affiliated with, endorsed by, or sponsored by VMware, Inc. or Broadcom\n> Inc.** \"VMware\" and \"vSphere\" are trademarks of Broadcom.\n\n---\n\n## 1. Alert-to-resource correlation\n\nThis is the sequence small models get wrong most reliably. An Aria alert does\nnot carry the affected object's name — only a `resourceId`. Correlating an\nalert with vCenter therefore means: fetch the alert, read its `resourceId`,\nfetch that resource, confirm its name and kind, then query vmware-monitor. Two\nUUIDs are in play, and they get swapped.\n\n**Use `investigate_alert` instead of chaining the steps by hand.** It performs\nthe whole sequence server-side and returns:\n\n- `alert` — Aria's own criticality / status / impact / control-state values,\n  passed through verbatim.\n- `resource` — the affected object, or `null`.\n- `correlation` — both UUIDs **explicitly labelled** (`alert_id` vs\n  `resource_id`), plus `resource_name`, `resource_kind`, and a `confirmed`\n  flag. Every key is always present; unresolved values are explicit `null`.\n- `next_step` — the exact vmware-monitor tool and argument to call next, or\n  `null` when there is nothing confirmed to hand off.\n- `warnings` — why anything above is incomplete; empty on success.\n\nA resource that cannot be resolved degrades to a warning plus explicit nulls\nrather than an error, so the alert you already fetched is never lost.\n\nIf you must drive the steps manually, state these rules explicitly:\n\n```text\n- Use investigate_alert to go from an alert to its affected resource.\n- Do not pass an alert UUID where a resource UUID is expected. They are\n  different objects; investigate_alert labels both.\n- Only query vCenter for a resource once correlation.confirmed is true.\n  The next_step block names the exact tool and argument to use.\n```\n\n---\n\n## 2. Aria-specific data fidelity\n\nAria's enum values carry operational meaning and must survive the model\nuntouched:\n\n```text\n- Preserve the exact critica"},{"path":"references/capabilities.md","content":"# Capabilities\n\n## Automation Level Reference\n\nEach operation is classified by autonomy level per the Enterprise Harness Engineering framework. **vmware-aria is heavily L1/L2 (34 read / 10 write)** — primarily a monitoring and analysis skill.\n\n| Level | Meaning | Agent autonomy | Examples in this skill |\n|:-:|---|---|---|\n| **L1** | Read-only, raw data | Always auto-run | `list_resources`, `get_resource`, `get_resource_metrics`, `list_metric_keys`, `get_resource_properties`, `get_resource_relationships`, `list_alerts`, `get_alert`, `list_alert_definitions`, `list_alert_notes`, `list_maintenance_schedules`, `get_aria_node_resources`, `list_adapters`, capacity / badge queries |\n| **L2** | Read + analysis / recommendation | Always auto-run | `investigate_alert` (alert → confirmed affected resource), `get_alert_recommendations` (alert → definition state → prioritized recommendations), anomaly counts, top-N consumer ranking, capacity trend forecasting, rightsizing recommendations |\n| **L3** | Single write — user must approve | Only after explicit confirmation | `acknowledge_alert` (via takeownership action), `cancel_alert`, `create_alert_definition`, `set_alert_definition_state`, `delete_alert_definition`, `generate_report`, `delete_report`, `start_resource_maintenance`, `end_resource_maintenance`, `add_alert_note` *(the only writes; all auditable. On MCP, `acknowledge_alert`, `cancel_alert`, `delete_alert_definition`, `delete_report`, `start_resource_maintenance` and `end_resource_maintenance` take `confirm` (default false): without `confirm=true` they return a `blast_radius` — what the call would change — and change nothing; `confirm=true` is refused when a blocker is found or a field the radius depends on could not be read. `confirmed` is a deprecated alias. `add_alert_note` is low risk and has no gate — a note changes neither the alert nor monitoring; the CLI still asks once)* |\n| **L4** | Multi-step plan / apply workflow | *N/A currently* | — *(no multi-step orchestration; Aria is observe/analyze, not configure)* |\n| **L5** | Auto-remediation from learned pattern | Pattern library only; requires `risk:low` + `reversible:true` + `repeatable:true` | *(roadmap — candidates: auto-acknowledge known-noisy alerts, auto-cancel resolved-by-event alerts)* |\n\n**Notes**:\n- L1/L2 tools are always safe for agents to call without confirmation.\n- L3 alert-state writes pass through the `@vmware_tool` decorator: connection check → policy check → audit log. Cancel is irreversible by Aria API design and treated as a destructive operation.\n- For VM/host operations see [vmware-aiops](https://github.com/vmware-skills/VMware-AIops); Aria recommendations are advisory, not actuating.\n\n## What vmware-aria Can Do\n\n### Resource Monitoring\n\n- **List any resource type**: VirtualMachine, HostSystem, ClusterComputeResource, Datastore, Datacenter, ResourcePool\n- **Filter by name**: substring match across any resource list\n- **Find objects that stopped reporting**: every row carri"},{"path":"references/cli-reference.md","content":"# CLI Reference\n\nComplete reference for `vmware-aria` command-line interface.\n\n## Global Options\n\nAll commands accept:\n- `--target / -t <name>` — Target name from config (uses default if omitted)\n- `--config / -c <path>` — Custom config file path\n\n---\n\n## `vmware-aria doctor`\n\nRun pre-flight diagnostics.\n\n```\nvmware-aria doctor [OPTIONS]\n\nOptions:\n  --skip-auth    Skip authentication check (only tests config + network)\n  --config -c    Path to config file\n```\n\n**Checks performed**:\n1. Config file exists at `~/.vmware-aria/config.yaml`\n2. `.env` file permissions (warns if wider than 600)\n3. Config parse succeeds (validates YAML and target structure)\n4. Password env vars are set for each target\n5. Network TCP connectivity to port 443 for each target\n6. Aria Operations token acquisition (unless `--skip-auth`)\n7. Aria version from `GET /versions/current`, e.g. `VMware Aria Operations 8.18.7 (8.x line, build 25423534)` (**WARN** `Not read: …` when it cannot be read — an unreadable version says nothing about whether the target works), and Aria platform health: PASS when HEALTHY, **WARN** when DEGRADED or UNKNOWN (the failed services are named), FAIL when DOWN. Only a failure to connect is an auth FAIL: an error after the token was acquired is a FAIL on the \"Aria platform\" row (`Checks did not complete: …`). The doctor disconnects from each target either way\n8. MCP server module importable\n\nError details in the doctor table do not tell you to run the doctor again.\n\n---\n\n## Resource Commands\n\n### `vmware-aria resource list`\n\nList resources by kind.\n\n```\nvmware-aria resource list [OPTIONS]\n\nOptions:\n  --kind -k TEXT             Resource kind [default: VirtualMachine]\n                             Values: VirtualMachine, HostSystem, ClusterComputeResource,\n                                     Datastore, Datacenter, ResourcePool, or all\n  --limit -n INT             Max results [default: 50]\n  --name TEXT                Filter by name substring (case-insensitive)\n  --collection-status TEXT   Keep objects with this data-collection status,\n                             e.g. NO_DATA_RECEIVING (case-insensitive)\n  --target -t TEXT           Target name\n```\n\n**Output**: Table with Name (and Kind with `--kind all`), ID, Health (color + score), Aria state, and Collection.\n**Aria state** is Aria's lifecycle state for the object — `STARTED` for a powered-off VM too, so it is not a power state.\n**Collection** is whether data is arriving (`DATA_RECEIVING`, `NO_DATA_RECEIVING`, …; `—` when Aria reports none).\nTo find the objects behind \"Objects are not receiving data from adapter instance\", run\n`vmware-aria resource list --kind all --collection-status NO_DATA_RECEIVING`.\nIf the filter matches nothing, a yellow line names the statuses the listing did contain.\n\n### `vmware-aria resource get`\n\nGet full resource details.\n\n```\nvmware-aria resource get <resource-id> [OPTIONS]\n\nArguments:\n  resource-id  Resource UUID (required)\n\nOptions:\n  --target -t TEXT  Target name\n```\n\n**Ou"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Use this skill whenever the user needs VMware Aria Operations (VMware VCF Operations in VCF 9+) data — metrics, alerts, capacity, anomalies, reports. Directly handles: resource metrics plus metric key/property/relationship lookup, list/acknowledge/cancel alerts with notes and recommendations, alert definitions, capacity forecasts, anomalies, reports, resource maintenance mode, Aria's own node health and adapter collection state. Always use this skill for \"check vSphere capacity\", \"what Aria Operations alerts are active\", \"show VMware anomalies\", \"generate an Aria report\", \"rightsizing recommendations\", \"VCF Operations alerts\", \"put this host in Aria maintenance mode\", \"is Aria Operations still collecting from vCenter\", \"what does Aria recommend for this alert\", or any Aria Operations / VCF Operations / vRealize Operations task. Do NOT use for real-time vCenter alarms/events (use vmware-monitor), VM operations (use vmware-aiops), or NSX networking (use vmware-nsx). For load balancing/AVI/AKO use vmware-avi. Skill: vmware-aria Owner: zw008 Summary: Use this skill whenever the user needs VMware Aria Operations (VMware VCF Operations in VCF 9+) data — metrics, alerts, capacity, anomalies, reports. Directly handles: resource metrics plus metric key/property/relationship lookup, list/acknowledge/cancel alerts with notes and recommendations, alert definitions, capacity forecasts, anomalies, reports, resource maintenance mode,","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1581,"uniquenessScore":47,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T04:21:42.936Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T04:21:42.936Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T16:27:51.275Z","emptyReason":null},"items":[{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-10T18:48:31.762Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}