{"id":"4a54d690-533f-4c34-9243-a7cce4c4a1ce","entityType":"agent","slug":"clawhub-zw008-vmware-privateai","name":"vmware-privateai","canonicalUrl":"https://www.xpersona.co/agent/clawhub-zw008-vmware-privateai","canonicalPath":"/agent/clawhub-zw008-vmware-privateai","generatedAt":"2026-10-11T14:15:03.710Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T11:46:55.190Z","emptyReason":null},"description":"Use this skill whenever the user needs the GPU / AI-infrastructure layer of VMware Private AI Foundation with NVIDIA (PAIF-N) on vSphere 9.x / VCF 9.1: inventory GPU hosts and physical GPU devices, see which VMs consume a vGPU and the profile each holds, read real-time GPU utilization, list the vGPU and DirectPath profile catalog, assign a VM's vGPU profile, and list Private AI Service (PAIS) served models and knowledge bases. Always use this skill for \"list GPU hosts\", \"which VMs are using a vGPU\", \"GPU utilization\", \"assign a vGPU profile\", \"list vGPU profiles\", \"list served models\" when the context is explicitly VMware / vSphere / VCF Private AI / NVIDIA vGPU. Do NOT use for the backing VM's power/snapshot/clone/migrate (use vmware-aiops), read-only vSphere inventory/alarms/host health (use vmware-monitor), or GPU-enabled Tanzu Kubernetes (use vmware-vks). This skill is the GPU lens; vmware-aiops owns the VM lifecycle behind it.","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.1K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s171xgnmqse0nqvgqvqnaq5f9183kyre:vmware-privateai","sourceUrl":"https://clawhub.ai/zw008/vmware-privateai","homepage":"https://clawhub.ai/zw008/skills/vmware-privateai","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/zw008/vmware-privateai","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/zw008/skills/vmware-privateai","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":61,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"vmware-privateai technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T11:46:55.190Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T11:46:55.190Z","emptyReason":null},"stars":null,"forks":null,"downloads":1072,"packageName":null,"latestVersion":"1.4.0","tractionLabel":"1.1K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T11:46:55.122Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T11:46:55.190Z","lastCrawledAt":"2026-10-11T11:46:55.122Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T11:46:55.122Z","lastVerifiedAt":null,"highlights":[{"version":"1.4.0","createdAt":"2026-09-20T14:53:06.383Z","changelog":"MCP instructions now name the configured targets and how to choose one; a config that cannot be read says so instead of falling silent.","fileCount":6,"zipByteSize":18412},{"version":"1.3.0","createdAt":"2026-09-19T03:55:49.524Z","changelog":"Destructive MCP tools preview by default (confirm=False) and state their blast radius; confirm=True refuses on blockers or unreadable measurements. Requires vmware-policy>=1.17.0.","fileCount":6,"zipByteSize":18446},{"version":"1.2.3","createdAt":"2026-09-16T05:19:06.101Z","changelog":"A stopped MCP server exits within five seconds even if its logout hangs","fileCount":6,"zipByteSize":17943},{"version":"1.2.2","createdAt":"2026-09-15T14:38:52.047Z","changelog":"Stopping the MCP server now logs out its vCenter session.","fileCount":6,"zipByteSize":17967},{"version":"1.2.1","createdAt":"2026-09-15T06:06:43.621Z","changelog":"CLI reads are audited under their MCP tool names; every CLI command declares what it reaches (needs vmware-policy 1.15.0)","fileCount":6,"zipByteSize":17922},{"version":"1.2.0","createdAt":"2026-09-12T00:16:45.127Z","changelog":"The policy environment and the connection now read the same config file, so a production target can no longer be judged by one file's label and connected to from another. CLI writes are authorised and audited under their MCP tool names.","fileCount":6,"zipByteSize":17925},{"version":"1.1.1","createdAt":"2026-09-05T01:06:46.871Z","changelog":"a dropped connection no longer keeps itself alive","fileCount":6,"zipByteSize":17918},{"version":"1.1.0","createdAt":"2026-08-31T07:25:44.290Z","changelog":"a doctor, so a failure has somewhere to look","fileCount":6,"zipByteSize":17959}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s171xgnmqse0nqvgqvqnaq5f9183kyre:vmware-privateai","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s171xgnmqse0nqvgqvqnaq5f9183kyre:vmware-privateai` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/zw008/vmware-privateai before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zw008-vmware-privateai/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zw008-vmware-privateai/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zw008-vmware-privateai/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-zw008-vmware-privateai/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-zw008-vmware-privateai/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-zw008-vmware-privateai/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T14:15:03.706Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zw008-vmware-privateai/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zw008-vmware-privateai/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zw008-vmware-privateai/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zw008-vmware-privateai/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T11:46:55.190Z","emptyReason":null},"readme":"Skill: vmware-privateai\n\nOwner: zw008\n\nSummary: Use this skill whenever the user needs the GPU / AI-infrastructure layer of VMware Private AI Foundation with NVIDIA (PAIF-N) on vSphere 9.x / VCF 9.1: inventory GPU hosts and physical GPU devices, see which VMs consume a vGPU and the profile each holds, read real-time GPU utilization, list the vGPU and DirectPath profile catalog, assign a VM's vGPU profile, and list Private AI Service (PAIS) served models and knowledge bases. Always use this skill for \"list GPU hosts\", \"which VMs are using a vGPU\", \"GPU utilization\", \"assign a vGPU profile\", \"list vGPU profiles\", \"list served models\" when the context is explicitly VMware / vSphere / VCF Private AI / NVIDIA vGPU. Do NOT use for the backing VM's power/snapshot/clone/migrate (use vmware-aiops), read-only vSphere inventory/alarms/host health (use vmware-monitor), or GPU-enabled Tanzu Kubernetes (use vmware-vks). This skill is the GPU lens; vmware-aiops owns the VM lifecycle behind it.\n\nTags: latest:1.4.0\n\nVersion history:\n\nv1.4.0 | 2026-09-20T14:53:06.383Z | user\n\nMCP instructions now name the configured targets and how to choose one; a config that cannot be read says so instead of falling silent.\n\nv1.3.0 | 2026-09-19T03:55:49.524Z | user\n\nDestructive MCP tools preview by default (confirm=False) and state their blast radius; confirm=True refuses on blockers or unreadable measurements. Requires vmware-policy>=1.17.0.\n\nv1.2.3 | 2026-09-16T05:19:06.101Z | user\n\nA stopped MCP server exits within five seconds even if its logout hangs\n\nv1.2.2 | 2026-09-15T14:38:52.047Z | user\n\nStopping the MCP server now logs out its vCenter session.\n\nv1.2.1 | 2026-09-15T06:06:43.621Z | user\n\nCLI reads are audited under their MCP tool names; every CLI command declares what it reaches (needs vmware-policy 1.15.0)\n\nv1.2.0 | 2026-09-12T00:16:45.127Z | user\n\nThe policy environment and the connection now read the same config file, so a production target can no longer be judged by one file's label and connected to from another. CLI writes are authorised and audited under their MCP tool names.\n\nv1.1.1 | 2026-09-05T01:06:46.871Z | user\n\na dropped connection no longer keeps itself alive\n\nv1.1.0 | 2026-08-31T07:25:44.290Z | user\n\na doctor, so a failure has somewhere to look\n\nv1.0.4 | 2026-08-31T00:36:35.285Z | user\n\nfix: run the suite on a non-UTF-8 machine, and stop one skill answering for another\n\nv1.0.3 | 2026-08-30T15:20:26.365Z | user\n\nSecond-round fixes from the 2026-08-30 VCF 9.1 re-test; vmware-policy floor raised to 1.11.0 (the engine no longer fails open when rules.yaml cannot be read).\n\nv1.0.2 | 2026-08-30T09:35:39.423Z | user\n\nParameter descriptions now reach the MCP JSON schema (0% -> 100% coverage); additionalProperties closed; vmware-policy floor raised to 1.10.0.\n\nv1.0.1 | 2026-08-28T05:12:35.404Z | user\n\nDistribution fix: ClawHub resolves latest by version order, so the withdrawn 1.0.0 outranked every 0.x release. 1.0.1 reclaims latest. No functional change; still beta in substance.\n\nv0.2.1 | 2026-08-28T02:56:40.123Z | user\n\nFixes the server's self-reported version and the advertised tool count; adds a Claude Code plugin manifest.\n\nv0.2.0 | 2026-08-10T04:20:49.982Z | user\n\n7 new read/pre-flight PAIS/GPU tools: sizing advisor, vGPU profile validate, host readiness, PAIS monitoring rollup, local air-gap bundle inspector, PAIS model-catalog + data-source reads (10→17 tools)\n\nv0.1.0 | 2026-08-06T09:57:29.270Z | user\n\nFirst beta (0.1.0): 10 tools for VMware Private AI Foundation on vSphere 9.x/VCF 9.1\n\nv1.0.0 | 2026-08-06T09:41:31.773Z | user\n\nFirst beta: 10 tools (GPU inventory/vGPU profiles/utilization/vGPU assign/PAIS model serving) for VMware Private AI Foundation on vSphere 9.x/VCF 9.1\n\nArchive index:\n\nArchive v1.4.0: 6 files, 18412 bytes\n\nFiles: references/capabilities.md (9124b), references/cli-reference.md (7276b), references/setup-guide.md (5951b), skill-card.md (2460b), SKILL.md (15504b), _meta.json (135b)\n\nFile v1.4.0:SKILL.md\n\n---\nname: vmware-privateai\ndescription: >\n  Use this skill whenever the user needs the GPU / AI-infrastructure layer of VMware Private AI\n  Foundation with NVIDIA (PAIF-N) on vSphere 9.x / VCF 9.1: inventory GPU hosts and physical GPU\n  devices, see which VMs consume a vGPU and the profile each holds, read real-time GPU utilization,\n  list the vGPU and DirectPath profile catalog, assign a VM's vGPU profile, and list Private AI\n  Service (PAIS) served models and knowledge bases. Always use this skill for \"list GPU hosts\",\n  \"which VMs are using a vGPU\", \"GPU utilization\", \"assign a vGPU profile\", \"list vGPU profiles\",\n  \"list served models\" when the context is explicitly VMware / vSphere / VCF Private AI / NVIDIA\n  vGPU. Do NOT use for the backing VM's power/snapshot/clone/migrate (use vmware-aiops), read-only\n  vSphere inventory/alarms/host health (use vmware-monitor), or GPU-enabled Tanzu Kubernetes\n  (use vmware-vks). This skill is the GPU lens; vmware-aiops owns the VM lifecycle behind it.\ninstaller:\n  kind: uv\n  package: vmware-privateai\nallowed-tools:\n  - Bash\nmetadata: {\"openclaw\":{\"requires\":{\"anyBins\":[\"vmware-privateai\",\"uvx\"]},\"optional\":{\"env\":[\"VMWARE_PRIVATEAI_CONFIG\"]}}}\n---\n\n# VMware Private AI (Foundation with NVIDIA) — GPU & Model-Serving Ops\n\n> **Disclaimer**: Community-maintained open-source project, **not affiliated with, endorsed by, or\n> sponsored by VMware, Inc., Broadcom Inc., or NVIDIA Corporation.** \"VMware\", \"vSphere\", and \"VCF\"\n> are trademarks of Broadcom; \"NVIDIA\" and \"vGPU\" are trademarks of NVIDIA. Source is publicly\n> auditable under the MIT license.\n\nThe GPU / AI-infrastructure lens for the VMware skill family — GPU host & device inventory, vGPU\nconsumers, real-time GPU utilization, the vGPU / DirectPath profile catalog, vGPU assignment, and\n**Private AI Service (PAIS)** served models and knowledge bases — over the **vSphere 9.x / VCF 9.1**\nWeb Services API (pyVmomi) plus the PAIS REST API.\n\n> **Companion skills**: [vmware-aiops](https://github.com/vmware-skills/VMware-AIops) (the vCenter VMs\n> behind AI workloads — power/snapshot/clone), [vmware-vks](https://github.com/vmware-skills/VMware-VKS)\n> (GPU-enabled Tanzu Kubernetes), [vmware-monitor](https://github.com/vmware-skills/VMware-Monitor)\n> (read-only vSphere health).\n\n> **Status: v1.0.1 — still beta in substance.** Skill #15 of the family. The jump from 0.2.x to\n> 1.0.1 is a distribution fix, not a maturity claim: the withdrawn first release used 1.0.0, and\n> ClawHub resolves `latest` by version order, so every 0.x release was invisible there. The beta\n> caveats below all still stand. Every API path is\n> verified against official Broadcom/NVIDIA sources before use (`tests/eval/spec/privateai_endpoints.py`)\n> — no endpoints written from memory. GET-response *field names* and the exact PAIS paths are\n> defensive and pending validation against live 9.x hardware (see Troubleshooting). Governed by the\n> family harness (audit + policy + teaching errors); read-vs-write authorization is delegated to the\n> vCenter service account's RBAC role.\n\n## What This Skill Does\n\n| Category | Tools | Count | Read/Write |\n|----------|-------|:-----:|:----------:|\n| **GPU inventory** | host list/get, device list, vGPU consumer list | 4 | 4 R |\n| **GPU utilization** | real-time per-vGPU-VM utilization (gpu %, mem %, temp) | 1 | 1 R |\n| **GPU readiness** | per-host vGPU/PAIS readiness verdict + blocking reasons | 1 | 1 R |\n| **Profile catalog** | vGPU profile list, DirectPath profile list | 2 | 2 R |\n| **Profile validation** | pre-flight a vGPU profile change (power state + host offers it) | 1 | 1 R |\n| **vGPU assignment** | set a VM's vGPU profile (VM must be powered off) | 1 | 1 W |\n| **Private AI Service** | served-model list, model catalog, knowledge-base list, data-source list | 4 | 4 R |\n| **PAIS monitoring** | fleet GPU rollup (util/mem/temp, hot/idle, busiest) | 1 | 1 R |\n| **Sizing & air-gap** | LLM GPU/storage sizing advisor, local pais.yml image inspector | 2 | 2 R |\n\n**17 MCP tools (16 read / 1 write).** Reads are strictly non-destructive. The single write\n(`vgpu_assign`) previews its blast radius, refuses a powered-on VM, never powers a VM off itself, is\ndouble-confirmed at the CLI, and is audit-logged. Pre-flight the write with `vgpu_profile_validate`.\n\n## Quick Install\n\n```bash\nuv tool install vmware-privateai==1.4.0\nvmware-privateai version\nvmware-privateai gpu host-list        # first read — lists hosts that have a GPU\n```\n\nConfig lives in `~/.vmware-privateai/config.yaml` (targets + optional `pais:` section); passwords and\nthe PAIS bearer token live in `~/.vmware-privateai/.env` (chmod 600). See `references/setup-guide.md`.\n\n## When to Use This Skill\n\nUse vmware-privateai for the **GPU / AI-infrastructure layer**: which hosts and physical devices have\nGPUs, which VMs hold a vGPU and what profile, real-time GPU utilization, the assignable vGPU /\nDirectPath profile catalog, changing a VM's vGPU profile, and the models / knowledge bases served by\nPrivate AI Service — when the context is explicitly VMware / vSphere / VCF Private AI / NVIDIA vGPU.\n\n**Do NOT use when**: the task is the backing VM's lifecycle — power on/off, snapshot, clone, migrate,\nreconfigure CPU/RAM (→ **vmware-aiops**); read-only vSphere inventory, alarms, or host health\n(→ **vmware-monitor**); or GPU-enabled Tanzu Kubernetes / Supervisor namespaces (→ **vmware-vks**).\n`vgpu_assign` deliberately does **not** power the VM off — that is vmware-aiops's job, kept separate\nso this skill's blast radius stays \"one VM, when it is already off\".\n\n## Related Skills — Skill Routing\n\n| The user wants… | Skill |\n|-----------------|-------|\n| Inventory GPUs / vGPU consumers / GPU utilization / assign a vGPU profile | **vmware-privateai** (this) |\n| List PAIS served models / knowledge bases | **vmware-privateai** (this) |\n| Power off / snapshot / clone / migrate the backing vCenter VM | vmware-aiops |\n| Read-only vSphere inventory / alarms / host health | vmware-monitor |\n| GPU-enabled Tanzu Kubernetes clusters / namespaces | vmware-vks |\n| Multi-step GPU workflow with approval + rollback | vmware-pilot |\n\n## Common Workflows\n\n**1. Find an idle GPU and reassign a VM's vGPU profile.**\n```\nvmware-privateai gpu device-list --vendor NVIDIA     # find GPUs; vm_count 0 = idle\nvmware-privateai gpu consumer-list                   # who holds a vGPU, and which profile\nvmware-privateai vgpu profile-list --host esx-07     # profiles that host can hand a VM\nvmware-privateai gpu vgpu-assign fin-train-01 grid_a100-4c --dry-run   # preview blast radius\n# power the VM off with vmware-aiops, THEN:\nvmware-privateai gpu vgpu-assign fin-train-01 grid_a100-4c             # double-confirm + audit\n```\n*Failure branch*: if `vgpu-assign` (confirm) refuses with \"VM is powered on — a vGPU change needs the\nVM powered off\", run `vmware-aiops vm_power_off 'fin-train-01'` first, then re-run. If it fails with\n\"profile not offered by the VM's host / GPU lacks free framebuffer\", run\n`vmware-privateai gpu host-get <that VM's host>` to see the valid profiles and free capacity.\n\n**2. Triage GPU utilization across the estate.**\n```\nvmware-privateai gpu utilization --top 10            # busiest vGPU VMs first\nvmware-privateai gpu host-list --vendor NVIDIA       # which hosts carry the load\n```\n*Failure branch*: a VM showing `metrics unavailable (no host driver?)` is not an error — the NVIDIA\nhost GPU driver is not exposing counters for it (`metrics_available:false`). Deep per-SM / per-process\n/ MIG-slice telemetry is **not** available via vSphere; use NVIDIA DCGM on the host for that.\n\n**3. See what Private AI Service is serving.**\n```\nvmware-privateai pais model-list                     # OpenAI-compatible /models\nvmware-privateai pais kb-list                         # RAG knowledge bases\n```\n*Failure branch*: HTTP 404 usually means a base-URL mismatch, not a bug — the `/api/v1` PAIS path\nprefix is deployment-specific and unconfirmed (beta). Check `pais.endpoint` in config.yaml. HTTP\n401/403 means the bearer token in `VMWARE_PRIVATEAI_PAIS_TOKEN` is expired or lacks scope — obtain a\nfresh token from your Identity Provider, re-export it, and retry.\n\n## Usage Mode\n\n- **CLI** — interactive inventory / triage, scripting, small or local models (lower context cost).\n- **MCP** — agent-driven operations with structured JSON; run `vmware-privateai mcp` (an installed\n  console script, so no `uvx` network re-resolve — works through enterprise TLS proxies, 踩坑 #25).\n\n## MCP Tools (17 — 16 read, 1 write)\n\n| Category | Tools | R/W |\n|----------|-------|:---:|\n| GPU inventory | `gpu_host_list`, `gpu_host_get`, `gpu_device_list`, `gpu_consumer_list` | Read |\n| GPU utilization | `gpu_utilization` | Read |\n| GPU readiness | `gpu_host_readiness` | Read |\n| Profile catalog | `vgpu_profile_list`, `directpath_profile_list` | Read |\n| Profile validation | `vgpu_profile_validate` | Read |\n| Private AI Service | `pais_model_list`, `pais_model_catalog`, `pais_knowledge_base_list`, `pais_data_source_list` | Read |\n| PAIS monitoring | `pais_monitoring_summary` | Read |\n| Sizing & air-gap | `pais_sizing_advise`, `pais_bundle_verify` | Read |\n| vGPU assignment | `vgpu_assign` | Write |\n\n**INFERRED PAIS paths**: `pais_model_catalog` and `pais_data_source_list` hit PAIS control-plane\npaths that are unconfirmed against a live OpenAPI (踩坑 #36) — a 404 returns a base-URL teaching\nmessage, not a bug. `pais_sizing_advise` and `pais_bundle_verify` need **no connection** (pure\ncomputation / local file parse). `gpu_host_readiness` reports what the vSphere API exposes and says\nso where it cannot (NVIDIA driver / MFT VIB / MIG need nvidia-smi on the host).\n\n**List envelope**: every `*_list` tool returns `{items, returned, limit, offset, total, truncated, hint}`\n— read rows from `items` and check `truncated` before concluding a listing is complete; empty `items`\nwith `truncated:false` means checked-and-none, not a failure. Lists paginate at `limit=50`; filter with\nthe tool's `name`/`vendor`/`host`/`profile`/`vm` arguments rather than paging the whole estate.\n\n**Write safety (normative)**: `vgpu_assign` with `confirm=false` (the default) returns `blast_radius`\nand changes nothing — VM name and id, current and target profile, `device_change` (`edit` the existing\nvGPU device or `add` one), other passthrough devices left untouched, power state, `blockers`,\n`unmeasured`. The acting response carries it too. Show it to the user and pass `confirm=true` only after\nthey agree — the user asking earlier is not agreement, they have not seen it. `confirm=true` is refused\nfor a powered-on or suspended VM, a name shared by several VMs, and an unreadable power state or device\nlist. It **never powers the VM off itself**, waits for the real ReconfigVM task outcome (never a premature \"ok\"), and audits every\napplied change to `~/.vmware/audit.db`.\n\n## CLI Quick Reference\n\n```bash\nvmware-privateai gpu host-list [--name N] [--vendor V]       # hosts with a GPU\nvmware-privateai gpu host-get <host>                          # full per-GPU detail\nvmware-privateai gpu device-list [--host H] [--vendor V]     # physical GPUs (vm_count 0 = idle)\nvmware-privateai gpu consumer-list [--profile P] [--vm V]    # VMs holding a vGPU + profile\nvmware-privateai gpu utilization [--vm V] [--top N]          # real-time GPU %, mem %, temp\nvmware-privateai gpu vgpu-assign <vm> <profile> [--dry-run]  # WRITE — VM must be off; double-confirm\nvmware-privateai gpu readiness [--host H]                   # per-host vGPU/PAIS readiness verdict\nvmware-privateai vgpu profile-list [--host H] [--model M]    # vGPU profile catalog\nvmware-privateai vgpu directpath-list [--vendor V]           # DirectPath profiles (vSphere 9.0+)\nvmware-privateai vgpu validate <vm> <profile>               # pre-flight a vGPU profile change (read-only)\nvmware-privateai pais model-list [--name N]                  # PAIS served models\nvmware-privateai pais model-catalog [--name N]               # PAIS deployable/approved model catalog\nvmware-privateai pais kb-list [--name N]                     # PAIS knowledge bases\nvmware-privateai pais data-source-list [--name N]            # PAIS RAG data sources\nvmware-privateai pais monitoring-summary [--top N]           # fleet GPU rollup (util/mem/temp, hot/idle)\nvmware-privateai pais sizing --model llama-70b               # LLM GPU/storage sizing (no connection)\nvmware-privateai pais bundle-verify <pais.yml>              # local air-gap image inspector (no network)\n```\nFull list: `references/cli-reference.md`. Per-tool response-token estimates: `references/capabilities.md`.\n\n## Troubleshooting\n\n- **`Password not found for target '<t>'. Set environment variable VMWARE_PRIVATEAI_<T>_PASSWORD`** —\n  add that line to `~/.vmware-privateai/.env` and `chmod 600` it, or export it (from a secret manager).\n  The `<T>` is the target name upper-cased with `-`→`_`.\n- **`TLS verification failed for target '<t>'`** — for a self-signed lab set `verify_ssl: false` for\n  that target in `config.yaml`; otherwise install the vCenter CA on this host.\n- **`gpu host-list` returns nothing on a cluster you know has GPUs** — only `shared` / `direct` /\n  `sharedDirect` graphics types count as compute GPUs (the plain host framebuffer is excluded). If real\n  9.x hardware surfaces a GPU under an unexpected type, that is a beta known-limitation — file an issue\n  with the raw `gpu host-get` output so the projection can be widened.\n- **`gpu utilization` shows a VM with `metrics unavailable`** — the NVIDIA host GPU driver is not\n  exposing counters for it (not an error). Note the `gpu.*` perf counters may report at host level on\n  some builds — verify the entity type on real hardware (beta caveat).\n- **`directpath-list` errors with \"needs vCenter 9.0+\"** — DirectPathProfileManager is new in vSphere\n  9.0; on 8.x use `vgpu profile-list` instead (the error routes you there, not an empty list).\n- **PAIS 404 / non-JSON response** — the `/api/v1` prefix is deployment-specific and unconfirmed;\n  check `pais.endpoint` (a proxy or login page returns non-JSON). PAIS 401/403 → refresh the bearer\n  token in `VMWARE_PRIVATEAI_PAIS_TOKEN`.\n\n## Audit & Safety\n\n1. **Source Code** — https://github.com/vmware-skills/VMware-PrivateAI (MIT).\n2. **Config File Contents** — `config.yaml` holds target host/username/port and the `pais.endpoint`\n   only; passwords and the PAIS bearer token live in `~/.vmware-privateai/.env` (0600, obfuscated to\n   `b64:` at rest — obfuscation, not encryption).\n3. **Webhook Data Scope** — none. This skill makes no outbound calls except to the configured\n   vCenter/ESXi targets and PAIS endpoint.\n4. **TLS Verification** — on by default; `verify_ssl: false` is per-target (and `pais.verify_ssl`) and\n   only for self-signed labs.\n5. **Prompt Injection Protection** — all vSphere-supplied and PAIS-supplied text (device/vendor/VM/\n   profile names, PAIS model ids, knowledge-base descriptions) passes through `vmware_policy.sanitize()`\n   (truncation ≤500 chars + C0/C1 control-char stripping); a KB description is the highest-value\n   injection surface here.\n6. **Least Privilege** — read-vs-write authorization is the vCenter role's job: a read-only service\n   account refuses `vgpu_assign`'s ReconfigVM at vCenter, un-bypassably. All writes are recorded in\n   `~/.vmware/audit.db`. See `references/setup-guide.md`.\n\n## License\n\nMIT\n\nFile v1.4.0:_meta.json\n\n{\n  \"ownerId\": \"kn7b067awq2s97bn3d7p5qfhw5827pxc\",\n  \"slug\": \"vmware-privateai\",\n  \"version\": \"1.4.0\",\n  \"publishedAt\": 1789915986383\n}\n\nFile v1.4.0:references/capabilities.md\n\n# vmware-privateai — Capabilities\n\n17 MCP tools (16 read / 1 write) over the vSphere 9.x / VCF 9.1 Web Services API (pyVmomi) plus the\nPrivate AI Service (PAIS) REST API, with two tools that need no connection at all (sizing / bundle).\nEvery vSphere tool accepts an optional `target`; PAIS tools use the `pais:` config section instead.\nTypical response tokens are estimates for a small estate; every `*_list` tool paginates at `limit=50`\nand returns the `{items, returned, limit, offset, total, truncated, hint}` envelope.\n\n## GPU inventory (4 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `gpu_host_list` | R | host, gpu_count, vendors[], vgpu_vms (filter name/vendor) | 60–400 |\n| `gpu_host_get` | R | one host's GPUs: device, type, vendor, memory_mb, pci_id, vm_count | 80–400 |\n| `gpu_device_list` | R | flattened physical GPUs: host, device, type, vendor, pci_id, memory_mb, vm_count (filter host/vendor) | 80–600 |\n| `gpu_consumer_list` | R | vm, profile — the \"who holds a vGPU\" view (filter profile/vm) | 60–500 |\n\n## GPU utilization (1 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `gpu_utilization` | R | vm, profile, gpu_pct, mem_pct, mem_used_kb, temp_c, metrics_available, idle; busiest first, `top` keeps N | 80–500 |\n\nReal-time 20s samples via the vSphere PerformanceManager `gpu.*` counters (require the NVIDIA host GPU\ndriver). A negative sample is vSphere's \"no data\" sentinel and is dropped; a VM with no samples reports\n`metrics_available:false`. **Beta caveat**: the `gpu.*` counters may report at host level on some\nbuilds — verify the entity type on real hardware. Deep per-SM / per-process / MIG-slice telemetry is\n**not** in vSphere (needs NVIDIA DCGM) and no endpoint is invented for it.\n\n## GPU readiness (1 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `gpu_host_readiness` | R | host, vgpu_ready, gpu_count, vendors[], total_gpu_memory_mb, default_graphics_type, vgpu_profiles_offered, active_vgpu_vms, blocking_reasons[], driver_note (filter/scope host) | 100–600 |\n\nCombines `config.graphicsInfo` + `config.graphicsConfig` + the per-host `QueryConfigTarget` profile\ncatalog into a `vgpu_ready` verdict (GPU present + `sharedDirect` mode + ≥1 profile offered). Only\nGPU hosts are returned; a per-host query failure lands in `unreachable_hosts`. The NVIDIA driver /\nMFT VIB version and MIG geometry are **not** in the vSphere API — every item carries a `driver_note`\nrouting to `nvidia-smi` / `esxcli` (spec NO_API; no endpoint invented).\n\n## Profile catalog (2 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `vgpu_profile_list` | R | profile, name, framebuffer_gib, profile_class, sharing, vendor_id, hosts[], host_count, unreachable_hosts[] (filter/scope host, filter model) | 80–600 |\n| `directpath_profile_list` | R | id, name, vendor, description — vCenter-level DirectPath profiles, **vSphere 9.0+** (filter name/vendor) | 60–400 |\n\n`vgpu_profile_list` polls `EnvironmentBrowser.QueryConfigTarget` **per host** — scope with `host` to\navoid polling the whole estate. A host whose per-host query fails lands in `unreachable_hosts` (a\nreachability problem, not \"no profiles\"). `directpath_profile_list` on a pre-9.0 vCenter raises a\nteaching error routing to `vgpu_profile_list` rather than returning an empty list.\n\n## Profile validation (1 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `vgpu_profile_validate` | R | vm, host, current_profile, target_profile, power_state, host_offers_target, target_framebuffer_gib, can_apply, blocking_reasons[] | 80–300 |\n\nRead-only pre-flight for `vgpu_assign`: checks the two ReconfigVM failure modes up front — VM powered\non, or the VM's own host does not offer the target profile. No reconfigure; run `vgpu_assign` once\n`can_apply` is true.\n\n## Private AI Service (4 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `pais_model_list` | R | id, owned_by, created — OpenAI-compatible served `/models` (filter name) | 60–400 |\n| `pais_model_catalog` | R | id, name, status, source — deployable/approved model catalog (filter name) | 60–500 |\n| `pais_knowledge_base_list` | R | id, name, status, description — RAG knowledge bases (filter name) | 80–500 |\n| `pais_data_source_list` | R | id, name, type, status — RAG ingest connectors (filter name) | 60–400 |\n\nPAIS reads go through the bearer-authenticated REST client (`VMWARE_PRIVATEAI_PAIS_TOKEN`). Responses\nare parsed defensively: a bare JSON array or a `{data|items|models|knowledge_bases|…: [...]}` envelope is\naccepted, and every field degrades via `.get()`. **Beta caveat**: the exact `/api/v1` path prefix and\nthe JSON field names are `INFERRED_EXACT` (corroborated by the rendered Broadcom developer portal, not a\ndownloaded OpenAPI or a live deployment) — a 404 usually means a base-URL mismatch, not a bug.\n`pais_model_catalog` (`/api/v1/control/models`) is the least-confirmed of these (a best-guess path).\n\n## PAIS monitoring (1 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `pais_monitoring_summary` | R | vgpu_vms, reporting_vms, idle_vms, hot_vms, gpu_pct avg/max, mem_pct avg/max, temp_c_max, by_profile{}, busiest[], scope_note | 150–500 |\n\nA fleet rollup of the same VERIFIED `gpu.*` perf counters `gpu_utilization` reads — the numbers you\nwould pin to a VCF Ops dashboard. Uses the vCenter connection (not PAIS REST). Deep per-SM / MIG /\npower telemetry needs NVIDIA DCGM and is out of scope (`scope_note` says so).\n\n## Sizing & air-gap (2 read — no connection)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `pais_sizing_advise` | R | model_billions, precision, weights_gib, serving_vram_gib, gpu_options[], storage{}, assumptions{} | 200–400 |\n| `pais_bundle_verify` | R | manifest, image_count, images[], registries_to_mirror[], public_registries[], mutable_images[], warnings[] | 150–800 |\n\nBoth are pure/local — **no vCenter or PAIS call**. `pais_sizing_advise` is a transparent first-\nprinciples heuristic (weights = params × bytes/precision; GPU count = ceil(serving / usable HBM)) and is\nexplicit that LLM inference is compute/HBM-bound, so random IOPS is the wrong axis for the weights.\n`pais_bundle_verify` parses a **local** pais.yml with a real YAML parser (踩坑 #38) and flags the two\nair-gap blockers (public registries to mirror, mutable tags); it never contacts a registry.\n\n## vGPU assignment (1 write)\n| Tool | R/W | Risk | Blast radius |\n|------|:---:|:----:|--------------|\n| `vgpu_assign` | W | high | one VM, **only when powered off** — sets/replaces its vGPU profile via ReconfigVM |\n\n`confirm=false` (default) returns `{\"action\": \"preview\", \"blast_radius\": {...}, \"hint\": ...}` and changes\nnothing. `blast_radius`: `vm`, `vm_id`, `same_name_vm_ids`, `power_state`, `current_profile`,\n`target_profile`, `device_change` (`edit`/`add`), `other_passthrough_devices`,\n`memory_reservation_locked_to_max`, `requires_power_off`, `blockers`, `unmeasured`; the acting response\n(`action: \"applied\"`) carries it too. The older top-level keys (`current_profile`, `target_profile`,\n`power_state`, `requires_power_off`) are still returned and are removed in the next minor release.\n`confirm=true` refuses a powered-on or suspended VM, a name shared by several VMs, and an unreadable power\nstate or device list. It edits the\nVM's **existing vGPU device** (selected by the same predicate the preview uses, never a plain\nDirectPath/SR-IOV passthrough), waits for the real ReconfigVM task outcome (never a premature \"ok\"), and\naudits the result. It never powers the VM off itself — that is `vmware-aiops`.\n\n## Verified API surface (anti-phantom-endpoint gate)\n\nEvery runtime path is pinned in `tests/eval/spec/privateai_endpoints.py` and asserted by a regression\ngate (踩坑 #36):\n\n- **pyVmomi**: `HostSystem.config.graphicsInfo` (VERIFIED), `HostSystem.config.graphicsConfig`\n  (VERIFIED — vGPU mode via `hostDefaultGraphicsType`), `EnvironmentBrowser.QueryConfigTarget →\n  ConfigTarget.vgpuProfileInfo[]` (VERIFIED — the attribute is `vgpuProfileInfo`, not the spec-doc's\n  phantom `vgpu[]`), `content.directPathProfileManager.ListDirectPathProfiles` (VERIFIED, 9.0+),\n  `VirtualMachine.config.hardware.device → VirtualPCIPassthrough.backing.vgpu` (VERIFIED),\n  `VirtualMachine.runtime.powerState` (VERIFIED), `VirtualMachine.runtime.host` (VERIFIED),\n  `ReconfigVM_Task` (VERIFIED, requires VM off).\n- **Perf counters**: `gpu.utilization.average`, `gpu.mem.used.average`, `gpu.mem.usage.average`,\n  `gpu.temperature.average` (VERIFIED).\n- **PAIS REST**: `/api/v1/compatibility/openai/v1/models`, `/api/v1/control/knowledge-bases`,\n  `/api/v1/control/data-sources`, `/api/v1/control/models` (INFERRED_EXACT — path prefix deferred to\n  first live run; `/control/models` is a best-guess pending a live OpenAPI).\n- **NO_API** (code must not invent an endpoint): MIG mode set, GPU driver version, deep GPU telemetry\n  (DCGM), one-call DL-VM deploy, pgvector.\n\nFile v1.4.0:references/cli-reference.md\n\n# vmware-privateai — CLI Reference\n\nFull command list for the `vmware-privateai` Typer CLI. Every read command prints a teaching error and\nexits 1 (never a traceback) on a config / not-found / connection problem. Every command accepts\n`--target <name>` (a vCenter/ESXi target from `config.yaml`; omit for the default) and `--config <path>`\n(override the `~/.vmware-privateai/config.yaml` location). Reads paginate at 50 rows; use the filter\noptions rather than paging the whole estate.\n\n## Top level\n\n```bash\nvmware-privateai version          # print the installed version\nvmware-privateai mcp              # run the stdio MCP server (used by MCP clients)\nvmware-privateai --help           # list command groups: gpu, vgpu, pais\n```\n\n> The `mcp` subcommand is the recommended MCP entry point — it is an installed console script, so it\n> never re-resolves from PyPI the way `uvx` does (踩坑 #25: `uvx` is fragile behind an enterprise TLS\n> proxy). MCP clients should launch `vmware-privateai mcp`.\n\n## `gpu` — GPU inventory, utilization, and vGPU assignment\n\n```bash\nvmware-privateai gpu host-list [--name N] [--vendor V] [--target T] [--config PATH]\n```\nList ESXi hosts that have at least one GPU. Columns: host, gpu_count, vendors, vgpu_vms. `--name`\nsubstring-matches the host name; `--vendor` substring-matches the GPU vendor (e.g. `NVIDIA`).\n\n```bash\nvmware-privateai gpu host-get <host_name> [--target T] [--config PATH]\n```\nFull per-GPU detail for one host: device, graphics type, vendor, memory (MB), pci id, vm_count. A wrong\nhost name prints a teaching error listing the hosts that do have GPUs.\n\n```bash\nvmware-privateai gpu device-list [--host H] [--vendor V] [--target T] [--config PATH]\n```\nFlattened list of physical GPU devices across hosts. Columns: host, device, type, vendor, memory (MB),\nvm_count. `vm_count 0` marks an idle GPU.\n\n```bash\nvmware-privateai gpu consumer-list [--profile P] [--vm V] [--target T] [--config PATH]\n```\nList VMs consuming a vGPU and the profile each holds (e.g. `grid_a100-4c`). `--profile` / `--vm`\nsubstring-filter.\n\n```bash\nvmware-privateai gpu utilization [--vm V] [--top N] [--target T] [--config PATH]\n```\nReal-time (20s sample) GPU utilization per vGPU VM, busiest first: gpu %, mem %, temp (C). `--top N`\nkeeps only the N busiest. A VM with no host-driver samples prints `metrics unavailable` (not an error).\n\n```bash\nvmware-privateai gpu vgpu-assign <vm_name> <profile> [--dry-run] [--target T] [--config PATH]\n```\n**WRITE.** Set a VM's vGPU profile. Always prints the preview (current → target profile, power state,\nrequires_power_off) first. `--dry-run` stops there. Otherwise requires **double confirmation**, then\napplies via ReconfigVM and audits the result. The VM must be powered **off** — a running or suspended VM\nis refused with a teaching error routing you to `vmware-aiops vm_power_off`, and so is a name shared by\nseveral VMs or a VM whose power state or devices could not be read. This command never powers the VM off\nitself.\n\n```bash\nvmware-privateai gpu readiness [--host H] [--target T] [--config PATH]\n```\nPer-host vGPU/PAIS readiness verdict. Columns: host, ready/not-ready, gpu_count, graphics mode,\nprofiles offered, plus `blocking_reasons` for any host that is not ready. Only hosts that have a GPU\nare shown. The NVIDIA driver / MFT VIB version and MIG are **not** in the vSphere API — verify with\n`nvidia-smi` / `esxcli software vib list` on the host.\n\n## `vgpu` — profile catalog\n\n```bash\nvmware-privateai vgpu profile-list [--host H] [--model M] [--target T] [--config PATH]\n```\nThe vGPU profile catalog aggregated across hosts. Columns: profile, framebuffer (GiB), class, sharing,\nhost_count. `--host` both filters and scopes the per-host `QueryConfigTarget` polling to one host;\n`--model` substring-matches the profile/model name. Hosts whose per-host query fails are reported in\n`unreachable_hosts` rather than sinking the whole catalog.\n\n```bash\nvmware-privateai vgpu directpath-list [--name N] [--vendor V] [--target T] [--config PATH]\n```\nvCenter-level DirectPath (dynamic passthrough) profiles. **vSphere 9.0+** — on an older vCenter this\nprints a teaching error routing you to `vgpu profile-list`. Columns: name, id, vendor, description.\n\n```bash\nvmware-privateai vgpu validate <vm_name> <target_profile> [--target T] [--config PATH]\n```\nRead-only pre-flight for a vGPU profile change (the pre-flight for `gpu vgpu-assign`). Checks the two\nReconfigVM failure modes up front — VM powered on, and whether the VM's own host offers the target\nprofile — and prints `can apply` / `blocked` with the reasons. No reconfigure.\n\n## `pais` — Private AI Service (REST), monitoring, sizing, and air-gap\n\nThese talk to the PAIS REST endpoint (`config.yaml` `pais:` section) with the bearer token in\n`VMWARE_PRIVATEAI_PAIS_TOKEN` — separate from the vCenter connection, so they take `--config` but not\n`--target`.\n\n```bash\nvmware-privateai pais model-list [--name N] [--config PATH]\n```\nList models served by PAIS (OpenAI-compatible `/models`). Columns: id, owned_by. `--name`\nsubstring-matches the model id.\n\n```bash\nvmware-privateai pais kb-list [--name N] [--config PATH]\n```\nList PAIS knowledge bases (RAG vector stores). Columns: name/id, status, description.\n\n```bash\nvmware-privateai pais model-catalog [--name N] [--config PATH]\nvmware-privateai pais data-source-list [--name N] [--config PATH]\n```\n`model-catalog` lists models available/approved to **deploy** (distinct from the served `/models`);\n`data-source-list` lists RAG ingest connectors. Both hit **INFERRED** PAIS control-plane paths — a 404\nprints a base-URL teaching message, not a bug.\n\n```bash\nvmware-privateai pais monitoring-summary [--hot-pct P] [--top N] [--target T] [--config PATH]\n```\nFleet GPU rollup (vgpu_vms, reporting/hot/idle, gpu%/mem% avg+max, busiest). Uses the **vCenter**\nconnection (takes `--target`), not PAIS REST. Deep per-SM/MIG telemetry needs NVIDIA DCGM (out of scope).\n\n```bash\nvmware-privateai pais sizing [--model llama-70b | --billions 70] [--precision fp16]\n```\nEstimate GPU memory / GPU count / storage to serve an LLM. **No connection** — a pure planning\nheuristic. Honest that random IOPS is the wrong axis for the model weights.\n\n```bash\nvmware-privateai pais bundle-verify <pais.yml>\n```\nParse a **local** pais.yml and list container images + registries to mirror for an air-gap, flagging\npublic registries and mutable tags. **No network** — does not contact any registry.\n\n## Exit codes\n\n| Code | Meaning |\n|------|---------|\n| 0 | Success |\n| 1 | Teaching error (config missing, target/host/VM not found, connection/TLS failure, PAIS 4xx) |\n\n## Environment variables\n\n| Variable | Purpose |\n|----------|---------|\n| `VMWARE_PRIVATEAI_<TARGET>_PASSWORD` | Password for a vCenter/ESXi target (`<TARGET>` = target name upper-cased, `-`→`_`) |\n| `VMWARE_PRIVATEAI_<TARGET>_USERNAME` | Optional username override (wins over `config.yaml`) |\n| `VMWARE_PRIVATEAI_PAIS_TOKEN` | OIDC/OAuth2 bearer token for the PAIS REST endpoint |\n| `VMWARE_PRIVATEAI_CONFIG` | Optional path to `config.yaml` (primary env for the skill) |\n\nSecret-bearing `*_PASSWORD` / `*_TOKEN` values are auto-obfuscated to `b64:` form in `.env` on load\n(grep-safe; obfuscation, not encryption).\n\nFile v1.4.0:references/setup-guide.md\n\n# vmware-privateai — Setup Guide\n\n> **Disclaimer**: Community-maintained open-source project, **not affiliated with, endorsed by, or\n> sponsored by VMware, Inc., Broadcom Inc., or NVIDIA Corporation.** \"VMware\", \"vSphere\", and \"VCF\" are\n> trademarks of Broadcom; \"NVIDIA\" and \"vGPU\" are trademarks of NVIDIA. Source is publicly auditable\n> under the MIT license.\n\nInstall, credential, and MCP-client configuration for vmware-privateai, plus the Security section.\n\n## 1. Install\n\n```bash\nuv tool install vmware-privateai==1.4.0       # isolated tool env; puts vmware-privateai on PATH\nvmware-privateai version\n```\n\nRequires Python 3.11+ (the MCP server is reflected by FastMCP/Pydantic; older interpreters can raise on\nPEP 604 unions — 踩坑 #33). Runtime deps: pyvmomi, httpx, typer, rich, pyyaml, python-dotenv, mcp, and\n`vmware-policy` (the family audit/policy harness, installed automatically).\n\n## 2. Configure targets\n\nCreate `~/.vmware-privateai/config.yaml`:\n\n```yaml\ntargets:\n  - name: vc-prod\n    host: vcenter-prod.example.com\n    username: administrator@vsphere.local   # optional; env var overrides this\n    type: vcenter                            # or esxi\n    port: 443\n    verify_ssl: true\n    environment: production                  # optional label for policy scoping\n\n# Optional — only needed for the pais model-list / kb-list tools:\npais:\n  endpoint: https://pais.example.com         # base URL; the /api/v1 prefix is added by the client\n  verify_ssl: true\n```\n\nThe first target is the default (used when `--target` / the `target` MCP arg is omitted).\n\n## 3. Credentials — never in config files\n\nPasswords and the PAIS bearer token live only in `~/.vmware-privateai/.env`:\n\n```bash\nmkdir -p ~/.vmware-privateai\ncat >> ~/.vmware-privateai/.env <<'EOF'\nVMWARE_PRIVATEAI_VC_PROD_PASSWORD=your-vcenter-password\nVMWARE_PRIVATEAI_PAIS_TOKEN=your-oidc-bearer-token\nEOF\nchmod 600 ~/.vmware-privateai/.env\n```\n\n- **Per-target password**: `VMWARE_PRIVATEAI_<TARGET>_PASSWORD`, where `<TARGET>` is the target `name`\n  upper-cased with `-` replaced by `_` (so target `vc-prod` → `VMWARE_PRIVATEAI_VC_PROD_PASSWORD`).\n- **Optional username override**: `VMWARE_PRIVATEAI_<TARGET>_USERNAME` wins over `config.yaml` (resolved\n  together with the password on every call, so a rotating sidecar never splits the pair).\n- **PAIS token**: `VMWARE_PRIVATEAI_PAIS_TOKEN` — a short-lived OIDC/OAuth2 bearer token from your\n  Identity Provider (the PAIS API uses `Authorization: Bearer <token>`). It is a secret and is\n  obfuscated to `b64:` at rest exactly like a password.\n- **Secret manager**: any of these vars can be injected from Vault / CyberArk / AWS Secrets Manager /\n  a Kubernetes Secret instead of `.env` — the code reads the environment either way.\n\nOn load, plaintext `*_PASSWORD` / `*_TOKEN` values in `.env` are auto-rewritten to grep-safe `b64:`\nform (obfuscation, not encryption — it defeats casual grep / shoulder-surfing, not a determined reader).\n\n## 4. MCP client configuration\n\nPreferred (installed console script — no `uvx` network re-resolve, works through enterprise TLS\nproxies, 踩坑 #25):\n\n```json\n{\n  \"mcpServers\": {\n    \"vmware-privateai\": {\n      \"command\": \"vmware-privateai\",\n      \"args\": [\"mcp\"],\n      \"env\": { \"VMWARE_PRIVATEAI_CONFIG\": \"~/.vmware-privateai/config.yaml\" }\n    }\n  }\n}\n```\n\nFallback (`uvx` — re-resolves from PyPI each start; if your network runs a TLS MitM proxy, add\n`\"UV_NATIVE_TLS\": \"true\"` to `env`):\n\n```json\n{\n  \"mcpServers\": {\n    \"vmware-privateai\": {\n      \"command\": \"uvx\",\n      \"args\": [\"--from\", \"vmware-privateai==1.4.0\", \"vmware-privateai-mcp\"],\n      \"env\": { \"VMWARE_PRIVATEAI_CONFIG\": \"~/.vmware-privateai/config.yaml\" }\n    }\n  }\n}\n```\n\nThis layout works with any MCP-compatible client (Claude Desktop, Claude Code, Goose, etc.).\n\n## 5. Verify\n\n```bash\nvmware-privateai gpu host-list             # lists hosts that have a GPU\nvmware-privateai vgpu profile-list         # the assignable vGPU profile catalog\nvmware-privateai pais model-list           # PAIS served models (if configured)\n```\n\nA teaching error here names exactly what to fix (missing password, unresolvable host, TLS, missing PAIS\nendpoint/token) — follow the message.\n\n## Security\n\n> **Disclaimer**: not affiliated with, endorsed by, or sponsored by VMware, Inc., Broadcom Inc., or\n> NVIDIA Corporation. See the full policy in the repo-root `SECURITY.md`.\n\n1. **Source Code** — https://github.com/vmware-skills/VMware-PrivateAI (MIT), publicly auditable.\n2. **Config File Contents** — `config.yaml` holds only target host/username/port and the\n   `pais.endpoint`. No passwords, no tokens. Secrets live in `~/.vmware-privateai/.env` (chmod 600,\n   `b64:` obfuscated at rest).\n3. **Webhook Data Scope** — none. No webhooks and no outbound network calls except to the configured\n   vCenter/ESXi targets and the PAIS endpoint.\n4. **TLS Verification** — on by default. `verify_ssl: false` (per target) and `pais.verify_ssl: false`\n   are intended only for self-signed lab certificates.\n5. **Prompt Injection Protection** — all vSphere-supplied and PAIS-supplied text (device / vendor / VM\n   / vGPU-profile names, PAIS model ids, and knowledge-base descriptions — the highest-value injection\n   surface here) passes through `vmware_policy.sanitize()` (≤500-char truncation + C0/C1 control-char\n   stripping) before it reaches the model.\n6. **Least Privilege** — read-vs-write authorization is delegated to the vCenter service account's RBAC\n   role: a read-only account refuses `vgpu_assign`'s ReconfigVM at vCenter, un-bypassably (the one place\n   the control cannot be stepped around by a shell). Recommend a dedicated service account scoped to the\n   GPU clusters. All writes are recorded to `~/.vmware/audit.db` (and the CLI companion\n   `~/.vmware-privateai/audit.log`).\n\n### Static analysis\n\n```bash\nuvx bandit -r vmware_privateai/            # target: 0 Medium+ issues\n```\n\nFile v1.4.0:skill-card.md\n\n## Description:\n\nProvides GPU and VMware Private AI Service operations for vSphere and VCF environments, including GPU inventory, vGPU consumers and utilization, profile catalogs and assignment, and PAIS model and knowledge-base listing.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[zw008](https://clawhub.ai/user/zw008)\n\n### License/Terms of Use:\n\nMIT\n\n## Use Case:\n\nDevelopers and infrastructure engineers use this skill to inspect GPU capacity, monitor vGPU usage, list Private AI Service resources, and safely plan or apply vGPU profile changes in VMware Private AI Foundation with NVIDIA environments.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: vCenter and PAIS credentials can expose infrastructure access if handled carelessly.\n\nMitigation: Use a dedicated least-privilege vCenter service account scoped to GPU clusters, prefer environment or secret-manager injection, and protect any local .env file with restricted permissions.\n\nRisk: The vGPU assignment tool can change VM configuration.\n\nMitigation: Review the blast-radius preview and approve the write only after confirming the target VM, profile, power state, and user intent.\n\nRisk: Disabling TLS verification outside self-signed lab environments can expose credentials or infrastructure data.\n\nMitigation: Keep TLS verification enabled for normal use and install the appropriate vCenter or PAIS certificate authority instead of disabling verification.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/zw008/skills/vmware-privateai)\n- [Capabilities reference](artifact/references/capabilities.md)\n- [CLI reference](artifact/references/cli-reference.md)\n- [Setup guide](artifact/references/setup-guide.md)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Shell commands, Configuration, Guidance]\n\n**Output Format:** [Markdown guidance with inline shell commands and structured JSON from MCP tools]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [List operations may be paginated; vGPU assignment is a write action that should be previewed and explicitly confirmed.]\n\n## Skill Version(s):\n\n1.4.0 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v1.3.0: 6 files, 18446 bytes\n\nFiles: references/capabilities.md (9124b), references/cli-reference.md (7276b), references/setup-guide.md (5951b), skill-card.md (2545b), SKILL.md (15504b), _meta.json (135b)\n\nFile v1.3.0:SKILL.md\n\n---\nname: vmware-privateai\ndescription: >\n  Use this skill whenever the user needs the GPU / AI-infrastructure layer of VMware Private AI\n  Foundation with NVIDIA (PAIF-N) on vSphere 9.x / VCF 9.1: inventory GPU hosts and physical GPU\n  devices, see which VMs consume a vGPU and the profile each holds, read real-time GPU utilization,\n  list the vGPU and DirectPath profile catalog, assign a VM's vGPU profile, and list Private AI\n  Service (PAIS) served models and knowledge bases. Always use this skill for \"list GPU hosts\",\n  \"which VMs are using a vGPU\", \"GPU utilization\", \"assign a vGPU profile\", \"list vGPU profiles\",\n  \"list served models\" when the context is explicitly VMware / vSphere / VCF Private AI / NVIDIA\n  vGPU. Do NOT use for the backing VM's power/snapshot/clone/migrate (use vmware-aiops), read-only\n  vSphere inventory/alarms/host health (use vmware-monitor), or GPU-enabled Tanzu Kubernetes\n  (use vmware-vks). This skill is the GPU lens; vmware-aiops owns the VM lifecycle behind it.\ninstaller:\n  kind: uv\n  package: vmware-privateai\nallowed-tools:\n  - Bash\nmetadata: {\"openclaw\":{\"requires\":{\"anyBins\":[\"vmware-privateai\",\"uvx\"]},\"optional\":{\"env\":[\"VMWARE_PRIVATEAI_CONFIG\"]}}}\n---\n\n# VMware Private AI (Foundation with NVIDIA) — GPU & Model-Serving Ops\n\n> **Disclaimer**: Community-maintained open-source project, **not affiliated with, endorsed by, or\n> sponsored by VMware, Inc., Broadcom Inc., or NVIDIA Corporation.** \"VMware\", \"vSphere\", and \"VCF\"\n> are trademarks of Broadcom; \"NVIDIA\" and \"vGPU\" are trademarks of NVIDIA. Source is publicly\n> auditable under the MIT license.\n\nThe GPU / AI-infrastructure lens for the VMware skill family — GPU host & device inventory, vGPU\nconsumers, real-time GPU utilization, the vGPU / DirectPath profile catalog, vGPU assignment, and\n**Private AI Service (PAIS)** served models and knowledge bases — over the **vSphere 9.x / VCF 9.1**\nWeb Services API (pyVmomi) plus the PAIS REST API.\n\n> **Companion skills**: [vmware-aiops](https://github.com/vmware-skills/VMware-AIops) (the vCenter VMs\n> behind AI workloads — power/snapshot/clone), [vmware-vks](https://github.com/vmware-skills/VMware-VKS)\n> (GPU-enabled Tanzu Kubernetes), [vmware-monitor](https://github.com/vmware-skills/VMware-Monitor)\n> (read-only vSphere health).\n\n> **Status: v1.0.1 — still beta in substance.** Skill #15 of the family. The jump from 0.2.x to\n> 1.0.1 is a distribution fix, not a maturity claim: the withdrawn first release used 1.0.0, and\n> ClawHub resolves `latest` by version order, so every 0.x release was invisible there. The beta\n> caveats below all still stand. Every API path is\n> verified against official Broadcom/NVIDIA sources before use (`tests/eval/spec/privateai_endpoints.py`)\n> — no endpoints written from memory. GET-response *field names* and the exact PAIS paths are\n> defensive and pending validation against live 9.x hardware (see Troubleshooting). Governed by the\n> family harness (audit + policy + teaching errors); read-vs-write authorization is delegated to the\n> vCenter service account's RBAC role.\n\n## What This Skill Does\n\n| Category | Tools | Count | Read/Write |\n|----------|-------|:-----:|:----------:|\n| **GPU inventory** | host list/get, device list, vGPU consumer list | 4 | 4 R |\n| **GPU utilization** | real-time per-vGPU-VM utilization (gpu %, mem %, temp) | 1 | 1 R |\n| **GPU readiness** | per-host vGPU/PAIS readiness verdict + blocking reasons | 1 | 1 R |\n| **Profile catalog** | vGPU profile list, DirectPath profile list | 2 | 2 R |\n| **Profile validation** | pre-flight a vGPU profile change (power state + host offers it) | 1 | 1 R |\n| **vGPU assignment** | set a VM's vGPU profile (VM must be powered off) | 1 | 1 W |\n| **Private AI Service** | served-model list, model catalog, knowledge-base list, data-source list | 4 | 4 R |\n| **PAIS monitoring** | fleet GPU rollup (util/mem/temp, hot/idle, busiest) | 1 | 1 R |\n| **Sizing & air-gap** | LLM GPU/storage sizing advisor, local pais.yml image inspector | 2 | 2 R |\n\n**17 MCP tools (16 read / 1 write).** Reads are strictly non-destructive. The single write\n(`vgpu_assign`) previews its blast radius, refuses a powered-on VM, never powers a VM off itself, is\ndouble-confirmed at the CLI, and is audit-logged. Pre-flight the write with `vgpu_profile_validate`.\n\n## Quick Install\n\n```bash\nuv tool install vmware-privateai==1.3.0\nvmware-privateai version\nvmware-privateai gpu host-list        # first read — lists hosts that have a GPU\n```\n\nConfig lives in `~/.vmware-privateai/config.yaml` (targets + optional `pais:` section); passwords and\nthe PAIS bearer token live in `~/.vmware-privateai/.env` (chmod 600). See `references/setup-guide.md`.\n\n## When to Use This Skill\n\nUse vmware-privateai for the **GPU / AI-infrastructure layer**: which hosts and physical devices have\nGPUs, which VMs hold a vGPU and what profile, real-time GPU utilization, the assignable vGPU /\nDirectPath profile catalog, changing a VM's vGPU profile, and the models / knowledge bases served by\nPrivate AI Service — when the context is explicitly VMware / vSphere / VCF Private AI / NVIDIA vGPU.\n\n**Do NOT use when**: the task is the backing VM's lifecycle — power on/off, snapshot, clone, migrate,\nreconfigure CPU/RAM (→ **vmware-aiops**); read-only vSphere inventory, alarms, or host health\n(→ **vmware-monitor**); or GPU-enabled Tanzu Kubernetes / Supervisor namespaces (→ **vmware-vks**).\n`vgpu_assign` deliberately does **not** power the VM off — that is vmware-aiops's job, kept separate\nso this skill's blast radius stays \"one VM, when it is already off\".\n\n## Related Skills — Skill Routing\n\n| The user wants… | Skill |\n|-----------------|-------|\n| Inventory GPUs / vGPU consumers / GPU utilization / assign a vGPU profile | **vmware-privateai** (this) |\n| List PAIS served models / knowledge bases | **vmware-privateai** (this) |\n| Power off / snapshot / clone / migrate the backing vCenter VM | vmware-aiops |\n| Read-only vSphere inventory / alarms / host health | vmware-monitor |\n| GPU-enabled Tanzu Kubernetes clusters / namespaces | vmware-vks |\n| Multi-step GPU workflow with approval + rollback | vmware-pilot |\n\n## Common Workflows\n\n**1. Find an idle GPU and reassign a VM's vGPU profile.**\n```\nvmware-privateai gpu device-list --vendor NVIDIA     # find GPUs; vm_count 0 = idle\nvmware-privateai gpu consumer-list                   # who holds a vGPU, and which profile\nvmware-privateai vgpu profile-list --host esx-07     # profiles that host can hand a VM\nvmware-privateai gpu vgpu-assign fin-train-01 grid_a100-4c --dry-run   # preview blast radius\n# power the VM off with vmware-aiops, THEN:\nvmware-privateai gpu vgpu-assign fin-train-01 grid_a100-4c             # double-confirm + audit\n```\n*Failure branch*: if `vgpu-assign` (confirm) refuses with \"VM is powered on — a vGPU change needs the\nVM powered off\", run `vmware-aiops vm_power_off 'fin-train-01'` first, then re-run. If it fails with\n\"profile not offered by the VM's host / GPU lacks free framebuffer\", run\n`vmware-privateai gpu host-get <that VM's host>` to see the valid profiles and free capacity.\n\n**2. Triage GPU utilization across the estate.**\n```\nvmware-privateai gpu utilization --top 10            # busiest vGPU VMs first\nvmware-privateai gpu host-list --vendor NVIDIA       # which hosts carry the load\n```\n*Failure branch*: a VM showing `metrics unavailable (no host driver?)` is not an error — the NVIDIA\nhost GPU driver is not exposing counters for it (`metrics_available:false`). Deep per-SM / per-process\n/ MIG-slice telemetry is **not** available via vSphere; use NVIDIA DCGM on the host for that.\n\n**3. See what Private AI Service is serving.**\n```\nvmware-privateai pais model-list                     # OpenAI-compatible /models\nvmware-privateai pais kb-list                         # RAG knowledge bases\n```\n*Failure branch*: HTTP 404 usually means a base-URL mismatch, not a bug — the `/api/v1` PAIS path\nprefix is deployment-specific and unconfirmed (beta). Check `pais.endpoint` in config.yaml. HTTP\n401/403 means the bearer token in `VMWARE_PRIVATEAI_PAIS_TOKEN` is expired or lacks scope — obtain a\nfresh token from your Identity Provider, re-export it, and retry.\n\n## Usage Mode\n\n- **CLI** — interactive inventory / triage, scripting, small or local models (lower context cost).\n- **MCP** — agent-driven operations with structured JSON; run `vmware-privateai mcp` (an installed\n  console script, so no `uvx` network re-resolve — works through enterprise TLS proxies, 踩坑 #25).\n\n## MCP Tools (17 — 16 read, 1 write)\n\n| Category | Tools | R/W |\n|----------|-------|:---:|\n| GPU inventory | `gpu_host_list`, `gpu_host_get`, `gpu_device_list`, `gpu_consumer_list` | Read |\n| GPU utilization | `gpu_utilization` | Read |\n| GPU readiness | `gpu_host_readiness` | Read |\n| Profile catalog | `vgpu_profile_list`, `directpath_profile_list` | Read |\n| Profile validation | `vgpu_profile_validate` | Read |\n| Private AI Service | `pais_model_list`, `pais_model_catalog`, `pais_knowledge_base_list`, `pais_data_source_list` | Read |\n| PAIS monitoring | `pais_monitoring_summary` | Read |\n| Sizing & air-gap | `pais_sizing_advise`, `pais_bundle_verify` | Read |\n| vGPU assignment | `vgpu_assign` | Write |\n\n**INFERRED PAIS paths**: `pais_model_catalog` and `pais_data_source_list` hit PAIS control-plane\npaths that are unconfirmed against a live OpenAPI (踩坑 #36) — a 404 returns a base-URL teaching\nmessage, not a bug. `pais_sizing_advise` and `pais_bundle_verify` need **no connection** (pure\ncomputation / local file parse). `gpu_host_readiness` reports what the vSphere API exposes and says\nso where it cannot (NVIDIA driver / MFT VIB / MIG need nvidia-smi on the host).\n\n**List envelope**: every `*_list` tool returns `{items, returned, limit, offset, total, truncated, hint}`\n— read rows from `items` and check `truncated` before concluding a listing is complete; empty `items`\nwith `truncated:false` means checked-and-none, not a failure. Lists paginate at `limit=50`; filter with\nthe tool's `name`/`vendor`/`host`/`profile`/`vm` arguments rather than paging the whole estate.\n\n**Write safety (normative)**: `vgpu_assign` with `confirm=false` (the default) returns `blast_radius`\nand changes nothing — VM name and id, current and target profile, `device_change` (`edit` the existing\nvGPU device or `add` one), other passthrough devices left untouched, power state, `blockers`,\n`unmeasured`. The acting response carries it too. Show it to the user and pass `confirm=true` only after\nthey agree — the user asking earlier is not agreement, they have not seen it. `confirm=true` is refused\nfor a powered-on or suspended VM, a name shared by several VMs, and an unreadable power state or device\nlist. It **never powers the VM off itself**, waits for the real ReconfigVM task outcome (never a premature \"ok\"), and audits every\napplied change to `~/.vmware/audit.db`.\n\n## CLI Quick Reference\n\n```bash\nvmware-privateai gpu host-list [--name N] [--vendor V]       # hosts with a GPU\nvmware-privateai gpu host-get <host>                          # full per-GPU detail\nvmware-privateai gpu device-list [--host H] [--vendor V]     # physical GPUs (vm_count 0 = idle)\nvmware-privateai gpu consumer-list [--profile P] [--vm V]    # VMs holding a vGPU + profile\nvmware-privateai gpu utilization [--vm V] [--top N]          # real-time GPU %, mem %, temp\nvmware-privateai gpu vgpu-assign <vm> <profile> [--dry-run]  # WRITE — VM must be off; double-confirm\nvmware-privateai gpu readiness [--host H]                   # per-host vGPU/PAIS readiness verdict\nvmware-privateai vgpu profile-list [--host H] [--model M]    # vGPU profile catalog\nvmware-privateai vgpu directpath-list [--vendor V]           # DirectPath profiles (vSphere 9.0+)\nvmware-privateai vgpu validate <vm> <profile>               # pre-flight a vGPU profile change (read-only)\nvmware-privateai pais model-list [--name N]                  # PAIS served models\nvmware-privateai pais model-catalog [--name N]               # PAIS deployable/approved model catalog\nvmware-privateai pais kb-list [--name N]                     # PAIS knowledge bases\nvmware-privateai pais data-source-list [--name N]            # PAIS RAG data sources\nvmware-privateai pais monitoring-summary [--top N]           # fleet GPU rollup (util/mem/temp, hot/idle)\nvmware-privateai pais sizing --model llama-70b               # LLM GPU/storage sizing (no connection)\nvmware-privateai pais bundle-verify <pais.yml>              # local air-gap image inspector (no network)\n```\nFull list: `references/cli-reference.md`. Per-tool response-token estimates: `references/capabilities.md`.\n\n## Troubleshooting\n\n- **`Password not found for target '<t>'. Set environment variable VMWARE_PRIVATEAI_<T>_PASSWORD`** —\n  add that line to `~/.vmware-privateai/.env` and `chmod 600` it, or export it (from a secret manager).\n  The `<T>` is the target name upper-cased with `-`→`_`.\n- **`TLS verification failed for target '<t>'`** — for a self-signed lab set `verify_ssl: false` for\n  that target in `config.yaml`; otherwise install the vCenter CA on this host.\n- **`gpu host-list` returns nothing on a cluster you know has GPUs** — only `shared` / `direct` /\n  `sharedDirect` graphics types count as compute GPUs (the plain host framebuffer is excluded). If real\n  9.x hardware surfaces a GPU under an unexpected type, that is a beta known-limitation — file an issue\n  with the raw `gpu host-get` output so the projection can be widened.\n- **`gpu utilization` shows a VM with `metrics unavailable`** — the NVIDIA host GPU driver is not\n  exposing counters for it (not an error). Note the `gpu.*` perf counters may report at host level on\n  some builds — verify the entity type on real hardware (beta caveat).\n- **`directpath-list` errors with \"needs vCenter 9.0+\"** — DirectPathProfileManager is new in vSphere\n  9.0; on 8.x use `vgpu profile-list` instead (the error routes you there, not an empty list).\n- **PAIS 404 / non-JSON response** — the `/api/v1` prefix is deployment-specific and unconfirmed;\n  check `pais.endpoint` (a proxy or login page returns non-JSON). PAIS 401/403 → refresh the bearer\n  token in `VMWARE_PRIVATEAI_PAIS_TOKEN`.\n\n## Audit & Safety\n\n1. **Source Code** — https://github.com/vmware-skills/VMware-PrivateAI (MIT).\n2. **Config File Contents** — `config.yaml` holds target host/username/port and the `pais.endpoint`\n   only; passwords and the PAIS bearer token live in `~/.vmware-privateai/.env` (0600, obfuscated to\n   `b64:` at rest — obfuscation, not encryption).\n3. **Webhook Data Scope** — none. This skill makes no outbound calls except to the configured\n   vCenter/ESXi targets and PAIS endpoint.\n4. **TLS Verification** — on by default; `verify_ssl: false` is per-target (and `pais.verify_ssl`) and\n   only for self-signed labs.\n5. **Prompt Injection Protection** — all vSphere-supplied and PAIS-supplied text (device/vendor/VM/\n   profile names, PAIS model ids, knowledge-base descriptions) passes through `vmware_policy.sanitize()`\n   (truncation ≤500 chars + C0/C1 control-char stripping); a KB description is the highest-value\n   injection surface here.\n6. **Least Privilege** — read-vs-write authorization is the vCenter role's job: a read-only service\n   account refuses `vgpu_assign`'s ReconfigVM at vCenter, un-bypassably. All writes are recorded in\n   `~/.vmware/audit.db`. See `references/setup-guide.md`.\n\n## License\n\nMIT\n\nFile v1.3.0:_meta.json\n\n{\n  \"ownerId\": \"kn7b067awq2s97bn3d7p5qfhw5827pxc\",\n  \"slug\": \"vmware-privateai\",\n  \"version\": \"1.3.0\",\n  \"publishedAt\": 1789790149524\n}\n\nFile v1.3.0:references/capabilities.md\n\n# vmware-privateai — Capabilities\n\n17 MCP tools (16 read / 1 write) over the vSphere 9.x / VCF 9.1 Web Services API (pyVmomi) plus the\nPrivate AI Service (PAIS) REST API, with two tools that need no connection at all (sizing / bundle).\nEvery vSphere tool accepts an optional `target`; PAIS tools use the `pais:` config section instead.\nTypical response tokens are estimates for a small estate; every `*_list` tool paginates at `limit=50`\nand returns the `{items, returned, limit, offset, total, truncated, hint}` envelope.\n\n## GPU inventory (4 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `gpu_host_list` | R | host, gpu_count, vendors[], vgpu_vms (filter name/vendor) | 60–400 |\n| `gpu_host_get` | R | one host's GPUs: device, type, vendor, memory_mb, pci_id, vm_count | 80–400 |\n| `gpu_device_list` | R | flattened physical GPUs: host, device, type, vendor, pci_id, memory_mb, vm_count (filter host/vendor) | 80–600 |\n| `gpu_consumer_list` | R | vm, profile — the \"who holds a vGPU\" view (filter profile/vm) | 60–500 |\n\n## GPU utilization (1 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `gpu_utilization` | R | vm, profile, gpu_pct, mem_pct, mem_used_kb, temp_c, metrics_available, idle; busiest first, `top` keeps N | 80–500 |\n\nReal-time 20s samples via the vSphere PerformanceManager `gpu.*` counters (require the NVIDIA host GPU\ndriver). A negative sample is vSphere's \"no data\" sentinel and is dropped; a VM with no samples reports\n`metrics_available:false`. **Beta caveat**: the `gpu.*` counters may report at host level on some\nbuilds — verify the entity type on real hardware. Deep per-SM / per-process / MIG-slice telemetry is\n**not** in vSphere (needs NVIDIA DCGM) and no endpoint is invented for it.\n\n## GPU readiness (1 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `gpu_host_readiness` | R | host, vgpu_ready, gpu_count, vendors[], total_gpu_memory_mb, default_graphics_type, vgpu_profiles_offered, active_vgpu_vms, blocking_reasons[], driver_note (filter/scope host) | 100–600 |\n\nCombines `config.graphicsInfo` + `config.graphicsConfig` + the per-host `QueryConfigTarget` profile\ncatalog into a `vgpu_ready` verdict (GPU present + `sharedDirect` mode + ≥1 profile offered). Only\nGPU hosts are returned; a per-host query failure lands in `unreachable_hosts`. The NVIDIA driver /\nMFT VIB version and MIG geometry are **not** in the vSphere API — every item carries a `driver_note`\nrouting to `nvidia-smi` / `esxcli` (spec NO_API; no endpoint invented).\n\n## Profile catalog (2 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `vgpu_profile_list` | R | profile, name, framebuffer_gib, profile_class, sharing, vendor_id, hosts[], host_count, unreachable_hosts[] (filter/scope host, filter model) | 80–600 |\n| `directpath_profile_list` | R | id, name, vendor, description — vCenter-level DirectPath profiles, **vSphere 9.0+** (filter name/vendor) | 60–400 |\n\n`vgpu_profile_list` polls `EnvironmentBrowser.QueryConfigTarget` **per host** — scope with `host` to\navoid polling the whole estate. A host whose per-host query fails lands in `unreachable_hosts` (a\nreachability problem, not \"no profiles\"). `directpath_profile_list` on a pre-9.0 vCenter raises a\nteaching error routing to `vgpu_profile_list` rather than returning an empty list.\n\n## Profile validation (1 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `vgpu_profile_validate` | R | vm, host, current_profile, target_profile, power_state, host_offers_target, target_framebuffer_gib, can_apply, blocking_reasons[] | 80–300 |\n\nRead-only pre-flight for `vgpu_assign`: checks the two ReconfigVM failure modes up front — VM powered\non, or the VM's own host does not offer the target profile. No reconfigure; run `vgpu_assign` once\n`can_apply` is true.\n\n## Private AI Service (4 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `pais_model_list` | R | id, owned_by, created — OpenAI-compatible served `/models` (filter name) | 60–400 |\n| `pais_model_catalog` | R | id, name, status, source — deployable/approved model catalog (filter name) | 60–500 |\n| `pais_knowledge_base_list` | R | id, name, status, description — RAG knowledge bases (filter name) | 80–500 |\n| `pais_data_source_list` | R | id, name, type, status — RAG ingest connectors (filter name) | 60–400 |\n\nPAIS reads go through the bearer-authenticated REST client (`VMWARE_PRIVATEAI_PAIS_TOKEN`). Responses\nare parsed defensively: a bare JSON array or a `{data|items|models|knowledge_bases|…: [...]}` envelope is\naccepted, and every field degrades via `.get()`. **Beta caveat**: the exact `/api/v1` path prefix and\nthe JSON field names are `INFERRED_EXACT` (corroborated by the rendered Broadcom developer portal, not a\ndownloaded OpenAPI or a live deployment) — a 404 usually means a base-URL mismatch, not a bug.\n`pais_model_catalog` (`/api/v1/control/models`) is the least-confirmed of these (a best-guess path).\n\n## PAIS monitoring (1 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `pais_monitoring_summary` | R | vgpu_vms, reporting_vms, idle_vms, hot_vms, gpu_pct avg/max, mem_pct avg/max, temp_c_max, by_profile{}, busiest[], scope_note | 150–500 |\n\nA fleet rollup of the same VERIFIED `gpu.*` perf counters `gpu_utilization` reads — the numbers you\nwould pin to a VCF Ops dashboard. Uses the vCenter connection (not PAIS REST). Deep per-SM / MIG /\npower telemetry needs NVIDIA DCGM and is out of scope (`scope_note` says so).\n\n## Sizing & air-gap (2 read — no connection)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `pais_sizing_advise` | R | model_billions, precision, weights_gib, serving_vram_gib, gpu_options[], storage{}, assumptions{} | 200–400 |\n| `pais_bundle_verify` | R | manifest, image_count, images[], registries_to_mirror[], public_registries[], mutable_images[], warnings[] | 150–800 |\n\nBoth are pure/local — **no vCenter or PAIS call**. `pais_sizing_advise` is a transparent first-\nprinciples heuristic (weights = params × bytes/precision; GPU count = ceil(serving / usable HBM)) and is\nexplicit that LLM inference is compute/HBM-bound, so random IOPS is the wrong axis for the weights.\n`pais_bundle_verify` parses a **local** pais.yml with a real YAML parser (踩坑 #38) and flags the two\nair-gap blockers (public registries to mirror, mutable tags); it never contacts a registry.\n\n## vGPU assignment (1 write)\n| Tool | R/W | Risk | Blast radius |\n|------|:---:|:----:|--------------|\n| `vgpu_assign` | W | high | one VM, **only when powered off** — sets/replaces its vGPU profile via ReconfigVM |\n\n`confirm=false` (default) returns `{\"action\": \"preview\", \"blast_radius\": {...}, \"hint\": ...}` and changes\nnothing. `blast_radius`: `vm`, `vm_id`, `same_name_vm_ids`, `power_state`, `current_profile`,\n`target_profile`, `device_change` (`edit`/`add`), `other_passthrough_devices`,\n`memory_reservation_locked_to_max`, `requires_power_off`, `blockers`, `unmeasured`; the acting response\n(`action: \"applied\"`) carries it too. The older top-level keys (`current_profile`, `target_profile`,\n`power_state`, `requires_power_off`) are still returned and are removed in the next minor release.\n`confirm=true` refuses a powered-on or suspended VM, a name shared by several VMs, and an unreadable power\nstate or device list. It edits the\nVM's **existing vGPU device** (selected by the same predicate the preview uses, never a plain\nDirectPath/SR-IOV passthrough), waits for the real ReconfigVM task outcome (never a premature \"ok\"), and\naudits the result. It never powers the VM off itself — that is `vmware-aiops`.\n\n## Verified API surface (anti-phantom-endpoint gate)\n\nEvery runtime path is pinned in `tests/eval/spec/privateai_endpoints.py` and asserted by a regression\ngate (踩坑 #36):\n\n- **pyVmomi**: `HostSystem.config.graphicsInfo` (VERIFIED), `HostSystem.config.graphicsConfig`\n  (VERIFIED — vGPU mode via `hostDefaultGraphicsType`), `EnvironmentBrowser.QueryConfigTarget →\n  ConfigTarget.vgpuProfileInfo[]` (VERIFIED — the attribute is `vgpuProfileInfo`, not the spec-doc's\n  phantom `vgpu[]`), `content.directPathProfileManager.ListDirectPathProfiles` (VERIFIED, 9.0+),\n  `VirtualMachine.config.hardware.device → VirtualPCIPassthrough.backing.vgpu` (VERIFIED),\n  `VirtualMachine.runtime.powerState` (VERIFIED), `VirtualMachine.runtime.host` (VERIFIED),\n  `ReconfigVM_Task` (VERIFIED, requires VM off).\n- **Perf counters**: `gpu.utilization.average`, `gpu.mem.used.average`, `gpu.mem.usage.average`,\n  `gpu.temperature.average` (VERIFIED).\n- **PAIS REST**: `/api/v1/compatibility/openai/v1/models`, `/api/v1/control/knowledge-bases`,\n  `/api/v1/control/data-sources`, `/api/v1/control/models` (INFERRED_EXACT — path prefix deferred to\n  first live run; `/control/models` is a best-guess pending a live OpenAPI).\n- **NO_API** (code must not invent an endpoint): MIG mode set, GPU driver version, deep GPU telemetry\n  (DCGM), one-call DL-VM deploy, pgvector.\n\nFile v1.3.0:references/cli-reference.md\n\n# vmware-privateai — CLI Reference\n\nFull command list for the `vmware-privateai` Typer CLI. Every read command prints a teaching error and\nexits 1 (never a traceback) on a config / not-found / connection problem. Every command accepts\n`--target <name>` (a vCenter/ESXi target from `config.yaml`; omit for the default) and `--config <path>`\n(override the `~/.vmware-privateai/config.yaml` location). Reads paginate at 50 rows; use the filter\noptions rather than paging the whole estate.\n\n## Top level\n\n```bash\nvmware-privateai version          # print the installed version\nvmware-privateai mcp              # run the stdio MCP server (used by MCP clients)\nvmware-privateai --help           # list command groups: gpu, vgpu, pais\n```\n\n> The `mcp` subcommand is the recommended MCP entry point — it is an installed console script, so it\n> never re-resolves from PyPI the way `uvx` does (踩坑 #25: `uvx` is fragile behind an enterprise TLS\n> proxy). MCP clients should launch `vmware-privateai mcp`.\n\n## `gpu` — GPU inventory, utilization, and vGPU assignment\n\n```bash\nvmware-privateai gpu host-list [--name N] [--vendor V] [--target T] [--config PATH]\n```\nList ESXi hosts that have at least one GPU. Columns: host, gpu_count, vendors, vgpu_vms. `--name`\nsubstring-matches the host name; `--vendor` substring-matches the GPU vendor (e.g. `NVIDIA`).\n\n```bash\nvmware-privateai gpu host-get <host_name> [--target T] [--config PATH]\n```\nFull per-GPU detail for one host: device, graphics type, vendor, memory (MB), pci id, vm_count. A wrong\nhost name prints a teaching error listing the hosts that do have GPUs.\n\n```bash\nvmware-privateai gpu device-list [--host H] [--vendor V] [--target T] [--config PATH]\n```\nFlattened list of physical GPU devices across hosts. Columns: host, device, type, vendor, memory (MB),\nvm_count. `vm_count 0` marks an idle GPU.\n\n```bash\nvmware-privateai gpu consumer-list [--profile P] [--vm V] [--target T] [--config PATH]\n```\nList VMs consuming a vGPU and the profile each holds (e.g. `grid_a100-4c`). `--profile` / `--vm`\nsubstring-filter.\n\n```bash\nvmware-privateai gpu utilization [--vm V] [--top N] [--target T] [--config PATH]\n```\nReal-time (20s sample) GPU utilization per vGPU VM, busiest first: gpu %, mem %, temp (C). `--top N`\nkeeps only the N busiest. A VM with no host-driver samples prints `metrics unavailable` (not an error).\n\n```bash\nvmware-privateai gpu vgpu-assign <vm_name> <profile> [--dry-run] [--target T] [--config PATH]\n```\n**WRITE.** Set a VM's vGPU profile. Always prints the preview (current → target profile, power state,\nrequires_power_off) first. `--dry-run` stops there. Otherwise requires **double confirmation**, then\napplies via ReconfigVM and audits the result. The VM must be powered **off** — a running or suspended VM\nis refused with a teaching error routing you to `vmware-aiops vm_power_off`, and so is a name shared by\nseveral VMs or a VM whose power state or devices could not be read. This command never powers the VM off\nitself.\n\n```bash\nvmware-privateai gpu readiness [--host H] [--target T] [--config PATH]\n```\nPer-host vGPU/PAIS readiness verdict. Columns: host, ready/not-ready, gpu_count, graphics mode,\nprofiles offered, plus `blocking_reasons` for any host that is not ready. Only hosts that have a GPU\nare shown. The NVIDIA driver / MFT VIB version and MIG are **not** in the vSphere API — verify with\n`nvidia-smi` / `esxcli software vib list` on the host.\n\n## `vgpu` — profile catalog\n\n```bash\nvmware-privateai vgpu profile-list [--host H] [--model M] [--target T] [--config PATH]\n```\nThe vGPU profile catalog aggregated across hosts. Columns: profile, framebuffer (GiB), class, sharing,\nhost_count. `--host` both filters and scopes the per-host `QueryConfigTarget` polling to one host;\n`--model` substring-matches the profile/model name. Hosts whose per-host query fails are reported in\n`unreachable_hosts` rather than sinking the whole catalog.\n\n```bash\nvmware-privateai vgpu directpath-list [--name N] [--vendor V] [--target T] [--config PATH]\n```\nvCenter-level DirectPath (dynamic passthrough) profiles. **vSphere 9.0+** — on an older vCenter this\nprints a teaching error routing you to `vgpu profile-list`. Columns: name, id, vendor, description.\n\n```bash\nvmware-privateai vgpu validate <vm_name> <target_profile> [--target T] [--config PATH]\n```\nRead-only pre-flight for a vGPU profile change (the pre-flight for `gpu vgpu-assign`). Checks the two\nReconfigVM failure modes up front — VM powered on, and whether the VM's own host offers the target\nprofile — and prints `can apply` / `blocked` with the reasons. No reconfigure.\n\n## `pais` — Private AI Service (REST), monitoring, sizing, and air-gap\n\nThese talk to the PAIS REST endpoint (`config.yaml` `pais:` section) with the bearer token in\n`VMWARE_PRIVATEAI_PAIS_TOKEN` — separate from the vCenter connection, so they take `--config` but not\n`--target`.\n\n```bash\nvmware-privateai pais model-list [--name N] [--config PATH]\n```\nList models served by PAIS (OpenAI-compatible `/models`). Columns: id, owned_by. `--name`\nsubstring-matches the model id.\n\n```bash\nvmware-privateai pais kb-list [--name N] [--config PATH]\n```\nList PAIS knowledge bases (RAG vector stores). Columns: name/id, status, description.\n\n```bash\nvmware-privateai pais model-catalog [--name N] [--config PATH]\nvmware-privateai pais data-source-list [--name N] [--config PATH]\n```\n`model-catalog` lists models available/approved to **deploy** (distinct from the served `/models`);\n`data-source-list` lists RAG ingest connectors. Both hit **INFERRED** PAIS control-plane paths — a 404\nprints a base-URL teaching message, not a bug.\n\n```bash\nvmware-privateai pais monitoring-summary [--hot-pct P] [--top N] [--target T] [--config PATH]\n```\nFleet GPU rollup (vgpu_vms, reporting/hot/idle, gpu%/mem% avg+max, busiest). Uses the **vCenter**\nconnection (takes `--target`), not PAIS REST. Deep per-SM/MIG telemetry needs NVIDIA DCGM (out of scope).\n\n```bash\nvmware-privateai pais sizing [--model llama-70b | --billions 70] [--precision fp16]\n```\nEstimate GPU memory / GPU count / storage to serve an LLM. **No connection** — a pure planning\nheuristic. Honest that random IOPS is the wrong axis for the model weights.\n\n```bash\nvmware-privateai pais bundle-verify <pais.yml>\n```\nParse a **local** pais.yml and list container images + registries to mirror for an air-gap, flagging\npublic registries and mutable tags. **No network** — does not contact any registry.\n\n## Exit codes\n\n| Code | Meaning |\n|------|---------|\n| 0 | Success |\n| 1 | Teaching error (config missing, target/host/VM not found, connection/TLS failure, PAIS 4xx) |\n\n## Environment variables\n\n| Variable | Purpose |\n|----------|---------|\n| `VMWARE_PRIVATEAI_<TARGET>_PASSWORD` | Password for a vCenter/ESXi target (`<TARGET>` = target name upper-cased, `-`→`_`) |\n| `VMWARE_PRIVATEAI_<TARGET>_USERNAME` | Optional username override (wins over `config.yaml`) |\n| `VMWARE_PRIVATEAI_PAIS_TOKEN` | OIDC/OAuth2 bearer token for the PAIS REST endpoint |\n| `VMWARE_PRIVATEAI_CONFIG` | Optional path to `config.yaml` (primary env for the skill) |\n\nSecret-bearing `*_PASSWORD` / `*_TOKEN` values are auto-obfuscated to `b64:` form in `.env` on load\n(grep-safe; obfuscation, not encryption).\n\nFile v1.3.0:references/setup-guide.md\n\n# vmware-privateai — Setup Guide\n\n> **Disclaimer**: Community-maintained open-source project, **not affiliated with, endorsed by, or\n> sponsored by VMware, Inc., Broadcom Inc., or NVIDIA Corporation.** \"VMware\", \"vSphere\", and \"VCF\" are\n> trademarks of Broadcom; \"NVIDIA\" and \"vGPU\" are trademarks of NVIDIA. Source is publicly auditable\n> under the MIT license.\n\nInstall, credential, and MCP-client configuration for vmware-privateai, plus the Security section.\n\n## 1. Install\n\n```bash\nuv tool install vmware-privateai==1.3.0       # isolated tool env; puts vmware-privateai on PATH\nvmware-privateai version\n```\n\nRequires Python 3.11+ (the MCP server is reflected by FastMCP/Pydantic; older interpreters can raise on\nPEP 604 unions — 踩坑 #33). Runtime deps: pyvmomi, httpx, typer, rich, pyyaml, python-dotenv, mcp, and\n`vmware-policy` (the family audit/policy harness, installed automatically).\n\n## 2. Configure targets\n\nCreate `~/.vmware-privateai/config.yaml`:\n\n```yaml\ntargets:\n  - name: vc-prod\n    host: vcenter-prod.example.com\n    username: administrator@vsphere.local   # optional; env var overrides this\n    type: vcenter                            # or esxi\n    port: 443\n    verify_ssl: true\n    environment: production                  # optional label for policy scoping\n\n# Optional — only needed for the pais model-list / kb-list tools:\npais:\n  endpoint: https://pais.example.com         # base URL; the /api/v1 prefix is added by the client\n  verify_ssl: true\n```\n\nThe first target is the default (used when `--target` / the `target` MCP arg is omitted).\n\n## 3. Credentials — never in config files\n\nPasswords and the PAIS bearer token live only in `~/.vmware-privateai/.env`:\n\n```bash\nmkdir -p ~/.vmware-privateai\ncat >> ~/.vmware-privateai/.env <<'EOF'\nVMWARE_PRIVATEAI_VC_PROD_PASSWORD=your-vcenter-password\nVMWARE_PRIVATEAI_PAIS_TOKEN=your-oidc-bearer-token\nEOF\nchmod 600 ~/.vmware-privateai/.env\n```\n\n- **Per-target password**: `VMWARE_PRIVATEAI_<TARGET>_PASSWORD`, where `<TARGET>` is the target `name`\n  upper-cased with `-` replaced by `_` (so target `vc-prod` → `VMWARE_PRIVATEAI_VC_PROD_PASSWORD`).\n- **Optional username override**: `VMWARE_PRIVATEAI_<TARGET>_USERNAME` wins over `config.yaml` (resolved\n  together with the password on every call, so a rotating sidecar never splits the pair).\n- **PAIS token**: `VMWARE_PRIVATEAI_PAIS_TOKEN` — a short-lived OIDC/OAuth2 bearer token from your\n  Identity Provider (the PAIS API uses `Authorization: Bearer <token>`). It is a secret and is\n  obfuscated to `b64:` at rest exactly like a password.\n- **Secret manager**: any of these vars can be injected from Vault / CyberArk / AWS Secrets Manager /\n  a Kubernetes Secret instead of `.env` — the code reads the environment either way.\n\nOn load, plaintext `*_PASSWORD` / `*_TOKEN` values in `.env` are auto-rewritten to grep-safe `b64:`\nform (obfuscation, not encryption — it defeats casual grep / shoulder-surfing, not a determined reader).\n\n## 4. MCP client configuration\n\nPreferred (installed console script — no `uvx` network re-resolve, works through enterprise TLS\nproxies, 踩坑 #25):\n\n```json\n{\n  \"mcpServers\": {\n    \"vmware-privateai\": {\n      \"command\": \"vmware-privateai\",\n      \"args\": [\"mcp\"],\n      \"env\": { \"VMWARE_PRIVATEAI_CONFIG\": \"~/.vmware-privateai/config.yaml\" }\n    }\n  }\n}\n```\n\nFallback (`uvx` — re-resolves from PyPI each start; if your network runs a TLS MitM proxy, add\n`\"UV_NATIVE_TLS\": \"true\"` to `env`):\n\n```json\n{\n  \"mcpServers\": {\n    \"vmware-privateai\": {\n      \"command\": \"uvx\",\n      \"args\": [\"--from\", \"vmware-privateai==1.3.0\", \"vmware-privateai-mcp\"],\n      \"env\": { \"VMWARE_PRIVATEAI_CONFIG\": \"~/.vmware-privateai/config.yaml\" }\n    }\n  }\n}\n```\n\nThis layout works with any MCP-compatible client (Claude Desktop, Claude Code, Goose, etc.).\n\n## 5. Verify\n\n```bash\nvmware-privateai gpu host-list             # lists hosts that have a GPU\nvmware-privateai vgpu profile-list         # the assignable vGPU profile catalog\nvmware-privateai pais model-list           # PAIS served models (if configured)\n```\n\nA teaching error here names exactly what to fix (missing password, unresolvable host, TLS, missing PAIS\nendpoint/token) — follow the message.\n\n## Security\n\n> **Disclaimer**: not affiliated with, endorsed by, or sponsored by VMware, Inc., Broadcom Inc., or\n> NVIDIA Corporation. See the full policy in the repo-root `SECURITY.md`.\n\n1. **Source Code** — https://github.com/vmware-skills/VMware-PrivateAI (MIT), publicly auditable.\n2. **Config File Contents** — `config.yaml` holds only target host/username/port and the\n   `pais.endpoint`. No passwords, no tokens. Secrets live in `~/.vmware-privateai/.env` (chmod 600,\n   `b64:` obfuscated at rest).\n3. **Webhook Data Scope** — none. No webhooks and no outbound network calls except to the configured\n   vCenter/ESXi targets and the PAIS endpoint.\n4. **TLS Verification** — on by default. `verify_ssl: false` (per target) and `pais.verify_ssl: false`\n   are intended only for self-signed lab certificates.\n5. **Prompt Injection Protection** — all vSphere-supplied and PAIS-supplied text (device / vendor / VM\n   / vGPU-profile names, PAIS model ids, and knowledge-base descriptions — the highest-value injection\n   surface here) passes through `vmware_policy.sanitize()` (≤500-char truncation + C0/C1 control-char\n   stripping) before it reaches the model.\n6. **Least Privilege** — read-vs-write authorization is delegated to the vCenter service account's RBAC\n   role: a read-only account refuses `vgpu_assign`'s ReconfigVM at vCenter, un-bypassably (the one place\n   the control cannot be stepped around by a shell). Recommend a dedicated service account scoped to the\n   GPU clusters. All writes are recorded to `~/.vmware/audit.db` (and the CLI companion\n   `~/.vmware-privateai/audit.log`).\n\n### Static analysis\n\n```bash\nuvx bandit -r vmware_privateai/            # target: 0 Medium+ issues\n```\n\nFile v1.3.0:skill-card.md\n\n## Description:\n\nvmware-privateai helps agents inspect VMware Private AI Foundation with NVIDIA GPU infrastructure, list PAIS model assets, and safely preview or apply one powered-off VM vGPU profile change.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[zw008](https://clawhub.ai/user/zw008)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and infrastructure operators use this skill to inventory GPU hosts and vGPU consumers, check utilization and readiness, inspect PAIS-served models and knowledge bases, and manage vGPU profile assignment within VMware Private AI environments.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill can read sensitive VMware GPU, vGPU, VM, and PAIS inventory through configured vCenter/ESXi and PAIS endpoints.\n\nMitigation: Use a dedicated least-privilege vCenter service account, scope it to the required GPU clusters, and inject credentials from a secret manager or short-lived environment variables.\n\nRisk: The vgpu_assign action can change one VM's vGPU profile when the VM is powered off.\n\nMitigation: Review the blast-radius preview, pre-flight with vgpu_profile_validate, and pass confirm=true only after explicit user approval.\n\nRisk: Disabling TLS verification or storing persistent secrets can weaken protection for vCenter and PAIS credentials.\n\nMitigation: Keep TLS verification enabled outside self-signed labs and avoid persistent .env secrets when managed secret injection is available.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/zw008/skills/vmware-privateai)\n- [Capabilities reference](artifact/references/capabilities.md)\n- [CLI reference](artifact/references/cli-reference.md)\n- [Setup guide](artifact/references/setup-guide.md)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown guidance with shell commands, configuration snippets, and structured MCP results]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Read tools return paginated inventory or status data; the vGPU assignment write first returns a blast-radius preview and requires explicit confirmation.]\n\n## Skill Version(s):\n\n1.3.0 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v1.2.3: 6 files, 17943 bytes\n\nFiles: references/capabilities.md (8615b), references/cli-reference.md (7168b), references/setup-guide.md (5951b), skill-card.md (2408b), SKILL.md (15156b), _meta.json (135b)\n\nFile v1.2.3:SKILL.md\n\n---\nname: vmware-privateai\ndescription: >\n  Use this skill whenever the user needs the GPU / AI-infrastructure layer of VMware Private AI\n  Foundation with NVIDIA (PAIF-N) on vSphere 9.x / VCF 9.1: inventory GPU hosts and physical GPU\n  devices, see which VMs consume a vGPU and the profile each holds, read real-time GPU utilization,\n  list the vGPU and DirectPath profile catalog, assign a VM's vGPU profile, and list Private AI\n  Service (PAIS) served models and knowledge bases. Always use this skill for \"list GPU hosts\",\n  \"which VMs are using a vGPU\", \"GPU utilization\", \"assign a vGPU profile\", \"list vGPU profiles\",\n  \"list served models\" when the context is explicitly VMware / vSphere / VCF Private AI / NVIDIA\n  vGPU. Do NOT use for the backing VM's power/snapshot/clone/migrate (use vmware-aiops), read-only\n  vSphere inventory/alarms/host health (use vmware-monitor), or GPU-enabled Tanzu Kubernetes\n  (use vmware-vks). This skill is the GPU lens; vmware-aiops owns the VM lifecycle behind it.\ninstaller:\n  kind: uv\n  package: vmware-privateai\nallowed-tools:\n  - Bash\nmetadata: {\"openclaw\":{\"requires\":{\"anyBins\":[\"vmware-privateai\",\"uvx\"]},\"optional\":{\"env\":[\"VMWARE_PRIVATEAI_CONFIG\"]}}}\n---\n\n# VMware Private AI (Foundation with NVIDIA) — GPU & Model-Serving Ops\n\n> **Disclaimer**: Community-maintained open-source project, **not affiliated with, endorsed by, or\n> sponsored by VMware, Inc., Broadcom Inc., or NVIDIA Corporation.** \"VMware\", \"vSphere\", and \"VCF\"\n> are trademarks of Broadcom; \"NVIDIA\" and \"vGPU\" are trademarks of NVIDIA. Source is publicly\n> auditable under the MIT license.\n\nThe GPU / AI-infrastructure lens for the VMware skill family — GPU host & device inventory, vGPU\nconsumers, real-time GPU utilization, the vGPU / DirectPath profile catalog, vGPU assignment, and\n**Private AI Service (PAIS)** served models and knowledge bases — over the **vSphere 9.x / VCF 9.1**\nWeb Services API (pyVmomi) plus the PAIS REST API.\n\n> **Companion skills**: [vmware-aiops](https://github.com/vmware-skills/VMware-AIops) (the vCenter VMs\n> behind AI workloads — power/snapshot/clone), [vmware-vks](https://github.com/vmware-skills/VMware-VKS)\n> (GPU-enabled Tanzu Kubernetes), [vmware-monitor](https://github.com/vmware-skills/VMware-Monitor)\n> (read-only vSphere health).\n\n> **Status: v1.0.1 — still beta in substance.** Skill #15 of the family. The jump from 0.2.x to\n> 1.0.1 is a distribution fix, not a maturity claim: the withdrawn first release used 1.0.0, and\n> ClawHub resolves `latest` by version order, so every 0.x release was invisible there. The beta\n> caveats below all still stand. Every API path is\n> verified against official Broadcom/NVIDIA sources before use (`tests/eval/spec/privateai_endpoints.py`)\n> — no endpoints written from memory. GET-response *field names* and the exact PAIS paths are\n> defensive and pending validation against live 9.x hardware (see Troubleshooting). Governed by the\n> family harness (audit + policy + teaching errors); read-vs-write authorization is delegated to the\n> vCenter service account's RBAC role.\n\n## What This Skill Does\n\n| Category | Tools | Count | Read/Write |\n|----------|-------|:-----:|:----------:|\n| **GPU inventory** | host list/get, device list, vGPU consumer list | 4 | 4 R |\n| **GPU utilization** | real-time per-vGPU-VM utilization (gpu %, mem %, temp) | 1 | 1 R |\n| **GPU readiness** | per-host vGPU/PAIS readiness verdict + blocking reasons | 1 | 1 R |\n| **Profile catalog** | vGPU profile list, DirectPath profile list | 2 | 2 R |\n| **Profile validation** | pre-flight a vGPU profile change (power state + host offers it) | 1 | 1 R |\n| **vGPU assignment** | set a VM's vGPU profile (VM must be powered off) | 1 | 1 W |\n| **Private AI Service** | served-model list, model catalog, knowledge-base list, data-source list | 4 | 4 R |\n| **PAIS monitoring** | fleet GPU rollup (util/mem/temp, hot/idle, busiest) | 1 | 1 R |\n| **Sizing & air-gap** | LLM GPU/storage sizing advisor, local pais.yml image inspector | 2 | 2 R |\n\n**17 MCP tools (16 read / 1 write).** Reads are strictly non-destructive. The single write\n(`vgpu_assign`) previews its blast radius, refuses a powered-on VM, never powers a VM off itself, is\ndouble-confirmed at the CLI, and is audit-logged. Pre-flight the write with `vgpu_profile_validate`.\n\n## Quick Install\n\n```bash\nuv tool install vmware-privateai==1.2.3\nvmware-privateai version\nvmware-privateai gpu host-list        # first read — lists hosts that have a GPU\n```\n\nConfig lives in `~/.vmware-privateai/config.yaml` (targets + optional `pais:` section); passwords and\nthe PAIS bearer token live in `~/.vmware-privateai/.env` (chmod 600). See `references/setup-guide.md`.\n\n## When to Use This Skill\n\nUse vmware-privateai for the **GPU / AI-infrastructure layer**: which hosts and physical devices have\nGPUs, which VMs hold a vGPU and what profile, real-time GPU utilization, the assignable vGPU /\nDirectPath profile catalog, changing a VM's vGPU profile, and the models / knowledge bases served by\nPrivate AI Service — when the context is explicitly VMware / vSphere / VCF Private AI / NVIDIA vGPU.\n\n**Do NOT use when**: the task is the backing VM's lifecycle — power on/off, snapshot, clone, migrate,\nreconfigure CPU/RAM (→ **vmware-aiops**); read-only vSphere inventory, alarms, or host health\n(→ **vmware-monitor**); or GPU-enabled Tanzu Kubernetes / Supervisor namespaces (→ **vmware-vks**).\n`vgpu_assign` deliberately does **not** power the VM off — that is vmware-aiops's job, kept separate\nso this skill's blast radius stays \"one VM, when it is already off\".\n\n## Related Skills — Skill Routing\n\n| The user wants… | Skill |\n|-----------------|-------|\n| Inventory GPUs / vGPU consumers / GPU utilization / assign a vGPU profile | **vmware-privateai** (this) |\n| List PAIS served models / knowledge bases | **vmware-privateai** (this) |\n| Power off / snapshot / clone / migrate the backing vCenter VM | vmware-aiops |\n| Read-only vSphere inventory / alarms / host health | vmware-monitor |\n| GPU-enabled Tanzu Kubernetes clusters / namespaces | vmware-vks |\n| Multi-step GPU workflow with approval + rollback | vmware-pilot |\n\n## Common Workflows\n\n**1. Find an idle GPU and reassign a VM's vGPU profile.**\n```\nvmware-privateai gpu device-list --vendor NVIDIA     # find GPUs; vm_count 0 = idle\nvmware-privateai gpu consumer-list                   # who holds a vGPU, and which profile\nvmware-privateai vgpu profile-list --host esx-07     # profiles that host can hand a VM\nvmware-privateai gpu vgpu-assign fin-train-01 grid_a100-4c --dry-run   # preview blast radius\n# power the VM off with vmware-aiops, THEN:\nvmware-privateai gpu vgpu-assign fin-train-01 grid_a100-4c             # double-confirm + audit\n```\n*Failure branch*: if `vgpu-assign` (confirm) refuses with \"VM is powered on — a vGPU change needs the\nVM powered off\", run `vmware-aiops vm_power_off 'fin-train-01'` first, then re-run. If it fails with\n\"profile not offered by the VM's host / GPU lacks free framebuffer\", run\n`vmware-privateai gpu host-get <that VM's host>` to see the valid profiles and free capacity.\n\n**2. Triage GPU utilization across the estate.**\n```\nvmware-privateai gpu utilization --top 10            # busiest vGPU VMs first\nvmware-privateai gpu host-list --vendor NVIDIA       # which hosts carry the load\n```\n*Failure branch*: a VM showing `metrics unavailable (no host driver?)` is not an error — the NVIDIA\nhost GPU driver is not exposing counters for it (`metrics_available:false`). Deep per-SM / per-process\n/ MIG-slice telemetry is **not** available via vSphere; use NVIDIA DCGM on the host for that.\n\n**3. See what Private AI Service is serving.**\n```\nvmware-privateai pais model-list                     # OpenAI-compatible /models\nvmware-privateai pais kb-list                         # RAG knowledge bases\n```\n*Failure branch*: HTTP 404 usually means a base-URL mismatch, not a bug — the `/api/v1` PAIS path\nprefix is deployment-specific and unconfirmed (beta). Check `pais.endpoint` in config.yaml. HTTP\n401/403 means the bearer token in `VMWARE_PRIVATEAI_PAIS_TOKEN` is expired or lacks scope — obtain a\nfresh token from your Identity Provider, re-export it, and retry.\n\n## Usage Mode\n\n- **CLI** — interactive inventory / triage, scripting, small or local models (lower context cost).\n- **MCP** — agent-driven operations with structured JSON; run `vmware-privateai mcp` (an installed\n  console script, so no `uvx` network re-resolve — works through enterprise TLS proxies, 踩坑 #25).\n\n## MCP Tools (17 — 16 read, 1 write)\n\n| Category | Tools | R/W |\n|----------|-------|:---:|\n| GPU inventory | `gpu_host_list`, `gpu_host_get`, `gpu_device_list`, `gpu_consumer_list` | Read |\n| GPU utilization | `gpu_utilization` | Read |\n| GPU readiness | `gpu_host_readiness` | Read |\n| Profile catalog | `vgpu_profile_list`, `directpath_profile_list` | Read |\n| Profile validation | `vgpu_profile_validate` | Read |\n| Private AI Service | `pais_model_list`, `pais_model_catalog`, `pais_knowledge_base_list`, `pais_data_source_list` | Read |\n| PAIS monitoring | `pais_monitoring_summary` | Read |\n| Sizing & air-gap | `pais_sizing_advise`, `pais_bundle_verify` | Read |\n| vGPU assignment | `vgpu_assign` | Write |\n\n**INFERRED PAIS paths**: `pais_model_catalog` and `pais_data_source_list` hit PAIS control-plane\npaths that are unconfirmed against a live OpenAPI (踩坑 #36) — a 404 returns a base-URL teaching\nmessage, not a bug. `pais_sizing_advise` and `pais_bundle_verify` need **no connection** (pure\ncomputation / local file parse). `gpu_host_readiness` reports what the vSphere API exposes and says\nso where it cannot (NVIDIA driver / MFT VIB / MIG need nvidia-smi on the host).\n\n**List envelope**: every `*_list` tool returns `{items, returned, limit, offset, total, truncated, hint}`\n— read rows from `items` and check `truncated` before concluding a listing is complete; empty `items`\nwith `truncated:false` means checked-and-none, not a failure. Lists paginate at `limit=50`; filter with\nthe tool's `name`/`vendor`/`host`/`profile`/`vm` arguments rather than paging the whole estate.\n\n**Write safety (normative)**: `vgpu_assign` with `confirm=false` (the default) previews only —\ncurrent profile, target profile, power state, and that a power-off is required — without acting.\n`confirm=true` applies it, but refuses a powered-on VM with a teaching error. It **never powers the VM\noff itself**, waits for the real ReconfigVM task outcome (never a premature \"ok\"), and audits every\napplied change to `~/.vmware/audit.db`.\n\n## CLI Quick Reference\n\n```bash\nvmware-privateai gpu host-list [--name N] [--vendor V]       # hosts with a GPU\nvmware-privateai gpu host-get <host>                          # full per-GPU detail\nvmware-privateai gpu device-list [--host H] [--vendor V]     # physical GPUs (vm_count 0 = idle)\nvmware-privateai gpu consumer-list [--profile P] [--vm V]    # VMs holding a vGPU + profile\nvmware-privateai gpu utilization [--vm V] [--top N]          # real-time GPU %, mem %, temp\nvmware-privateai gpu vgpu-assign <vm> <profile> [--dry-run]  # WRITE — VM must be off; double-confirm\nvmware-privateai gpu readiness [--host H]                   # per-host vGPU/PAIS readiness verdict\nvmware-privateai vgpu profile-list [--host H] [--model M]    # vGPU profile catalog\nvmware-privateai vgpu directpath-list [--vendor V]           # DirectPath profiles (vSphere 9.0+)\nvmware-privateai vgpu validate <vm> <profile>               # pre-flight a vGPU profile change (read-only)\nvmware-privateai pais model-list [--name N]                  # PAIS served models\nvmware-privateai pais model-catalog [--name N]               # PAIS deployable/approved model catalog\nvmware-privateai pais kb-list [--name N]                     # PAIS knowledge bases\nvmware-privateai pais data-source-list [--name N]            # PAIS RAG data sources\nvmware-privateai pais monitoring-summary [--top N]           # fleet GPU rollup (util/mem/temp, hot/idle)\nvmware-privateai pais sizing --model llama-70b               # LLM GPU/storage sizing (no connection)\nvmware-privateai pais bundle-verify <pais.yml>              # local air-gap image inspector (no network)\n```\nFull list: `references/cli-reference.md`. Per-tool response-token estimates: `references/capabilities.md`.\n\n## Troubleshooting\n\n- **`Password not found for target '<t>'. Set environment variable VMWARE_PRIVATEAI_<T>_PASSWORD`** —\n  add that line to `~/.vmware-privateai/.env` and `chmod 600` it, or export it (from a secret manager).\n  The `<T>` is the target name upper-cased with `-`→`_`.\n- **`TLS verification failed for target '<t>'`** — for a self-signed lab set `verify_ssl: false` for\n  that target in `config.yaml`; otherwise install the vCenter CA on this host.\n- **`gpu host-list` returns nothing on a cluster you know has GPUs** — only `shared` / `direct` /\n  `sharedDirect` graphics types count as compute GPUs (the plain host framebuffer is excluded). If real\n  9.x hardware surfaces a GPU under an unexpected type, that is a beta known-limitation — file an issue\n  with the raw `gpu host-get` output so the projection can be widened.\n- **`gpu utilization` shows a VM with `metrics unavailable`** — the NVIDIA host GPU driver is not\n  exposing counters for it (not an error). Note the `gpu.*` perf counters may report at host level on\n  some builds — verify the entity type on real hardware (beta caveat).\n- **`directpath-list` errors with \"needs vCenter 9.0+\"** — DirectPathProfileManager is new in vSphere\n  9.0; on 8.x use `vgpu profile-list` instead (the error routes you there, not an empty list).\n- **PAIS 404 / non-JSON response** — the `/api/v1` prefix is deployment-specific and unconfirmed;\n  check `pais.endpoint` (a proxy or login page returns non-JSON). PAIS 401/403 → refresh the bearer\n  token in `VMWARE_PRIVATEAI_PAIS_TOKEN`.\n\n## Audit & Safety\n\n1. **Source Code** — https://github.com/vmware-skills/VMware-PrivateAI (MIT).\n2. **Config File Contents** — `config.yaml` holds target host/username/port and the `pais.endpoint`\n   only; passwords and the PAIS bearer token live in `~/.vmware-privateai/.env` (0600, obfuscated to\n   `b64:` at rest — obfuscation, not encryption).\n3. **Webhook Data Scope** — none. This skill makes no outbound calls except to the configured\n   vCenter/ESXi targets and PAIS endpoint.\n4. **TLS Verification** — on by default; `verify_ssl: false` is per-target (and `pais.verify_ssl`) and\n   only for self-signed labs.\n5. **Prompt Injection Protection** — all vSphere-supplied and PAIS-supplied text (device/vendor/VM/\n   profile names, PAIS model ids, knowledge-base descriptions) passes through `vmware_policy.sanitize()`\n   (truncation ≤500 chars + C0/C1 control-char stripping); a KB description is the highest-value\n   injection surface here.\n6. **Least Privilege** — read-vs-write authorization is the vCenter role's job: a read-only service\n   account refuses `vgpu_assign`'s ReconfigVM at vCenter, un-bypassably. All writes are recorded in\n   `~/.vmware/audit.db`. See `references/setup-guide.md`.\n\n## License\n\nMIT\n\nFile v1.2.3:_meta.json\n\n{\n  \"ownerId\": \"kn7b067awq2s97bn3d7p5qfhw5827pxc\",\n  \"slug\": \"vmware-privateai\",\n  \"version\": \"1.2.3\",\n  \"publishedAt\": 1789535946101\n}\n\nFile v1.2.3:references/capabilities.md\n\n# vmware-privateai — Capabilities\n\n17 MCP tools (16 read / 1 write) over the vSphere 9.x / VCF 9.1 Web Services API (pyVmomi) plus the\nPrivate AI Service (PAIS) REST API, with two tools that need no connection at all (sizing / bundle).\nEvery vSphere tool accepts an optional `target`; PAIS tools use the `pais:` config section instead.\nTypical response tokens are estimates for a small estate; every `*_list` tool paginates at `limit=50`\nand returns the `{items, returned, limit, offset, total, truncated, hint}` envelope.\n\n## GPU inventory (4 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `gpu_host_list` | R | host, gpu_count, vendors[], vgpu_vms (filter name/vendor) | 60–400 |\n| `gpu_host_get` | R | one host's GPUs: device, type, vendor, memory_mb, pci_id, vm_count | 80–400 |\n| `gpu_device_list` | R | flattened physical GPUs: host, device, type, vendor, pci_id, memory_mb, vm_count (filter host/vendor) | 80–600 |\n| `gpu_consumer_list` | R | vm, profile — the \"who holds a vGPU\" view (filter profile/vm) | 60–500 |\n\n## GPU utilization (1 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `gpu_utilization` | R | vm, profile, gpu_pct, mem_pct, mem_used_kb, temp_c, metrics_available, idle; busiest first, `top` keeps N | 80–500 |\n\nReal-time 20s samples via the vSphere PerformanceManager `gpu.*` counters (require the NVIDIA host GPU\ndriver). A negative sample is vSphere's \"no data\" sentinel and is dropped; a VM with no samples reports\n`metrics_available:false`. **Beta caveat**: the `gpu.*` counters may report at host level on some\nbuilds — verify the entity type on real hardware. Deep per-SM / per-process / MIG-slice telemetry is\n**not** in vSphere (needs NVIDIA DCGM) and no endpoint is invented for it.\n\n## GPU readiness (1 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `gpu_host_readiness` | R | host, vgpu_ready, gpu_count, vendors[], total_gpu_memory_mb, default_graphics_type, vgpu_profiles_offered, active_vgpu_vms, blocking_reasons[], driver_note (filter/scope host) | 100–600 |\n\nCombines `config.graphicsInfo` + `config.graphicsConfig` + the per-host `QueryConfigTarget` profile\ncatalog into a `vgpu_ready` verdict (GPU present + `sharedDirect` mode + ≥1 profile offered). Only\nGPU hosts are returned; a per-host query failure lands in `unreachable_hosts`. The NVIDIA driver /\nMFT VIB version and MIG geometry are **not** in the vSphere API — every item carries a `driver_note`\nrouting to `nvidia-smi` / `esxcli` (spec NO_API; no endpoint invented).\n\n## Profile catalog (2 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `vgpu_profile_list` | R | profile, name, framebuffer_gib, profile_class, sharing, vendor_id, hosts[], host_count, unreachable_hosts[] (filter/scope host, filter model) | 80–600 |\n| `directpath_profile_list` | R | id, name, vendor, description — vCenter-level DirectPath profiles, **vSphere 9.0+** (filter name/vendor) | 60–400 |\n\n`vgpu_profile_list` polls `EnvironmentBrowser.QueryConfigTarget` **per host** — scope with `host` to\navoid polling the whole estate. A host whose per-host query fails lands in `unreachable_hosts` (a\nreachability problem, not \"no profiles\"). `directpath_profile_list` on a pre-9.0 vCenter raises a\nteaching error routing to `vgpu_profile_list` rather than returning an empty list.\n\n## Profile validation (1 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `vgpu_profile_validate` | R | vm, host, current_profile, target_profile, power_state, host_offers_target, target_framebuffer_gib, can_apply, blocking_reasons[] | 80–300 |\n\nRead-only pre-flight for `vgpu_assign`: checks the two ReconfigVM failure modes up front — VM powered\non, or the VM's own host does not offer the target profile. No reconfigure; run `vgpu_assign` once\n`can_apply` is true.\n\n## Private AI Service (4 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `pais_model_list` | R | id, owned_by, created — OpenAI-compatible served `/models` (filter name) | 60–400 |\n| `pais_model_catalog` | R | id, name, status, source — deployable/approved model catalog (filter name) | 60–500 |\n| `pais_knowledge_base_list` | R | id, name, status, description — RAG knowledge bases (filter name) | 80–500 |\n| `pais_data_source_list` | R | id, name, type, status — RAG ingest connectors (filter name) | 60–400 |\n\nPAIS reads go through the bearer-authenticated REST client (`VMWARE_PRIVATEAI_PAIS_TOKEN`). Responses\nare parsed defensively: a bare JSON array or a `{data|items|models|knowledge_bases|…: [...]}` envelope is\naccepted, and every field degrades via `.get()`. **Beta caveat**: the exact `/api/v1` path prefix and\nthe JSON field names are `INFERRED_EXACT` (corroborated by the rendered Broadcom developer portal, not a\ndownloaded OpenAPI or a live deployment) — a 404 usually means a base-URL mismatch, not a bug.\n`pais_model_catalog` (`/api/v1/control/models`) is the least-confirmed of these (a best-guess path).\n\n## PAIS monitoring (1 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `pais_monitoring_summary` | R | vgpu_vms, reporting_vms, idle_vms, hot_vms, gpu_pct avg/max, mem_pct avg/max, temp_c_max, by_profile{}, busiest[], scope_note | 150–500 |\n\nA fleet rollup of the same VERIFIED `gpu.*` perf counters `gpu_utilization` reads — the numbers you\nwould pin to a VCF Ops dashboard. Uses the vCenter connection (not PAIS REST). Deep per-SM / MIG /\npower telemetry needs NVIDIA DCGM and is out of scope (`scope_note` says so).\n\n## Sizing & air-gap (2 read — no connection)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `pais_sizing_advise` | R | model_billions, precision, weights_gib, serving_vram_gib, gpu_options[], storage{}, assumptions{} | 200–400 |\n| `pais_bundle_verify` | R | manifest, image_count, images[], registries_to_mirror[], public_registries[], mutable_images[], warnings[] | 150–800 |\n\nBoth are pure/local — **no vCenter or PAIS call**. `pais_sizing_advise` is a transparent first-\nprinciples heuristic (weights = params × bytes/precision; GPU count = ceil(serving / usable HBM)) and is\nexplicit that LLM inference is compute/HBM-bound, so random IOPS is the wrong axis for the weights.\n`pais_bundle_verify` parses a **local** pais.yml with a real YAML parser (踩坑 #38) and flags the two\nair-gap blockers (public registries to mirror, mutable tags); it never contacts a registry.\n\n## vGPU assignment (1 write)\n| Tool | R/W | Risk | Blast radius |\n|------|:---:|:----:|--------------|\n| `vgpu_assign` | W | high | one VM, **only when powered off** — sets/replaces its vGPU profile via ReconfigVM |\n\n`confirm=false` (default) previews (current→target profile, power_state, requires_power_off, applied:false)\nwithout acting. `confirm=true` applies it but refuses a powered-on VM with a teaching error. It edits the\nVM's **existing vGPU device** (selected by the same predicate the preview uses, never a plain\nDirectPath/SR-IOV passthrough), waits for the real ReconfigVM task outcome (never a premature \"ok\"), and\naudits the result. It never powers the VM off itself — that is `vmware-aiops`.\n\n## Verified API surface (anti-phantom-endpoint gate)\n\nEvery runtime path is pinned in `tests/eval/spec/privateai_endpoints.py` and asserted by a regression\ngate (踩坑 #36):\n\n- **pyVmomi**: `HostSystem.config.graphicsInfo` (VERIFIED), `HostSystem.config.graphicsConfig`\n  (VERIFIED — vGPU mode via `hostDefaultGraphicsType`), `EnvironmentBrowser.QueryConfigTarget →\n  ConfigTarget.vgpuProfileInfo[]` (VERIFIED — the attribute is `vgpuProfileInfo`, not the spec-doc's\n  phantom `vgpu[]`), `content.directPathProfileManager.ListDirectPathProfiles` (VERIFIED, 9.0+),\n  `VirtualMachine.config.hardware.device → VirtualPCIPassthrough.backing.vgpu` (VERIFIED),\n  `VirtualMachine.runtime.powerState` (VERIFIED), `VirtualMachine.runtime.host` (VERIFIED),\n  `ReconfigVM_Task` (VERIFIED, requires VM off).\n- **Perf counters**: `gpu.utilization.average`, `gpu.mem.used.average`, `gpu.mem.usage.average`,\n  `gpu.temperature.average` (VERIFIED).\n- **PAIS REST**: `/api/v1/compatibility/openai/v1/models`, `/api/v1/control/knowledge-bases`,\n  `/api/v1/control/data-sources`, `/api/v1/control/models` (INFERRED_EXACT — path prefix deferred to\n  first live run; `/control/models` is a best-guess pending a live OpenAPI).\n- **NO_API** (code must not invent an endpoint): MIG mode set, GPU driver version, deep GPU telemetry\n  (DCGM), one-call DL-VM deploy, pgvector.\n\nFile v1.2.3:references/cli-reference.md\n\n# vmware-privateai — CLI Reference\n\nFull command list for the `vmware-privateai` Typer CLI. Every read command prints a teaching error and\nexits 1 (never a traceback) on a config / not-found / connection problem. Every command accepts\n`--target <name>` (a vCenter/ESXi target from `config.yaml`; omit for the default) and `--config <path>`\n(override the `~/.vmware-privateai/config.yaml` location). Reads paginate at 50 rows; use the filter\noptions rather than paging the whole estate.\n\n## Top level\n\n```bash\nvmware-privateai version          # print the installed version\nvmware-privateai mcp              # run the stdio MCP server (used by MCP clients)\nvmware-privateai --help           # list command groups: gpu, vgpu, pais\n```\n\n> The `mcp` subcommand is the recommended MCP entry point — it is an installed console script, so it\n> never re-resolves from PyPI the way `uvx` does (踩坑 #25: `uvx` is fragile behind an enterprise TLS\n> proxy). MCP clients should launch `vmware-privateai mcp`.\n\n## `gpu` — GPU inventory, utilization, and vGPU assignment\n\n```bash\nvmware-privateai gpu host-list [--name N] [--vendor V] [--target T] [--config PATH]\n```\nList ESXi hosts that have at least one GPU. Columns: host, gpu_count, vendors, vgpu_vms. `--name`\nsubstring-matches the host name; `--vendor` substring-matches the GPU vendor (e.g. `NVIDIA`).\n\n```bash\nvmware-privateai gpu host-get <host_name> [--target T] [--config PATH]\n```\nFull per-GPU detail for one host: device, graphics type, vendor, memory (MB), pci id, vm_count. A wrong\nhost name prints a teaching error listing the hosts that do have GPUs.\n\n```bash\nvmware-privateai gpu device-list [--host H] [--vendor V] [--target T] [--config PATH]\n```\nFlattened list of physical GPU devices across hosts. Columns: host, device, type, vendor, memory (MB),\nvm_count. `vm_count 0` marks an idle GPU.\n\n```bash\nvmware-privateai gpu consumer-list [--profile P] [--vm V] [--target T] [--config PATH]\n```\nList VMs consuming a vGPU and the profile each holds (e.g. `grid_a100-4c`). `--profile` / `--vm`\nsubstring-filter.\n\n```bash\nvmware-privateai gpu utilization [--vm V] [--top N] [--target T] [--config PATH]\n```\nReal-time (20s sample) GPU utilization per vGPU VM, busiest first: gpu %, mem %, temp (C). `--top N`\nkeeps only the N busiest. A VM with no host-driver samples prints `metrics unavailable` (not an error).\n\n```bash\nvmware-privateai gpu vgpu-assign <vm_name> <profile> [--dry-run] [--target T] [--config PATH]\n```\n**WRITE.** Set a VM's vGPU profile. Always prints the preview (current → target profile, power state,\nrequires_power_off) first. `--dry-run` stops there. Otherwise requires **double confirmation**, then\napplies via ReconfigVM and audits the result. The VM must be powered **off** — a running VM is refused\nwith a teaching error routing you to `vmware-aiops vm_power_off`. This command never powers the VM off\nitself.\n\n```bash\nvmware-privateai gpu readiness [--host H] [--target T] [--config PATH]\n```\nPer-host vGPU/PAIS readiness verdict. Columns: host, ready/not-ready, gpu_count, graphics mode,\nprofiles offered, plus `blocking_reasons` for any host that is not ready. Only hosts that have a GPU\nare shown. The NVIDIA driver / MFT VIB version and MIG are **not** in the vSphere API — verify with\n`nvidia-smi` / `esxcli software vib list` on the host.\n\n## `vgpu` — profile catalog\n\n```bash\nvmware-privateai vgpu profile-list [--host H] [--model M] [--target T] [--config PATH]\n```\nThe vGPU profile catalog aggregated across hosts. Columns: profile, framebuffer (GiB), class, sharing,\nhost_count. `--host` both filters and scopes the per-host `QueryConfigTarget` polling to one host;\n`--model` substring-matches the profile/model name. Hosts whose per-host query fails are reported in\n`unreachable_hosts` rather than sinking the whole catalog.\n\n```bash\nvmware-privateai vgpu directpath-list [--name N] [--vendor V] [--target T] [--config PATH]\n```\nvCenter-level DirectPath (dynamic passthrough) profiles. **vSphere 9.0+** — on an older vCenter this\nprints a teaching error routing you to `vgpu profile-list`. Columns: name, id, vendor, description.\n\n```bash\nvmware-privateai vgpu validate <vm_name> <target_profile> [--target T] [--config PATH]\n```\nRead-only pre-flight for a vGPU profile change (the pre-flight for `gpu vgpu-assign`). Checks the two\nReconfigVM failure modes up front — VM powered on, and whether the VM's own host offers the target\nprofile — and prints `can apply` / `blocked` with the reasons. No reconfigure.\n\n## `pais` — Private AI Service (REST), monitoring, sizing, and air-gap\n\nThese talk to the PAIS REST endpoint (`config.yaml` `pais:` section) with the bearer token in\n`VMWARE_PRIVATEAI_PAIS_TOKEN` — separate from the vCenter connection, so they take `--config` but not\n`--target`.\n\n```bash\nvmware-privateai pais model-list [--name N] [--config PATH]\n```\nList models served by PAIS (OpenAI-compatible `/models`). Columns: id, owned_by. `--name`\nsubstring-matches the model id.\n\n```bash\nvmware-privateai pais kb-list [--name N] [--config PATH]\n```\nList PAIS knowledge bases (RAG vector stores). Columns: name/id, status, description.\n\n```bash\nvmware-privateai pais model-catalog [--name N] [--config PATH]\nvmware-privateai pais data-source-list [--name N] [--config PATH]\n```\n`model-catalog` lists models available/approved to **deploy** (distinct from the served `/models`);\n`data-source-list` lists RAG ingest connectors. Both hit **INFERRED** PAIS control-plane paths — a 404\nprints a base-URL teaching message, not a bug.\n\n```bash\nvmware-privateai pais monitoring-summary [--hot-pct P] [--top N] [--target T] [--config PATH]\n```\nFleet GPU rollup (vgpu_vms, reporting/hot/idle, gpu%/mem% avg+max, busiest). Uses the **vCenter**\nconnection (takes `--target`), not PAIS REST. Deep per-SM/MIG telemetry needs NVIDIA DCGM (out of scope).\n\n```bash\nvmware-privateai pais sizing [--model llama-70b | --billions 70] [--precision fp16]\n```\nEstimate GPU memory / GPU count / storage to serve an LLM. **No connection** — a pure planning\nheuristic. Honest that random IOPS is the wrong axis for the model weights.\n\n```bash\nvmware-privateai pais bundle-verify <pais.yml>\n```\nParse a **local** pais.yml and list container images + registries to mirror for an air-gap, flagging\npublic registries and mutable tags. **No network** — does not contact any registry.\n\n## Exit codes\n\n| Code | Meaning |\n|------|---------|\n| 0 | Success |\n| 1 | Teaching error (config missing, target/host/VM not found, connection/TLS failure, PAIS 4xx) |\n\n## Environment variables\n\n| Variable | Purpose |\n|----------|---------|\n| `VMWARE_PRIVATEAI_<TARGET>_PASSWORD` | Password for a vCenter/ESXi target (`<TARGET>` = target name upper-cased, `-`→`_`) |\n| `VMWARE_PRIVATEAI_<TARGET>_USERNAME` | Optional username override (wins over `config.yaml`) |\n| `VMWARE_PRIVATEAI_PAIS_TOKEN` | OIDC/OAuth2 bearer token for the PAIS REST endpoint |\n| `VMWARE_PRIVATEAI_CONFIG` | Optional path to `config.yaml` (primary env for the skill) |\n\nSecret-bearing `*_PASSWORD` / `*_TOKEN` values are auto-obfuscated to `b64:` form in `.env` on load\n(grep-safe; obfuscation, not encryption).\n\nFile v1.2.3:references/setup-guide.md\n\n# vmware-privateai — Setup Guide\n\n> **Disclaimer**: Community-maintained open-source project, **not affiliated with, endorsed by, or\n> sponsored by VMware, Inc., Broadcom Inc., or NVIDIA Corporation.** \"VMware\", \"vSphere\", and \"VCF\" are\n> trademarks of Broadcom; \"NVIDIA\" and \"vGPU\" are trademarks of NVIDIA. Source is publicly auditable\n> under the MIT license.\n\nInstall, credential, and MCP-client configuration for vmware-privateai, plus the Security section.\n\n## 1. Install\n\n```bash\nuv tool install vmware-privateai==1.2.3       # isolated tool env; puts vmware-privateai on PATH\nvmware-privateai version\n```\n\nRequires Python 3.11+ (the MCP server is reflected by FastMCP/Pydantic; older interpreters can raise on\nPEP 604 unions — 踩坑 #33). Runtime deps: pyvmomi, httpx, typer, rich, pyyaml, python-dotenv, mcp, and\n`vmware-policy` (the family audit/policy harness, installed automatically).\n\n## 2. Configure targets\n\nCreate `~/.vmware-privateai/config.yaml`:\n\n```yaml\ntargets:\n  - name: vc-prod\n    host: vcenter-prod.example.com\n    username: administrator@vsphere.local   # optional; env var overrides this\n    type: vcenter                            # or esxi\n    port: 443\n    verify_ssl: true\n    environment: production                  # optional label for policy scoping\n\n# Optional — only needed for the pais model-list / kb-list tools:\npais:\n  endpoint: https://pais.example.com         # base URL; the /api/v1 prefix is added by the client\n  verify_ssl: true\n```\n\nThe first target is the default (used when `--target` / the `target` MCP arg is omitted).\n\n## 3. Credentials — never in config files\n\nPasswords and the PAIS bearer token live only in `~/.vmware-privateai/.env`:\n\n```bash\nmkdir -p ~/.vmware-privateai\ncat >> ~/.vmware-privateai/.env <<'EOF'\nVMWARE_PRIVATEAI_VC_PROD_PASSWORD=your-vcenter-password\nVMWARE_PRIVATEAI_PAIS_TOKEN=your-oidc-bearer-token\nEOF\nchmod 600 ~/.vmware-privateai/.env\n```\n\n- **Per-target password**: `VMWARE_PRIVATEAI_<TARGET>_PASSWORD`, where `<TARGET>` is the target `name`\n  upper-cased with `-` replaced by `_` (so target `vc-prod` → `VMWARE_PRIVATEAI_VC_PROD_PASSWORD`).\n- **Optional username override**: `VMWARE_PRIVATEAI_<TARGET>_USERNAME` wins over `config.yaml` (resolved\n  together with the password on every call, so a rotating sidecar never splits the pair).\n- **PAIS token**: `VMWARE_PRIVATEAI_PAIS_TOKEN` — a short-lived OIDC/OAuth2 bearer token from your\n  Identity Provider (the PAIS API uses `Authorization: Bearer <token>`). It is a secret and is\n  obfuscated to `b64:` at rest exactly like a password.\n- **Secret manager**: any of these vars can be injected from Vault / CyberArk / AWS Secrets Manager /\n  a Kubernetes Secret instead of `.env` — the code reads the environment either way.\n\nOn load, plaintext `*_PASSWORD` / `*_TOKEN` values in `.env` are auto-rewritten to grep-safe `b64:`\nform (obfuscation, not encryption — it defeats casual grep / shoulder-surfing, not a determined reader).\n\n## 4. MCP client configuration\n\nPreferred (installed console script — no `uvx` network re-resolve, works through enterprise TLS\nproxies, 踩坑 #25):\n\n```json\n{\n  \"mcpServers\": {\n    \"vmware-privateai\": {\n      \"command\": \"vmware-privateai\",\n      \"args\": [\"mcp\"],\n      \"env\": { \"VMWARE_PRIVATEAI_CONFIG\": \"~/.vmware-privateai/config.yaml\" }\n    }\n  }\n}\n```\n\nFallback (`uvx` — re-resolves from PyPI each start; if your network runs a TLS MitM proxy, add\n`\"UV_NATIVE_TLS\": \"true\"` to `env`):\n\n```json\n{\n  \"mcpServers\": {\n    \"vmware-privateai\": {\n      \"command\": \"uvx\",\n      \"args\": [\"--from\", \"vmware-privateai==1.2.3\", \"vmware-privateai-mcp\"],\n      \"env\": { \"VMWARE_PRIVATEAI_CONFIG\": \"~/.vmware-privateai/config.yaml\" }\n    }\n  }\n}\n```\n\nThis layout works with any MCP-compatible client (Claude Desktop, Claude Code, Goose, etc.).\n\n## 5. Verify\n\n```bash\nvmware-privateai gpu host-list             # lists hosts that have a GPU\nvmware-privateai vgpu profile-list         # the assignable vGPU profile catalog\nvmware-privateai pais model-list           # PAIS served models (if configured)\n```\n\nA teaching error here names exactly what to fix (missing password, unresolvable host, TLS, missing PAIS\nendpoint/token) — follow the message.\n\n## Security\n\n> **Disclaimer**: not affiliated with, endorsed by, or sponsored by VMware, Inc., Broadcom Inc., or\n> NVIDIA Corporation. See the full policy in the repo-root `SECURITY.md`.\n\n1. **Source Code** — https://github.com/vmware-skills/VMware-PrivateAI (MIT), publicly auditable.\n2. **Config File Contents** — `config.yaml` holds only target host/username/port and the\n   `pais.endpoint`. No passwords, no tokens. Secrets live in `~/.vmware-privateai/.env` (chmod 600,\n   `b64:` obfuscated at rest).\n3. **Webhook Data Scope** — none. No webhooks and no outbound network calls except to the configured\n   vCenter/ESXi targets and the PAIS endpoint.\n4. **TLS Verification** — on by default. `verify_ssl: false` (per target) and `pais.verify_ssl: false`\n   are intended only for self-signed lab certificates.\n5. **Prompt Injection Protection** — all vSphere-supplied and PAIS-supplied text (device / vendor / VM\n   / vGPU-profile names, PAIS model ids, and knowledge-base descriptions — the highest-value injection\n   surface here) passes through `vmware_policy.sanitize()` (≤500-char truncation + C0/C1 control-char\n   stripping) before it reaches the model.\n6. **Least Privilege** — read-vs-write authorization is delegated to the vCenter service account's RBAC\n   role: a read-only account refuses `vgpu_assign`'s ReconfigVM at vCenter, un-bypassably (the one place\n   the control cannot be stepped around by a shell). Recommend a dedicated service account scoped to the\n   GPU clusters. All writes are recorded to `~/.vmware/audit.db` (and the CLI companion\n   `~/.vmware-privateai/audit.log`).\n\n### Static analysis\n\n```bash\nuvx bandit -r vmware_privateai/            # target: 0 Medium+ issues\n```\n\nFile v1.2.3:skill-card.md\n\n## Description:\n\nProvides VMware Private AI Foundation with NVIDIA GPU and model-serving operations for vSphere 9.x and VCF 9.1, including GPU inventory, vGPU consumers, utilization, profile validation and assignment, and PAIS model and knowledge-base listings.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[zw008](https://clawhub.ai/user/zw008)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers, platform engineers, and infrastructure operators use this skill to inspect VMware GPU capacity, triage vGPU utilization, validate or apply powered-off vGPU profile changes, and query Private AI Service model-serving resources.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill installs external runtime code that handles infrastructure credentials.\n\nMitigation: Verify package provenance before production use, prefer the installed console script over uvx, and inject secrets through a secret manager or environment variables.\n\nRisk: The skill can perform a limited VM reconfiguration through vGPU assignment.\n\nMitigation: Use a dedicated least-privileged vCenter service account and grant write permissions only when vgpu_assign is needed.\n\nRisk: Stored credentials may be exposed if local files are mismanaged.\n\nMitigation: Prefer secret-manager or environment injection over storing credentials in ~/.vmware-privateai/.env.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/zw008/skills/vmware-privateai)\n- [Capabilities](artifact/references/capabilities.md)\n- [CLI Reference](artifact/references/cli-reference.md)\n- [Setup Guide](artifact/references/setup-guide.md)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown guidance with command examples and structured MCP or CLI results]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Operational outputs may include vSphere or PAIS resource names, status fields, utilization metrics, validation reasons, and audit-oriented write previews.]\n\n## Skill Version(s):\n\n1.2.3 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v1.2.2: 6 files, 17967 bytes\n\nFiles: references/capabilities.md (8615b), references/cli-reference.md (7168b), references/setup-guide.md (5951b), skill-card.md (2513b), SKILL.md (15156b), _meta.json (135b)\n\nFile v1.2.2:SKILL.md\n\n---\nname: vmware-privateai\ndescription: >\n  Use this skill whenever the user needs the GPU / AI-infrastructure layer of VMware Private AI\n  Foundation with NVIDIA (PAIF-N) on vSphere 9.x / VCF 9.1: inventory GPU hosts and physical GPU\n  devices, see which VMs consume a vGPU and the profile each holds, read real-time GPU utilization,\n  list the vGPU and DirectPath profile catalog, assign a VM's vGPU profile, and list Private AI\n  Service (PAIS) served models and knowledge bases. Always use this skill for \"list GPU hosts\",\n  \"which VMs are using a vGPU\", \"GPU utilization\", \"assign a vGPU profile\", \"list vGPU profiles\",\n  \"list served models\" when the context is explicitly VMware / vSphere / VCF Private AI / NVIDIA\n  vGPU. Do NOT use for the backing VM's power/snapshot/clone/migrate (use vmware-aiops), read-only\n  vSphere inventory/alarms/host health (use vmware-monitor), or GPU-enabled Tanzu Kubernetes\n  (use vmware-vks). This skill is the GPU lens; vmware-aiops owns the VM lifecycle behind it.\ninstaller:\n  kind: uv\n  package: vmware-privateai\nallowed-tools:\n  - Bash\nmetadata: {\"openclaw\":{\"requires\":{\"anyBins\":[\"vmware-privateai\",\"uvx\"]},\"optional\":{\"env\":[\"VMWARE_PRIVATEAI_CONFIG\"]}}}\n---\n\n# VMware Private AI (Foundation with NVIDIA) — GPU & Model-Serving Ops\n\n> **Disclaimer**: Community-maintained open-source project, **not affiliated with, endorsed by, or\n> sponsored by VMware, Inc., Broadcom Inc., or NVIDIA Corporation.** \"VMware\", \"vSphere\", and \"VCF\"\n> are trademarks of Broadcom; \"NVIDIA\" and \"vGPU\" are trademarks of NVIDIA. Source is publicly\n> auditable under the MIT license.\n\nThe GPU / AI-infrastructure lens for the VMware skill family — GPU host & device inventory, vGPU\nconsumers, real-time GPU utilization, the vGPU / DirectPath profile catalog, vGPU assignment, and\n**Private AI Service (PAIS)** served models and knowledge bases — over the **vSphere 9.x / VCF 9.1**\nWeb Services API (pyVmomi) plus the PAIS REST API.\n\n> **Companion skills**: [vmware-aiops](https://github.com/vmware-skills/VMware-AIops) (the vCenter VMs\n> behind AI workloads — power/snapshot/clone), [vmware-vks](https://github.com/vmware-skills/VMware-VKS)\n> (GPU-enabled Tanzu Kubernetes), [vmware-monitor](https://github.com/vmware-skills/VMware-Monitor)\n> (read-only vSphere health).\n\n> **Status: v1.0.1 — still beta in substance.** Skill #15 of the family. The jump from 0.2.x to\n> 1.0.1 is a distribution fix, not a maturity claim: the withdrawn first release used 1.0.0, and\n> ClawHub resolves `latest` by version order, so every 0.x release was invisible there. The beta\n> caveats below all still stand. Every API path is\n> verified against official Broadcom/NVIDIA sources before use (`tests/eval/spec/privateai_endpoints.py`)\n> — no endpoints written from memory. GET-response *field names* and the exact PAIS paths are\n> defensive and pending validation against live 9.x hardware (see Troubleshooting). Governed by the\n> family harness (audit + policy + teaching errors); read-vs-write authorization is delegated to the\n> vCenter service account's RBAC role.\n\n## What This Skill Does\n\n| Category | Tools | Count | Read/Write |\n|----------|-------|:-----:|:----------:|\n| **GPU inventory** | host list/get, device list, vGPU consumer list | 4 | 4 R |\n| **GPU utilization** | real-time per-vGPU-VM utilization (gpu %, mem %, temp) | 1 | 1 R |\n| **GPU readiness** | per-host vGPU/PAIS readiness verdict + blocking reasons | 1 | 1 R |\n| **Profile catalog** | vGPU profile list, DirectPath profile list | 2 | 2 R |\n| **Profile validation** | pre-flight a vGPU profile change (power state + host offers it) | 1 | 1 R |\n| **vGPU assignment** | set a VM's vGPU profile (VM must be powered off) | 1 | 1 W |\n| **Private AI Service** | served-model list, model catalog, knowledge-base list, data-source list | 4 | 4 R |\n| **PAIS monitoring** | fleet GPU rollup (util/mem/temp, hot/idle, busiest) | 1 | 1 R |\n| **Sizing & air-gap** | LLM GPU/storage sizing advisor, local pais.yml image inspector | 2 | 2 R |\n\n**17 MCP tools (16 read / 1 write).** Reads are strictly non-destructive. The single write\n(`vgpu_assign`) previews its blast radius, refuses a powered-on VM, never powers a VM off itself, is\ndouble-confirmed at the CLI, and is audit-logged. Pre-flight the write with `vgpu_profile_validate`.\n\n## Quick Install\n\n```bash\nuv tool install vmware-privateai==1.2.2\nvmware-privateai version\nvmware-privateai gpu host-list        # first read — lists hosts that have a GPU\n```\n\nConfig lives in `~/.vmware-privateai/config.yaml` (targets + optional `pais:` section); passwords and\nthe PAIS bearer token live in `~/.vmware-privateai/.env` (chmod 600). See `references/setup-guide.md`.\n\n## When to Use This Skill\n\nUse vmware-privateai for the **GPU / AI-infrastructure layer**: which hosts and physical devices have\nGPUs, which VMs hold a vGPU and what profile, real-time GPU utilization, the assignable vGPU /\nDirectPath profile catalog, changing a VM's vGPU profile, and the models / knowledge bases served by\nPrivate AI Service — when the context is explicitly VMware / vSphere / VCF Private AI / NVIDIA vGPU.\n\n**Do NOT use when**: the task is the backing VM's lifecycle — power on/off, snapshot, clone, migrate,\nreconfigure CPU/RAM (→ **vmware-aiops**); read-only vSphere inventory, alarms, or host health\n(→ **vmware-monitor**); or GPU-enabled Tanzu Kubernetes / Supervisor namespaces (→ **vmware-vks**).\n`vgpu_assign` deliberately does **not** power the VM off — that is vmware-aiops's job, kept separate\nso this skill's blast radius stays \"one VM, when it is already off\".\n\n## Related Skills — Skill Routing\n\n| The user wants… | Skill |\n|-----------------|-------|\n| Inventory GPUs / vGPU consumers / GPU utilization / assign a vGPU profile | **vmware-privateai** (this) |\n| List PAIS served models / knowledge bases | **vmware-privateai** (this) |\n| Power off / snapshot / clone / migrate the backing vCenter VM | vmware-aiops |\n| Read-only vSphere inventory / alarms / host health | vmware-monitor |\n| GPU-enabled Tanzu Kubernetes clusters / namespaces | vmware-vks |\n| Multi-step GPU workflow with approval + rollback | vmware-pilot |\n\n## Common Workflows\n\n**1. Find an idle GPU and reassign a VM's vGPU profile.**\n```\nvmware-privateai gpu device-list --vendor NVIDIA     # find GPUs; vm_count 0 = idle\nvmware-privateai gpu consumer-list                   # who holds a vGPU, and which profile\nvmware-privateai vgpu profile-list --host esx-07     # profiles that host can hand a VM\nvmware-privateai gpu vgpu-assign fin-train-01 grid_a100-4c --dry-run   # preview blast radius\n# power the VM off with vmware-aiops, THEN:\nvmware-privateai gpu vgpu-assign fin-train-01 grid_a100-4c             # double-confirm + audit\n```\n*Failure branch*: if `vgpu-assign` (confirm) refuses with \"VM is powered on — a vGPU change needs the\nVM powered off\", run `vmware-aiops vm_power_off 'fin-train-01'` first, then re-run. If it fails with\n\"profile not offered by the VM's host / GPU lacks free framebuffer\", run\n`vmware-privateai gpu host-get <that VM's host>` to see the valid profiles and free capacity.\n\n**2. Triage GPU utilization across the estate.**\n```\nvmware-privateai gpu utilization --top 10            # busiest vGPU VMs first\nvmware-privateai gpu host-list --vendor NVIDIA       # which hosts carry the load\n```\n*Failure branch*: a VM showing `metrics unavailable (no host driver?)` is not an error — the NVIDIA\nhost GPU driver is not exposing counters for it (`metrics_available:false`). Deep per-SM / per-process\n/ MIG-slice telemetry is **not** available via vSphere; use NVIDIA DCGM on the host for that.\n\n**3. See what Private AI Service is serving.**\n```\nvmware-privateai pais model-list                     # OpenAI-compatible /models\nvmware-privateai pais kb-list                         # RAG knowledge bases\n```\n*Failure branch*: HTTP 404 usually means a base-URL mismatch, not a bug — the `/api/v1` PAIS path\nprefix is deployment-specific and unconfirmed (beta). Check `pais.endpoint` in config.yaml. HTTP\n401/403 means the bearer token in `VMWARE_PRIVATEAI_PAIS_TOKEN` is expired or lacks scope — obtain a\nfresh token from your Identity Provider, re-export it, and retry.\n\n## Usage Mode\n\n- **CLI** — interactive inventory / triage, scripting, small or local models (lower context cost).\n- **MCP** — agent-driven operations with structured JSON; run `vmware-privateai mcp` (an installed\n  console script, so no `uvx` network re-resolve — works through enterprise TLS proxies, 踩坑 #25).\n\n## MCP Tools (17 — 16 read, 1 write)\n\n| Category | Tools | R/W |\n|----------|-------|:---:|\n| GPU inventory | `gpu_host_list`, `gpu_host_get`, `gpu_device_list`, `gpu_consumer_list` | Read |\n| GPU utilization | `gpu_utilization` | Read |\n| GPU readiness | `gpu_host_readiness` | Read |\n| Profile catalog | `vgpu_profile_list`, `directpath_profile_list` | Read |\n| Profile validation | `vgpu_profile_validate` | Read |\n| Private AI Service | `pais_model_list`, `pais_model_catalog`, `pais_knowledge_base_list`, `pais_data_source_list` | Read |\n| PAIS monitoring | `pais_monitoring_summary` | Read |\n| Sizing & air-gap | `pais_sizing_advise`, `pais_bundle_verify` | Read |\n| vGPU assignment | `vgpu_assign` | Write |\n\n**INFERRED PAIS paths**: `pais_model_catalog` and `pais_data_source_list` hit PAIS control-plane\npaths that are unconfirmed against a live OpenAPI (踩坑 #36) — a 404 returns a base-URL teaching\nmessage, not a bug. `pais_sizing_advise` and `pais_bundle_verify` need **no connection** (pure\ncomputation / local file parse). `gpu_host_readiness` reports what the vSphere API exposes and says\nso where it cannot (NVIDIA driver / MFT VIB / MIG need nvidia-smi on the host).\n\n**List envelope**: every `*_list` tool returns `{items, returned, limit, offset, total, truncated, hint}`\n— read rows from `items` and check `truncated` before concluding a listing is complete; empty `items`\nwith `truncated:false` means checked-and-none, not a failure. Lists paginate at `limit=50`; filter with\nthe tool's `name`/`vendor`/`host`/`profile`/`vm` arguments rather than paging the whole estate.\n\n**Write safety (normative)**: `vgpu_assign` with `confirm=false` (the default) previews only —\ncurrent profile, target profile, power state, and that a power-off is required — without acting.\n`confirm=true` applies it, but refuses a powered-on VM with a teaching error. It **never powers the VM\noff itself**, waits for the real ReconfigVM task outcome (never a premature \"ok\"), and audits every\napplied change to `~/.vmware/audit.db`.\n\n## CLI Quick Reference\n\n```bash\nvmware-privateai gpu host-list [--name N] [--vendor V]       # hosts with a GPU\nvmware-privateai gpu host-get <host>                          # full per-GPU detail\nvmware-privateai gpu device-list [--host H] [--vendor V]     # physical GPUs (vm_count 0 = idle)\nvmware-privateai gpu consumer-list [--profile P] [--vm V]    # VMs holding a vGPU + profile\nvmware-privateai gpu utilization [--vm V] [--top N]          # real-time GPU %, mem %, temp\nvmware-privateai gpu vgpu-assign <vm> <profile> [--dry-run]  # WRITE — VM must be off; double-confirm\nvmware-privateai gpu readiness [--host H]                   # per-host vGPU/PAIS readiness verdict\nvmware-privateai vgpu profile-list [--host H] [--model M]    # vGPU profile catalog\nvmware-privateai vgpu directpath-list [--vendor V]           # DirectPath profiles (vSphere 9.0+)\nvmware-privateai vgpu validate <vm> <profile>               # pre-flight a vGPU profile change (read-only)\nvmware-privateai pais model-list [--name N]                  # PAIS served models\nvmware-privateai pais model-catalog [--name N]               # PAIS deployable/approved model catalog\nvmware-privateai pais kb-list [--name N]                     # PAIS knowledge bases\nvmware-privateai pais data-source-list [--name N]            # PAIS RAG data sources\nvmware-privateai pais monitoring-summary [--top N]           # fleet GPU rollup (util/mem/temp, hot/idle)\nvmware-privateai pais sizing --model llama-70b               # LLM GPU/storage sizing (no connection)\nvmware-privateai pais bundle-verify <pais.yml>              # local air-gap image inspector (no network)\n```\nFull list: `references/cli-reference.md`. Per-tool response-token estimates: `references/capabilities.md`.\n\n## Troubleshooting\n\n- **`Password not found for target '<t>'. Set environment variable VMWARE_PRIVATEAI_<T>_PASSWORD`** —\n  add that line to `~/.vmware-privateai/.env` and `chmod 600` it, or export it (from a secret manager).\n  The `<T>` is the target name upper-cased with `-`→`_`.\n- **`TLS verification failed for target '<t>'`** — for a self-signed lab set `verify_ssl: false` for\n  that target in `config.yaml`; otherwise install the vCenter CA on this host.\n- **`gpu host-list` returns nothing on a cluster you know has GPUs** — only `shared` / `direct` /\n  `sharedDirect` graphics types count as compute GPUs (the plain host framebuffer is excluded). If real\n  9.x hardware surfaces a GPU under an unexpected type, that is a beta known-limitation — file an issue\n  with the raw `gpu host-get` output so the projection can be widened.\n- **`gpu utilization` shows a VM with `metrics unavailable`** — the NVIDIA host GPU driver is not\n  exposing counters for it (not an error). Note the `gpu.*` perf counters may report at host level on\n  some builds — verify the entity type on real hardware (beta caveat).\n- **`directpath-list` errors with \"needs vCenter 9.0+\"** — DirectPathProfileManager is new in vSphere\n  9.0; on 8.x use `vgpu profile-list` instead (the error routes you there, not an empty list).\n- **PAIS 404 / non-JSON response** — the `/api/v1` prefix is deployment-specific and unconfirmed;\n  check `pais.endpoint` (a proxy or login page returns non-JSON). PAIS 401/403 → refresh the bearer\n  token in `VMWARE_PRIVATEAI_PAIS_TOKEN`.\n\n## Audit & Safety\n\n1. **Source Code** — https://github.com/vmware-skills/VMware-PrivateAI (MIT).\n2. **Config File Contents** — `config.yaml` holds target host/username/port and the `pais.endpoint`\n   only; passwords and the PAIS bearer token live in `~/.vmware-privateai/.env` (0600, obfuscated to\n   `b64:` at rest — obfuscation, not encryption).\n3. **Webhook Data Scope** — none. This skill makes no outbound calls except to the configured\n   vCenter/ESXi targets and PAIS endpoint.\n4. **TLS Verification** — on by default; `verify_ssl: false` is per-target (and `pais.verify_ssl`) and\n   only for self-signed labs.\n5. **Prompt Injection Protection** — all vSphere-supplied and PAIS-supplied text (device/vendor/VM/\n   profile names, PAIS model ids, knowledge-base descriptions) passes through `vmware_policy.sanitize()`\n   (truncation ≤500 chars + C0/C1 control-char stripping); a KB description is the highest-value\n   injection surface here.\n6. **Least Privilege** — read-vs-write authorization is the vCenter role's job: a read-only service\n   account refuses `vgpu_assign`'s ReconfigVM at vCenter, un-bypassably. All writes are recorded in\n   `~/.vmware/audit.db`. See `references/setup-guide.md`.\n\n## License\n\nMIT\n\nFile v1.2.2:_meta.json\n\n{\n  \"ownerId\": \"kn7b067awq2s97bn3d7p5qfhw5827pxc\",\n  \"slug\": \"vmware-privateai\",\n  \"version\": \"1.2.2\",\n  \"publishedAt\": 1789483132047\n}\n\nFile v1.2.2:references/capabilities.md\n\n# vmware-privateai — Capabilities\n\n17 MCP tools (16 read / 1 write) over the vSphere 9.x / VCF 9.1 Web Services API (pyVmomi) plus the\nPrivate AI Service (PAIS) REST API, with two tools that need no connection at all (sizing / bundle).\nEvery vSphere tool accepts an optional `target`; PAIS tools use the `pais:` config section instead.\nTypical response tokens are estimates for a small estate; every `*_list` tool paginates at `limit=50`\nand returns the `{items, returned, limit, offset, total, truncated, hint}` envelope.\n\n## GPU inventory (4 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `gpu_host_list` | R | host, gpu_count, vendors[], vgpu_vms (filter name/vendor) | 60–400 |\n| `gpu_host_get` | R | one host's GPUs: device, type, vendor, memory_mb, pci_id, vm_count | 80–400 |\n| `gpu_device_list` | R | flattened physical GPUs: host, device, type, vendor, pci_id, memory_mb, vm_count (filter host/vendor) | 80–600 |\n| `gpu_consumer_list` | R | vm, profile — the \"who holds a vGPU\" view (filter profile/vm) | 60–500 |\n\n## GPU utilization (1 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `gpu_utilization` | R | vm, profile, gpu_pct, mem_pct, mem_used_kb, temp_c, metrics_available, idle; busiest first, `top` keeps N | 80–500 |\n\nReal-time 20s samples via the vSphere PerformanceManager `gpu.*` counters (require the NVIDIA host GPU\ndriver). A negative sample is vSphere's \"no data\" sentinel and is dropped; a VM with no samples reports\n`metrics_available:false`. **Beta caveat**: the `gpu.*` counters may report at host level on some\nbuilds — verify the entity type on real hardware. Deep per-SM / per-process / MIG-slice telemetry is\n**not** in vSphere (needs NVIDIA DCGM) and no endpoint is invented for it.\n\n## GPU readiness (1 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `gpu_host_readiness` | R | host, vgpu_ready, gpu_count, vendors[], total_gpu_memory_mb, default_graphics_type, vgpu_profiles_offered, active_vgpu_vms, blocking_reasons[], driver_note (filter/scope host) | 100–600 |\n\nCombines `config.graphicsInfo` + `config.graphicsConfig` + the per-host `QueryConfigTarget` profile\ncatalog into a `vgpu_ready` verdict (GPU present + `sharedDirect` mode + ≥1 profile offered). Only\nGPU hosts are returned; a per-host query failure lands in `unreachable_hosts`. The NVIDIA driver /\nMFT VIB version and MIG geometry are **not** in the vSphere API — every item carries a `driver_note`\nrouting to `nvidia-smi` / `esxcli` (spec NO_API; no endpoint invented).\n\n## Profile catalog (2 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `vgpu_profile_list` | R | profile, name, framebuffer_gib, profile_class, sharing, vendor_id, hosts[], host_count, unreachable_hosts[] (filter/scope host, filter model) | 80–600 |\n| `directpath_profile_list` | R | id, name, vendor, description — vCenter-level DirectPath profiles, **vSphere 9.0+** (filter name/vendor) | 60–400 |\n\n`vgpu_profile_list` polls `EnvironmentBrowser.QueryConfigTarget` **per host** — scope with `host` to\navoid polling the whole estate. A host whose per-host query fails lands in `unreachable_hosts` (a\nreachability problem, not \"no profiles\"). `directpath_profile_list` on a pre-9.0 vCenter raises a\nteaching error routing to `vgpu_profile_list` rather than returning an empty list.\n\n## Profile validation (1 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `vgpu_profile_validate` | R | vm, host, current_profile, target_profile, power_state, host_offers_target, target_framebuffer_gib, can_apply, blocking_reasons[] | 80–300 |\n\nRead-only pre-flight for `vgpu_assign`: checks the two ReconfigVM failure modes up front — VM powered\non, or the VM's own host does not offer the target profile. No reconfigure; run `vgpu_assign` once\n`can_apply` is true.\n\n## Private AI Service (4 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `pais_model_list` | R | id, owned_by, created — OpenAI-compatible served `/models` (filter name) | 60–400 |\n| `pais_model_catalog` | R | id, name, status, source — deployable/approved model catalog (filter name) | 60–500 |\n| `pais_knowledge_base_list` | R | id, name, status, description — RAG knowledge bases (filter name) | 80–500 |\n| `pais_data_source_list` | R | id, name, type, status — RAG ingest connectors (filter name) | 60–400 |\n\nPAIS reads go through the bearer-authenticated REST client (`VMWARE_PRIVATEAI_PAIS_TOKEN`). Responses\nare parsed defensively: a bare JSON array or a `{data|items|models|knowledge_bases|…: [...]}` envelope is\naccepted, and every field degrades via `.get()`. **Beta caveat**: the exact `/api/v1` path prefix and\nthe JSON field names are `INFERRED_EXACT` (corroborated by the rendered Broadcom developer portal, not a\ndownloaded OpenAPI or a live deployment) — a 404 usually means a base-URL mismatch, not a bug.\n`pais_model_catalog` (`/api/v1/control/models`) is the least-confirmed of these (a best-guess path).\n\n## PAIS monitoring (1 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `pais_monitoring_summary` | R | vgpu_vms, reporting_vms, idle_vms, hot_vms, gpu_pct avg/max, mem_pct\n\nArchive v1.2.1: 6 files, 17922 bytes\n\nFiles: references/capabilities.md (8615b), references/cli-reference.md (7168b), references/setup-guide.md (5951b), skill-card.md (2336b), SKILL.md (15156b), _meta.json (135b)\n\nArchive v1.2.0: 6 files, 17925 bytes\n\nFiles: references/capabilities.md (8615b), references/cli-reference.md (7168b), references/setup-guide.md (5951b), skill-card.md (2444b), SKILL.md (15156b), _meta.json (135b)\n\nArchive v1.1.1: 6 files, 17918 bytes\n\nFiles: references/capabilities.md (8615b), references/cli-reference.md (7168b), references/setup-guide.md (5937b), skill-card.md (2400b), SKILL.md (15211b), _meta.json (135b)\n\nArchive v1.1.0: 6 files, 17959 bytes\n\nFiles: references/capabilities.md (8615b), references/cli-reference.md (7168b), references/setup-guide.md (5937b), skill-card.md (2520b), SKILL.md (15211b), _meta.json (135b)\n\nArchive v1.0.4: 6 files, 17938 bytes\n\nFiles: references/capabilities.md (8615b), references/cli-reference.md (7168b), references/setup-guide.md (5937b), skill-card.md (2442b), SKILL.md (15211b), _meta.json (135b)\n\nArchive v1.0.3: 6 files, 18180 bytes\n\nFiles: references/capabilities.md (8615b), references/cli-reference.md (7168b), references/setup-guide.md (5937b), skill-card.md (2908b), SKILL.md (15211b), _meta.json (135b)","readmeExcerpt":"Skill: vmware-privateai Owner: zw008 Summary: Use this skill whenever the user needs the GPU / AI-infrastructure layer of VMware Private AI Foundation with NVIDIA (PAIF-N) on vSphere 9.x / VCF 9.1: inventory GPU hosts and physical GPU devices, see which VMs consume a vGPU and the profile each holds, read real-time GPU utilization, list the vGPU and DirectPath profile catalog, assign a VM's vGPU profile, and list Priv","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"uv tool install vmware-privateai==1.4.0\nvmware-privateai version\nvmware-privateai gpu host-list        # first read — lists hosts that have a GPU"},{"language":"text","snippet":"vmware-privateai gpu device-list --vendor NVIDIA     # find GPUs; vm_count 0 = idle\nvmware-privateai gpu consumer-list                   # who holds a vGPU, and which profile\nvmware-privateai vgpu profile-list --host esx-07     # profiles that host can hand a VM\nvmware-privateai gpu vgpu-assign fin-train-01 grid_a100-4c --dry-run   # preview blast radius\n# power the VM off with vmware-aiops, THEN:\nvmware-privateai gpu vgpu-assign fin-train-01 grid_a100-4c             # double-confirm + audit"},{"language":"text","snippet":"vmware-privateai gpu utilization --top 10            # busiest vGPU VMs first\nvmware-privateai gpu host-list --vendor NVIDIA       # which hosts carry the load"},{"language":"text","snippet":"vmware-privateai pais model-list                     # OpenAI-compatible /models\nvmware-privateai pais kb-list                         # RAG knowledge bases"},{"language":"bash","snippet":"vmware-privateai gpu host-list [--name N] [--vendor V]       # hosts with a GPU\nvmware-privateai gpu host-get <host>                          # full per-GPU detail\nvmware-privateai gpu device-list [--host H] [--vendor V]     # physical GPUs (vm_count 0 = idle)\nvmware-privateai gpu consumer-list [--profile P] [--vm V]    # VMs holding a vGPU + profile\nvmware-privateai gpu utilization [--vm V] [--top N]          # real-time GPU %, mem %, temp\nvmware-privateai gpu vgpu-assign <vm> <profile> [--dry-run]  # WRITE — VM must be off; double-confirm\nvmware-privateai gpu readiness [--host H]                   # per-host vGPU/PAIS readiness verdict\nvmware-privateai vgpu profile-list [--host H] [--model M]    # vGPU profile catalog\nvmware-privateai vgpu directpath-list [--vendor V]           # DirectPath profiles (vSphere 9.0+)\nvmware-privateai vgpu validate <vm> <profile>               # pre-flight a vGPU profile change (read-only)\nvmware-privateai pais model-list [--name N]                  # PAIS served models\nvmware-privateai pais model-catalog [--name N]               # PAIS deployable/approved model catalog\nvmware-privateai pais kb-list [--name N]                     # PAIS knowledge bases\nvmware-privateai pais data-source-list [--name N]            # PAIS RAG data sources\nvmware-privateai pais monitoring-summary [--top N]           # fleet GPU rollup (util/mem/temp, hot/idle)\nvmware-privateai pais sizing --model llama-70b               # LLM GPU/storage sizing (no connection)\nvmware-privateai pais bundle-verify <pais.yml>              # local air-gap image inspector (no network)"},{"language":"bash","snippet":"vmware-privateai version          # print the installed version\nvmware-privateai mcp              # run the stdio MCP server (used by MCP clients)\nvmware-privateai --help           # list command groups: gpu, vgpu, pais"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: vmware-privateai\ndescription: >\n  Use this skill whenever the user needs the GPU / AI-infrastructure layer of VMware Private AI\n  Foundation with NVIDIA (PAIF-N) on vSphere 9.x / VCF 9.1: inventory GPU hosts and physical GPU\n  devices, see which VMs consume a vGPU and the profile each holds, read real-time GPU utilization,\n  list the vGPU and DirectPath profile catalog, assign a VM's vGPU profile, and list Private AI\n  Service (PAIS) served models and knowledge bases. Always use this skill for \"list GPU hosts\",\n  \"which VMs are using a vGPU\", \"GPU utilization\", \"assign a vGPU profile\", \"list vGPU profiles\",\n  \"list served models\" when the context is explicitly VMware / vSphere / VCF Private AI / NVIDIA\n  vGPU. Do NOT use for the backing VM's power/snapshot/clone/migrate (use vmware-aiops), read-only\n  vSphere inventory/alarms/host health (use vmware-monitor), or GPU-enabled Tanzu Kubernetes\n  (use vmware-vks). This skill is the GPU lens; vmware-aiops owns the VM lifecycle behind it.\ninstaller:\n  kind: uv\n  package: vmware-privateai\nallowed-tools:\n  - Bash\nmetadata: {\"openclaw\":{\"requires\":{\"anyBins\":[\"vmware-privateai\",\"uvx\"]},\"optional\":{\"env\":[\"VMWARE_PRIVATEAI_CONFIG\"]}}}\n---\n\n# VMware Private AI (Foundation with NVIDIA) — GPU & Model-Serving Ops\n\n> **Disclaimer**: Community-maintained open-source project, **not affiliated with, endorsed by, or\n> sponsored by VMware, Inc., Broadcom Inc., or NVIDIA Corporation.** \"VMware\", \"vSphere\", and \"VCF\"\n> are trademarks of Broadcom; \"NVIDIA\" and \"vGPU\" are trademarks of NVIDIA. Source is publicly\n> auditable under the MIT license.\n\nThe GPU / AI-infrastructure lens for the VMware skill family — GPU host & device inventory, vGPU\nconsumers, real-time GPU utilization, the vGPU / DirectPath profile catalog, vGPU assignment, and\n**Private AI Service (PAIS)** served models and knowledge bases — over the **vSphere 9.x / VCF 9.1**\nWeb Services API (pyVmomi) plus the PAIS REST API.\n\n> **Companion skills**: [vmware-aiops](https://github.com/vmware-skills/VMware-AIops) (the vCenter VMs\n> behind AI workloads — power/snapshot/clone), [vmware-vks](https://github.com/vmware-skills/VMware-VKS)\n> (GPU-enabled Tanzu Kubernetes), [vmware-monitor](https://github.com/vmware-skills/VMware-Monitor)\n> (read-only vSphere health).\n\n> **Status: v1.0.1 — still beta in substance.** Skill #15 of the family. The jump from 0.2.x to\n> 1.0.1 is a distribution fix, not a maturity claim: the withdrawn first release used 1.0.0, and\n> ClawHub resolves `latest` by version order, so every 0.x release was invisible there. The beta\n> caveats below all still stand. Every API path is\n> verified against official Broadcom/NVIDIA sources before use (`tests/eval/spec/privateai_endpoints.py`)\n> — no endpoints written from memory. GET-response *field names* and the exact PAIS paths are\n> defensive and pending validation against live 9.x hardware (see Troubleshooting). Governed by the\n> family harness (audit + policy + teaching errors); read-vs-"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7b067awq2s97bn3d7p5qfhw5827pxc\",\n  \"slug\": \"vmware-privateai\",\n  \"version\": \"1.4.0\",\n  \"publishedAt\": 1789915986383\n}"},{"path":"references/capabilities.md","content":"# vmware-privateai — Capabilities\n\n17 MCP tools (16 read / 1 write) over the vSphere 9.x / VCF 9.1 Web Services API (pyVmomi) plus the\nPrivate AI Service (PAIS) REST API, with two tools that need no connection at all (sizing / bundle).\nEvery vSphere tool accepts an optional `target`; PAIS tools use the `pais:` config section instead.\nTypical response tokens are estimates for a small estate; every `*_list` tool paginates at `limit=50`\nand returns the `{items, returned, limit, offset, total, truncated, hint}` envelope.\n\n## GPU inventory (4 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `gpu_host_list` | R | host, gpu_count, vendors[], vgpu_vms (filter name/vendor) | 60–400 |\n| `gpu_host_get` | R | one host's GPUs: device, type, vendor, memory_mb, pci_id, vm_count | 80–400 |\n| `gpu_device_list` | R | flattened physical GPUs: host, device, type, vendor, pci_id, memory_mb, vm_count (filter host/vendor) | 80–600 |\n| `gpu_consumer_list` | R | vm, profile — the \"who holds a vGPU\" view (filter profile/vm) | 60–500 |\n\n## GPU utilization (1 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `gpu_utilization` | R | vm, profile, gpu_pct, mem_pct, mem_used_kb, temp_c, metrics_available, idle; busiest first, `top` keeps N | 80–500 |\n\nReal-time 20s samples via the vSphere PerformanceManager `gpu.*` counters (require the NVIDIA host GPU\ndriver). A negative sample is vSphere's \"no data\" sentinel and is dropped; a VM with no samples reports\n`metrics_available:false`. **Beta caveat**: the `gpu.*` counters may report at host level on some\nbuilds — verify the entity type on real hardware. Deep per-SM / per-process / MIG-slice telemetry is\n**not** in vSphere (needs NVIDIA DCGM) and no endpoint is invented for it.\n\n## GPU readiness (1 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `gpu_host_readiness` | R | host, vgpu_ready, gpu_count, vendors[], total_gpu_memory_mb, default_graphics_type, vgpu_profiles_offered, active_vgpu_vms, blocking_reasons[], driver_note (filter/scope host) | 100–600 |\n\nCombines `config.graphicsInfo` + `config.graphicsConfig` + the per-host `QueryConfigTarget` profile\ncatalog into a `vgpu_ready` verdict (GPU present + `sharedDirect` mode + ≥1 profile offered). Only\nGPU hosts are returned; a per-host query failure lands in `unreachable_hosts`. The NVIDIA driver /\nMFT VIB version and MIG geometry are **not** in the vSphere API — every item carries a `driver_note`\nrouting to `nvidia-smi` / `esxcli` (spec NO_API; no endpoint invented).\n\n## Profile catalog (2 read)\n| Tool | R/W | Returns | ~tokens |\n|------|:---:|---------|:------:|\n| `vgpu_profile_list` | R | profile, name, framebuffer_gib, profile_class, sharing, vendor_id, hosts[], host_count, unreachable_hosts[] (filter/scope host, filter model) | 80–600 |\n| `directpath_profile_list` | R | id, name, vendor, description — vCenter-level DirectPath profiles, **vSphere 9.0+** (filter name/vendor) | 60–400 |\n\n`vgpu_pro"},{"path":"references/cli-reference.md","content":"# vmware-privateai — CLI Reference\n\nFull command list for the `vmware-privateai` Typer CLI. Every read command prints a teaching error and\nexits 1 (never a traceback) on a config / not-found / connection problem. Every command accepts\n`--target <name>` (a vCenter/ESXi target from `config.yaml`; omit for the default) and `--config <path>`\n(override the `~/.vmware-privateai/config.yaml` location). Reads paginate at 50 rows; use the filter\noptions rather than paging the whole estate.\n\n## Top level\n\n```bash\nvmware-privateai version          # print the installed version\nvmware-privateai mcp              # run the stdio MCP server (used by MCP clients)\nvmware-privateai --help           # list command groups: gpu, vgpu, pais\n```\n\n> The `mcp` subcommand is the recommended MCP entry point — it is an installed console script, so it\n> never re-resolves from PyPI the way `uvx` does (踩坑 #25: `uvx` is fragile behind an enterprise TLS\n> proxy). MCP clients should launch `vmware-privateai mcp`.\n\n## `gpu` — GPU inventory, utilization, and vGPU assignment\n\n```bash\nvmware-privateai gpu host-list [--name N] [--vendor V] [--target T] [--config PATH]\n```\nList ESXi hosts that have at least one GPU. Columns: host, gpu_count, vendors, vgpu_vms. `--name`\nsubstring-matches the host name; `--vendor` substring-matches the GPU vendor (e.g. `NVIDIA`).\n\n```bash\nvmware-privateai gpu host-get <host_name> [--target T] [--config PATH]\n```\nFull per-GPU detail for one host: device, graphics type, vendor, memory (MB), pci id, vm_count. A wrong\nhost name prints a teaching error listing the hosts that do have GPUs.\n\n```bash\nvmware-privateai gpu device-list [--host H] [--vendor V] [--target T] [--config PATH]\n```\nFlattened list of physical GPU devices across hosts. Columns: host, device, type, vendor, memory (MB),\nvm_count. `vm_count 0` marks an idle GPU.\n\n```bash\nvmware-privateai gpu consumer-list [--profile P] [--vm V] [--target T] [--config PATH]\n```\nList VMs consuming a vGPU and the profile each holds (e.g. `grid_a100-4c`). `--profile` / `--vm`\nsubstring-filter.\n\n```bash\nvmware-privateai gpu utilization [--vm V] [--top N] [--target T] [--config PATH]\n```\nReal-time (20s sample) GPU utilization per vGPU VM, busiest first: gpu %, mem %, temp (C). `--top N`\nkeeps only the N busiest. A VM with no host-driver samples prints `metrics unavailable` (not an error).\n\n```bash\nvmware-privateai gpu vgpu-assign <vm_name> <profile> [--dry-run] [--target T] [--config PATH]\n```\n**WRITE.** Set a VM's vGPU profile. Always prints the preview (current → target profile, power state,\nrequires_power_off) first. `--dry-run` stops there. Otherwise requires **double confirmation**, then\napplies via ReconfigVM and audits the result. The VM must be powered **off** — a running or suspended VM\nis refused with a teaching error routing you to `vmware-aiops vm_power_off`, and so is a name shared by\nseveral VMs or a VM whose power state or devices could not be read. This command never powers the VM off\nitself.\n\n```bas"},{"path":"references/setup-guide.md","content":"# vmware-privateai — Setup Guide\n\n> **Disclaimer**: Community-maintained open-source project, **not affiliated with, endorsed by, or\n> sponsored by VMware, Inc., Broadcom Inc., or NVIDIA Corporation.** \"VMware\", \"vSphere\", and \"VCF\" are\n> trademarks of Broadcom; \"NVIDIA\" and \"vGPU\" are trademarks of NVIDIA. Source is publicly auditable\n> under the MIT license.\n\nInstall, credential, and MCP-client configuration for vmware-privateai, plus the Security section.\n\n## 1. Install\n\n```bash\nuv tool install vmware-privateai==1.4.0       # isolated tool env; puts vmware-privateai on PATH\nvmware-privateai version\n```\n\nRequires Python 3.11+ (the MCP server is reflected by FastMCP/Pydantic; older interpreters can raise on\nPEP 604 unions — 踩坑 #33). Runtime deps: pyvmomi, httpx, typer, rich, pyyaml, python-dotenv, mcp, and\n`vmware-policy` (the family audit/policy harness, installed automatically).\n\n## 2. Configure targets\n\nCreate `~/.vmware-privateai/config.yaml`:\n\n```yaml\ntargets:\n  - name: vc-prod\n    host: vcenter-prod.example.com\n    username: administrator@vsphere.local   # optional; env var overrides this\n    type: vcenter                            # or esxi\n    port: 443\n    verify_ssl: true\n    environment: production                  # optional label for policy scoping\n\n# Optional — only needed for the pais model-list / kb-list tools:\npais:\n  endpoint: https://pais.example.com         # base URL; the /api/v1 prefix is added by the client\n  verify_ssl: true\n```\n\nThe first target is the default (used when `--target` / the `target` MCP arg is omitted).\n\n## 3. Credentials — never in config files\n\nPasswords and the PAIS bearer token live only in `~/.vmware-privateai/.env`:\n\n```bash\nmkdir -p ~/.vmware-privateai\ncat >> ~/.vmware-privateai/.env <<'EOF'\nVMWARE_PRIVATEAI_VC_PROD_PASSWORD=your-vcenter-password\nVMWARE_PRIVATEAI_PAIS_TOKEN=your-oidc-bearer-token\nEOF\nchmod 600 ~/.vmware-privateai/.env\n```\n\n- **Per-target password**: `VMWARE_PRIVATEAI_<TARGET>_PASSWORD`, where `<TARGET>` is the target `name`\n  upper-cased with `-` replaced by `_` (so target `vc-prod` → `VMWARE_PRIVATEAI_VC_PROD_PASSWORD`).\n- **Optional username override**: `VMWARE_PRIVATEAI_<TARGET>_USERNAME` wins over `config.yaml` (resolved\n  together with the password on every call, so a rotating sidecar never splits the pair).\n- **PAIS token**: `VMWARE_PRIVATEAI_PAIS_TOKEN` — a short-lived OIDC/OAuth2 bearer token from your\n  Identity Provider (the PAIS API uses `Authorization: Bearer <token>`). It is a secret and is\n  obfuscated to `b64:` at rest exactly like a password.\n- **Secret manager**: any of these vars can be injected from Vault / CyberArk / AWS Secrets Manager /\n  a Kubernetes Secret instead of `.env` — the code reads the environment either way.\n\nOn load, plaintext `*_PASSWORD` / `*_TOKEN` values in `.env` are auto-rewritten to grep-safe `b64:`\nform (obfuscation, not encryption — it defeats casual grep / shoulder-surfing, not a determined reader).\n\n## 4. MCP client configuration\n\nPrefer"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":2079,"uniquenessScore":39,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T11:46:55.190Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T11:46:55.190Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T14:15:03.710Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}