{"id":"9440d813-4f8f-4efd-8713-929f1a2563c8","entityType":"agent","slug":"clawhub-agents365-ai-agent-native-design","name":"Agent Native Design","canonicalUrl":"https://www.xpersona.co/agent/clawhub-agents365-ai-agent-native-design","canonicalPath":"/agent/clawhub-agents365-ai-agent-native-design","generatedAt":"2026-10-11T03:55:06.544Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-11T01:06:57.128Z","emptyReason":null},"description":"Use when designing, reviewing, or refactoring a CLI that must serve AI agents alongside humans, or when converting an API or SDK into an agent-usable CLI int... Skill: Agent Native Design Owner: agents365-ai Summary: Use when designing, reviewing, or refactoring a CLI that must serve AI agents alongside humans, or when converting an API or SDK into an agent-usable CLI int... Tags: agent-native:1.3.3, cli:1.3.3, interface-design:1.3.3, latest:1.3.5, schema-driven:1.3.3, structured-output:1.3.3 Version history: v1.3.5 | 2026-07-08T17:35:37.031Z | auto - No file changes detecte","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.2K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s177ks88grnjrm2y5799tcjned83g880:agent-native-design","sourceUrl":"https://clawhub.ai/agents365-ai/agent-native-design","homepage":"https://clawhub.ai/agents365-ai/skills/agent-native-design","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/agents365-ai/agent-native-design","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/agents365-ai/skills/agent-native-design","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":62,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Use when designing, reviewing, or refactoring a CLI that must serve AI agents alongside humans, or when converting an API or SDK into an agent-usable CLI int..."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T01:06:57.128Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T01:06:57.128Z","emptyReason":null},"stars":null,"forks":null,"downloads":1210,"packageName":null,"latestVersion":"1.3.5","tractionLabel":"1.2K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T01:06:57.058Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T01:06:57.128Z","lastCrawledAt":"2026-10-11T01:06:57.058Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T01:06:57.058Z","lastVerifiedAt":null,"highlights":[{"version":"1.3.5","createdAt":"2026-07-08T17:35:37.031Z","changelog":"- No file changes detected in this version. - No updates to logic, instructions, or metadata. - Documentation and features remain unchanged from the previous release.","fileCount":12,"zipByteSize":31933},{"version":"1.3.4","createdAt":"2026-07-08T17:34:56.325Z","changelog":"- License changed from MIT to CC-BY-NC-4.0. - Skill version and metadata updated. - Redundant documentation files (e.g., README.md, skill-card.md, several docs) removed for consolidation. - Existing core references, rubric, design-patterns, and checklists updated or clarified. - Example links and help contract references corrected in SKILL.md.","fileCount":12,"zipByteSize":31719},{"version":"1.3.3","createdAt":"2026-05-05T08:00:56.300Z","changelog":"v1.3.3 — Step 0 is now notify-only (not silent-pull). Throttled to 1 check per 24h, surfaces version delta, only pulls on user consent. Corrects the v1.3.2 trust-boundary issue. https://github.com/Agents365-ai/agent-native-design/releases/tag/v1.3.3","fileCount":18,"zipByteSize":62882},{"version":"1.3.2","createdAt":"2026-05-05T07:53:32.369Z","changelog":"v1.3.2 — added Step 0 auto-update to standard workflow (silent git pull --ff-only on first use per conversation, ignored if not a git checkout). See https://github.com/Agents365-ai/agent-native-design/releases/tag/v1.3.2","fileCount":17,"zipByteSize":60393},{"version":"1.3.1","createdAt":"2026-05-05T07:47:08.274Z","changelog":"v1.3.1 — added references/testing.md (CI recipes for envelope contracts, idempotency replay, TTY behavior, schema drift, etc.) and bilingual concept hero. Earlier v1.3.0 split SKILL.md into core + references/. See https://github.com/Agents365-ai/agent-native-design/releases/tag/v1.3.1","fileCount":17,"zipByteSize":59896}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s177ks88grnjrm2y5799tcjned83g880:agent-native-design","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-agents365-ai-agent-native-design/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-agents365-ai-agent-native-design/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-agents365-ai-agent-native-design/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-agents365-ai-agent-native-design/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-agents365-ai-agent-native-design/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-agents365-ai-agent-native-design/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T03:55:06.540Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-agents365-ai-agent-native-design/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-agents365-ai-agent-native-design/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-agents365-ai-agent-native-design/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-agents365-ai-agent-native-design/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-11T01:06:57.128Z","emptyReason":null},"readme":"Skill: Agent Native Design\n\nOwner: agents365-ai\n\nSummary: Use when designing, reviewing, or refactoring a CLI that must serve AI agents alongside humans, or when converting an API or SDK into an agent-usable CLI int...\n\nTags: agent-native:1.3.3, cli:1.3.3, interface-design:1.3.3, latest:1.3.5, schema-driven:1.3.3, structured-output:1.3.3\n\nVersion history:\n\nv1.3.5 | 2026-07-08T17:35:37.031Z | auto\n\n- No file changes detected in this version.\n- No updates to logic, instructions, or metadata.\n- Documentation and features remain unchanged from the previous release.\n\nv1.3.4 | 2026-07-08T17:34:56.325Z | auto\n\n- License changed from MIT to CC-BY-NC-4.0.\n- Skill version and metadata updated.\n- Redundant documentation files (e.g., README.md, skill-card.md, several docs) removed for consolidation.\n- Existing core references, rubric, design-patterns, and checklists updated or clarified.\n- Example links and help contract references corrected in SKILL.md.\n\nv1.3.3 | 2026-05-05T08:00:56.300Z | user\n\nv1.3.3 — Step 0 is now notify-only (not silent-pull). Throttled to 1 check per 24h, surfaces version delta, only pulls on user consent. Corrects the v1.3.2 trust-boundary issue. https://github.com/Agents365-ai/agent-native-design/releases/tag/v1.3.3\n\nv1.3.2 | 2026-05-05T07:53:32.369Z | user\n\nv1.3.2 — added Step 0 auto-update to standard workflow (silent git pull --ff-only on first use per conversation, ignored if not a git checkout). See https://github.com/Agents365-ai/agent-native-design/releases/tag/v1.3.2\n\nv1.3.1 | 2026-05-05T07:47:08.274Z | user\n\nv1.3.1 — added references/testing.md (CI recipes for envelope contracts, idempotency replay, TTY behavior, schema drift, etc.) and bilingual concept hero. Earlier v1.3.0 split SKILL.md into core + references/. See https://github.com/Agents365-ai/agent-native-design/releases/tag/v1.3.1\n\nArchive index:\n\nArchive v1.3.5: 12 files, 31933 bytes\n\nFiles: agents/openai.yaml (1124b), references/checklists.md (3696b), references/citations.md (3748b), references/design-patterns.md (12673b), references/examples.md (8337b), references/hybrid-mcp-cli.md (3986b), references/rubric.md (3566b), references/testing.md (12206b), scripts/validate-metadata.sh (2722b), skill-card.md (3251b), SKILL.md (12880b), _meta.json (138b)\n\nFile v1.3.5:SKILL.md\n\n---\nname: agent-native-design\ndescription: Use when designing, reviewing, or refactoring a CLI that must serve AI agents alongside humans, or when converting an API or SDK into an agent-usable CLI interface.\nlicense: CC-BY-NC-4.0\nhomepage: https://github.com/Agents365-ai/agent-native-design\ncompatibility: Includes sidecar metadata for OpenClaw, Hermes, pi-mono, and OpenAI Codex; the core SKILL.md is portable to any agent runtime that supports Agent Skills-style instructions.\nplatforms: [macos, linux, windows]\nmetadata: {\"openclaw\":{\"requires\":{},\"emoji\":\"⌨️\",\"os\":[\"darwin\",\"linux\",\"win32\"]},\"hermes\":{\"tags\":[\"cli\",\"agent-native\",\"interface-design\",\"structured-output\",\"schema-driven\"],\"category\":\"engineering\",\"requires_tools\":[],\"related_skills\":[]},\"pimo\":{\"category\":\"engineering\",\"tags\":[\"cli\",\"agent-native\",\"interface-design\",\"structured-output\",\"schema-driven\"]},\"author\":\"Agents365-ai\",\"version\":\"1.3.5\"}\n---\n\n# agent-native-design\n\n## Purpose\n\nThis skill helps analyze, design, and refactor command-line tools so they can reliably serve **humans**, **AI agents**, and **orchestration systems** at the same time.\n\nIt is not a skill for merely *using* a CLI. It is a skill for designing and reviewing a CLI as an **agent-native interface**.\n\nThe skill focuses on four goals:\n\n1. Make CLI behavior predictable for AI agents.\n2. Make CLI output readable and recoverable for humans.\n3. Make CLI execution manageable for systems and orchestrators.\n4. Define a complete interaction loop from authentication to error routing.\n\n---\n\n## When to use this skill\n\nUse this skill when the user wants to:\n\n* evaluate whether an existing CLI is agent-friendly\n* redesign a CLI to better support AI agents\n* convert an API or SDK into an agent-native CLI\n* review help output, schema design, exit codes, or JSON contracts\n* design dry-run, auth delegation, or safety boundaries\n* generate CLI skills, docs, or interface conventions from schema\n* refactor a human-oriented CLI into a machine-friendly one\n* define how a CLI should interact with an agent runtime\n\nTypical prompts include:\n\n* \"Review this CLI and tell me whether it is agent-native.\"\n* \"Design a CLI for this API that an AI agent can use reliably.\"\n* \"Refactor this tool so stdout is machine-readable and safer for agents.\"\n* \"Help me define schema introspection, dry-run, and exit code semantics.\"\n\n## When not to use this skill\n\nDo not use this skill when the user only wants:\n\n* help running a specific command\n* installation help for a CLI\n* shell troubleshooting unrelated to interface design\n* generic Linux or terminal tutorials\n* agent planning or memory design unrelated to tools\n* API business logic review without any CLI/tooling layer\n\n---\n\n## Core model\n\nAn agent-native CLI must simultaneously serve three audiences.\n\n### 1. Human\n\nNeeds: readable output, friendly error messages, onboarding guidance.\nChannels: `stderr`, optional `--format table`, interactive TUI when appropriate.\n\n### 2. AI Agent\n\nNeeds: structured data, stable contracts, self-description.\nChannels: `stdout` as JSON, stable exit codes, schema introspection, dry-run previews, generated skills/docs.\n\n### 3. System / Orchestrator\n\nNeeds: delegated authentication, process management, deterministic error routing.\nChannels: environment variables, exit codes, dry-run mode, stable command semantics.\n\n### Foundational contract\n\n| Channel | Primary audience |\n|---------|-----------------|\n| `stdout` | Machines and agents |\n| `stderr` | Humans |\n| `exit codes` | Systems and orchestrators |\n\nThis skill teaches how to make CLI a first-class interface for agents. Production agents (Claude Code, Cursor, Gemini CLI) often pair a CLI with an MCP server — CLI for state changes and local/scriptable work, MCP for multi-tenant SaaS and per-user auth. When a design needs the MCP side as well, see `references/hybrid-mcp-cli.md` for the decision matrix and the benchmark data behind the CLI/MCP tradeoff.\n\n---\n\n## The complete interaction loop\n\n| Phase | Step | Description |\n|-------|------|-------------|\n| 0. Bootstrap | 1 | Human/system obtains auth token or credentials |\n| 0. Bootstrap | 2 | Set trusted env vars: token, profile, safety mode |\n| 1. Discovery | 3 | Agent loads skills or command summaries |\n| 1. Discovery | 4 | Agent queries schema/help for parameters |\n| 2. Planning | 5 | Agent uses `--dry-run` to preview request shape |\n| 3. Execution | 6 | Agent executes with validated inputs |\n| 4. Interpretation | 7 | Agent parses structured result |\n| 5. Recovery | 8 | Agent uses exit code + error object to retry, re-auth, repair, or escalate |\n\nA CLI that does not support every phase is incomplete from the agent's perspective.\n\n---\n\n## Seven principles\n\nThese are load-bearing. Each principle has at least one rubric criterion and at least one example backing it.\n\n### Principle 0. One CLI, Three Audiences\n\nThe CLI must serve human, agent, and system simultaneously. A design that serves only one audience is incomplete.\n\n### Principle 1. Structured Output Is the Interface\n\n`stdout` should always be parseable and stable. Both success and failure are structured JSON. The CLI must decide for itself which audience is reading: detect at startup whether stdout is a TTY, default to JSON when it is not, default to human-readable when it is. `NO_COLOR` and an explicit `--format json|table` flag override the auto-detection. Agents should never have to remember to pass `--format json` — if they have to, they will forget, and the run will silently produce un-parseable prose. Envelope and error contract: `references/design-patterns.md#output-envelopes`.\n\n### Principle 2. Trust Is Directional\n\nCLI arguments are not inherently trusted — they may come from a hallucinating or prompt-injected agent. Environment-level configuration set by the human or system is more trusted. The agent chooses *what to do* within a bounded surface; the human defines *where and how it is allowed to operate*.\n\n### Principle 3. The CLI Must Describe Itself\n\nThe CLI must be self-describing enough that an agent can use it without reading external README files. Self-description must be **progressive**, not eager: top-level `--help` lists resources; resource help lists actions; action help lists flags; a separate `schema <resource.action>` returns the full typed schema. A CLI with hundreds of commands that dumps everything into the first `--help` pays that token cost on every agent invocation. See `references/design-patterns.md#help-design` and `references/examples.md` Examples 2 and 5.\n\n### Principle 4. Safety Through Graduated Visibility\n\nRead commands are easy to discover; mutating commands carry explicit warnings; destructive commands are hidden from skills or gated separately. Tier table and rationale: `references/design-patterns.md#safety-design`. Tiers are necessary but not sufficient — they are a prompt-side defense and approval fatigue degrades them quickly. Assume the agent runtime will additionally sandbox the CLI at the OS level (filesystem, network, processes), and design destructive commands to fail closed inside that sandbox.\n\n### Principle 5. Validate at the Boundary, Not in the Middle\n\nInputs are validated once at the CLI entry point. Internal code operates on validated, typed, trusted structures. Validation functions are centralized and tested for both pass and reject cases.\n\n### Principle 6. The Schema Is the Source of Truth\n\nIf a schema exists, everything derives from it: CLI command structure, validation rules, help text, generated docs, generated skills, type definitions, dry-run contracts. The schema is never manually duplicated. The schema must also carry its own version and deprecation signals, surfaced in the `meta` block of every response, so agents that have cached an older view can detect drift and re-discover rather than silently calling a removed method. Full versioning contract and example: `references/design-patterns.md#schema-versioning`.\n\n### Principle 7. Authentication Must Be Delegatable\n\nAuthentication is obtained and refreshed by human/system-managed flows. The agent uses credentials; it never owns the auth lifecycle. Preferred mechanisms: environment variables, config files, OS keychain integration, externally refreshed tokens. Canonical pattern: `references/examples.md` Example 3.\n\n---\n\n## Standard review workflow\n\n### Step 1. Classify the input\n\nDecide whether the user is providing: an existing CLI, an API to be wrapped, a conceptual design, a partial interface, or a failure case.\n\n### Step 2. Map the three audiences\n\n**Human:** Is there readable output? Are errors understandable? Is onboarding supported?\n\n**Agent:** Is stdout stable JSON? Can the CLI describe itself? Is there schema introspection and dry-run?\n\n**System:** Is auth delegatable? Are exit codes stable? Can failures be routed deterministically?\n\n### Step 3. Review the interaction loop\n\nCheck whether the CLI supports: bootstrap, discovery, parameter understanding, preview, execution, parsing, recovery.\n\n### Step 4. Score the CLI with the rubric, then map back to principles\n\nUse the 14-criterion rubric to score the CLI. The full rubric lives in `references/rubric.md`. Every one of the seven principles has at least one rubric row backing it, so the score-to-principle mapping is total: P0 → Three-audience support, Non-interactive operation; P1 → Stdout contract, Stderr separation, Idempotent retries, Error recoverability; P2 → Trust boundary; P3 → Self-description (help), Dry-run; P4 → Safety tiers; P5 → Boundary validation; P6 → Schema introspection; P7 → Auth delegation. Then summarize per principle with evidence, risk, and recommendation. The full review checklists live in `references/checklists.md`.\n\n### Step 5. Produce a refactor plan\n\n- **P0** must fix\n- **P1** should improve\n- **P2** long-term enhancements\n\n---\n\n## Default output format\n\n### 1. Overall verdict\n\nState whether the CLI is **agent-native**, **partially agent-native**, or **not yet agent-native**.\n\n### 2. Three-audience contract review\n\nAssess support for human, agent, system.\n\n### 3. Interaction loop coverage\n\nAssess each phase: auth bootstrap → env setup → skill/help discovery → schema introspection → dry-run → execution → parsing and recovery.\n\n### 4. Rubric score + seven-principle review\n\nReport the 14-criterion rubric score first, then summarize the seven principles as: status · evidence · issue · recommendation.\n\n### 5. Key risks\n\nSummarize design failures: human-only output, unstable JSON, no schema introspection, destructive commands overexposed, auth coupled to agent, ambiguous exit codes.\n\n### 6. Refactor plan\n\nPrioritized recommendations with examples drawn from `references/examples.md`.\n\n---\n\n## Things this skill should avoid recommending\n\n* Human-readable prose as the only output contract\n* README required for basic command discovery\n* Schema and validation that drift apart\n* Auth supplied primarily via agent-generated arguments\n* Destructive actions exposed by default\n* CLI behavior that depends on undocumented conventions\n* Errors that are only textual and not machine-routable\n* Mutating commands that are not idempotent under retry\n* Confirmation prompts with no `--yes` escape and no TTY-aware fallback\n* Eager schema dumps in top-level `--help` — agents that call the CLI in loops pay this cost on every invocation; use progressive disclosure instead. The token-cost rationale lives in `references/hybrid-mcp-cli.md`.\n\n---\n\n## Reference files\n\nLoad on demand — these are not in the agent's context until needed:\n\n| File | Read when |\n|------|-----------|\n| `references/examples.md` | Showing the user a good envelope, error, dry-run, batch response, or anti-pattern |\n| `references/rubric.md` | Producing the score component of a CLI review |\n| `references/checklists.md` | Walking through a CLI auditing list with the user, or sanity-checking a new design |\n| `references/design-patterns.md` | Writing the contract for envelopes, exit codes, idempotency, non-interactive mode, long-running commands, schema versioning, locale/time |\n| `references/hybrid-mcp-cli.md` | Deciding CLI vs. MCP vs. both, or citing benchmark numbers behind the CLI efficiency claim |\n| `references/testing.md` | Showing the user how to verify their CLI actually upholds the contract (envelope shape, idempotency replay, TTY behavior, schema drift, dry-run safety, locale determinism) — load this when the design review converges on \"how do we keep it agent-native over time?\" |\n| `references/citations.md` | Citing the primary sources behind a recommendation |\n\n---\n\n## One-sentence summary\n\nThis skill helps turn a CLI into a trustworthy execution interface for **humans, AI agents, and systems** through **structured output, self-description, delegated authentication, safety boundaries, and a complete interaction loop**.\n\nFile v1.3.5:_meta.json\n\n{\n  \"ownerId\": \"kn74y2dmpaszt959h16p11k53983ht10\",\n  \"slug\": \"agent-native-design\",\n  \"version\": \"1.3.5\",\n  \"publishedAt\": 1783532137031\n}\n\nFile v1.3.5:references/checklists.md\n\n# Review Checklists\n\nUse this when evaluating a CLI for agent readiness, or when sanity-checking a new design before shipping.\n\n## Output\n\n- [ ] `stdout` is valid JSON when stdout is not a TTY or when `--format json` is passed (success and failure)\n- [ ] `stderr` carries human-readable diagnostics only\n- [ ] JSON envelope is stable: `{ \"ok\": bool, \"data\": ... }` or `{ \"ok\": false, \"error\": ... }`\n- [ ] Error object includes: `code`, `message`, `retryable`\n- [ ] No prose mixed into `stdout`\n\n## Exit codes\n\n- [ ] Exit codes are documented\n- [ ] Exit codes are stable across versions\n- [ ] Distinct codes for: success (0), runtime error, auth error, validation error\n- [ ] Exit code mapping is available via `--help` or `schema`\n\n## Retry and interaction mode\n\n- [ ] Every mutating command accepts `--idempotency-key`\n- [ ] Retried calls with the same idempotency key return the original result\n- [ ] `retryable` field in error envelope is meaningful and correct\n- [ ] CLI never prompts for input when stdin is not a TTY\n- [ ] `--yes` / `--no-input` / `--force` supported on every command that would otherwise prompt\n- [ ] Returns structured `confirmation_required` error instead of blocking on missing confirmation\n- [ ] stdout defaults to JSON when stdout is not a TTY (no `--format json` required)\n- [ ] Pagers (`less`, `more`) disabled when stdout is not a TTY\n\n## Self-description\n\n- [ ] Top-level `--help` lists all resources/commands\n- [ ] Resource-level `--help` lists actions\n- [ ] Action-level `--help` lists all flags with types\n- [ ] Schema introspection command available (`tool schema <resource.action>`)\n- [ ] Dry-run available for all mutating commands\n\n## Safety\n\n- [ ] Read commands clearly discoverable\n- [ ] Write/mutating commands carry explicit warning in help\n- [ ] Destructive commands (delete/purge) hidden from skills or gated\n- [ ] Dry-run covers all write operations\n\n## Auth\n\n- [ ] Human/system manages token acquisition (browser flow, keychain)\n- [ ] Agent receives credential via env var or pre-fetched token\n- [ ] Agent never navigates OAuth2 or browser flows\n- [ ] Token refresh handled outside agent runtime\n\n## Trust\n\n- [ ] CLI args treated as untrusted (validated at boundary)\n- [ ] Environment variables used for config/safety settings (human-set)\n- [ ] Agent cannot escalate its own privileges via CLI args\n\n## Schema\n\n- [ ] Schema is the single source of truth\n- [ ] CLI command structure derives from schema\n- [ ] Validation derives from schema\n- [ ] Help text derives from schema\n- [ ] Generated skills derive from schema (if applicable)\n- [ ] Schema version included in every response's `meta` block\n- [ ] Deprecation signals included in schema responses (`deprecated_fields`, `replaced_by`, `removed_in`)\n- [ ] Schema introspection is incremental (not eager): `--help` is small; full schema via `schema` subcommand only\n\n## Token efficiency\n\nFor agents that call the CLI in loops (orchestration, multi-turn planning), context cost matters. These items help CLIs realize the per-call token advantage over eager-loaded MCP servers (see `hybrid-mcp-cli.md` for the benchmark data this is based on).\n\n- [ ] Top-level `--help` response is under 500 tokens\n- [ ] Full schema is not dumped in top-level `--help`; accessed via `schema <resource.action>` instead\n- [ ] Field selection supported on list responses (`--json field1,field2,...` or similar)\n- [ ] Default list responses are compact (3–5 fields); full detail via `--full` flag\n- [ ] Requests that would normally require two CLI calls are collapsed (e.g., `count` + `list` → return count in the envelope)\n- [ ] Schema versioning allows agents to cache and avoid re-discovery on every invocation\n\nFile v1.3.5:references/citations.md\n\n# Citations\n\nSources actually quoted or directly relied on in SKILL.md and the other reference files. The broader landscape (clig.dev, the gh / aws / kubectl design corpus) is implicit background.\n\n## Anthropic engineering\n\n- *Code execution with MCP: Building more efficient agents* — https://www.anthropic.com/engineering/code-execution-with-mcp (Nov 4, 2025). Progressive disclosure of tool definitions; the 150K → 2K token reduction case study cited under Principle 3 / `design-patterns.md#help-design`.\n- *Beyond permission prompts: making Claude Code more secure and autonomous* — https://www.anthropic.com/engineering/claude-code-sandboxing (Oct 20, 2025). Approval fatigue and the 84% prompt-reduction figure cited under Principle 4 / `design-patterns.md#safety-design`.\n\n## CLI-for-agents writing\n\n- Ugo Enyioha, *Writing CLI Tools That AI Agents Actually Want to Use* — https://dev.to/uenyioha/writing-cli-tools-that-ai-agents-actually-want-to-use-39no (Feb 27, 2025). Idempotency-on-retry and \"agents cannot type 'y'\" framings cited in the Idempotency and Non-interactive sections of `design-patterns.md`.\n- Thibault Le Ouay Ducasse / openstatus, *Building a CLI That Works for Humans and Machines* — https://www.openstatus.dev/blog/building-cli-for-human-and-agents (Apr 2, 2026). TTY detection as the human/machine switch cited under Principle 1.\n- Mario Zechner, *MCP vs CLI: Benchmarking Tools for Coding Agents* — https://mariozechner.at/posts/2025-08-15-mcp-vs-cli/ (Aug 15, 2025). Empirical case that many MCP servers could be CLI invocations.\n- Armin Ronacher, *Skills vs Dynamic MCP Loadouts* — https://lucumr.pocoo.org/2025/12/13/skills-vs-mcp/ (Dec 13, 2025). Schema/API stability as a first-class concern cited under Principle 6.\n\n## CLI-vs-MCP benchmarks (2026)\n\n- Jannik Reinhard, *CLI Tools vs MCP: Better AI Agents With Less Context* — https://jannikreinhard.com/2026/02/22/why-cli-tools-are-beating-mcp-for-ai-agents/ (Feb 22, 2026). Source for the 28% / 33% / 55K / 35× numbers in `hybrid-mcp-cli.md`.\n- Manveer Chawla, *MCP vs. CLI for AI agents: When to Use Each (A Practical Decision Framework for 2026)* — https://manveerc.substack.com/p/mcp-vs-cli-ai-agents (Mar 8, 2026). Per-integration decision framework; production examples (Claude Code, Cowork) using both transports.\n- Soumyadeb Mitra / RudderStack, *CLI or MCP or both? The design pattern for AI agents managing your data stack* — https://www.rudderstack.com/blog/ai-agents-cli-mcp-design-pattern/ (Mar 18, 2026). The \"writes via CLI, reads via MCP\" split underpinning `hybrid-mcp-cli.md`.\n\n## Pre-agent baseline\n\n- *Scripting with GitHub CLI* — https://github.blog/engineering/engineering-principles/scripting-with-github-cli/ (Mar 11, 2021). `gh api --jq` and structured JSON output as a first-class CLI pattern. The post predates the `gh <resource> --json field1,field2` field-selection flag (cli/cli#1089) but lays out the design philosophy that flag is built on.\n- *Command Line Interface Guidelines* — https://clig.dev/. The pre-agent baseline for human-first CLI design. This skill extends it; it does not replace it.\n- `sysexits.h` — https://manpages.ubuntu.com/manpages/noble/man3/sysexits.h.3head.html. The BSD exit-code vocabulary that `design-patterns.md#exit-code-model` deliberately simplifies away from.\n\n---\n\n## A note on the \"agent-native CLI\" framing\n\nThe term used throughout this skill is one of several competing framings in current writing. *Agent-first CLI* (Propel, Keyboards Down) is more common; *CLI for humans and machines* (openstatus, Linearis) is the most descriptive. \"Native\" is chosen here to emphasize that agent support is a first-class design goal, not a retrofit on top of a human-only CLI.\n\nFile v1.3.5:references/design-patterns.md\n\n# Design Patterns\n\nReference details for specific design areas. SKILL.md states the principles; this file holds the concrete contracts and rules.\n\n---\n\n## Output envelopes\n\nSuccess:\n\n```json\n{ \"ok\": true, \"data\": {} }\n```\n\nFailure:\n\n```json\n{\n  \"ok\": false,\n  \"error\": {\n    \"code\": \"validation_error\",\n    \"message\": \"Missing required field: email\",\n    \"field\": \"email\",\n    \"retryable\": false\n  }\n}\n```\n\nPartial success (batch operations):\n\n```json\n{\n  \"ok\": \"partial\",\n  \"data\": {\n    \"succeeded\": [\n      { \"id\": \"msg_001\", \"status\": \"sent\" },\n      { \"id\": \"msg_002\", \"status\": \"sent\" }\n    ],\n    \"failed\": [\n      {\n        \"id\": \"msg_003\",\n        \"error\": { \"code\": \"rate_limited\", \"message\": \"Rate limit exceeded for recipient\", \"retryable\": true }\n      }\n    ]\n  }\n}\n```\n\nA batch command that collapses any per-item failure into a top-level `ok: false` forces every agent to re-process its successful items on retry. AWS SQS's `ReportBatchItemFailures` and similar APIs settled on this per-item shape for the same reason. Pair partial success with an idempotency key on the batch as a whole so that re-running with the same key only re-issues the *failed* items.\n\nOptional `meta` slot for observability:\n\n```json\n{\n  \"ok\": true,\n  \"data\": { \"id\": \"abc123\" },\n  \"meta\": {\n    \"request_id\": \"req_8fa9c1\",\n    \"latency_ms\": 412,\n    \"schema_version\": \"1.4.0\"\n  }\n}\n```\n\n`meta` is a freeform slot for telemetry the orchestrator may want without polluting `data`: `request_id` for log correlation, `latency_ms` for SLO tracking, `schema_version` so an agent can detect drift against a cached schema. When the underlying call has a token / quota cost the CLI knows about, surface it here too — `tokens_used`, `quota_remaining`. Optional and additive: agents that don't need it ignore it; agents that do gain observability without an extra round-trip. Aligns with where the OpenTelemetry GenAI semantic conventions are heading for tool-call traces.\n\n---\n\n## Schema versioning\n\nThis is the contract that lets agents cache schemas safely across calls.\n\nEvery response should carry `schema_version` in the optional `meta` block (see envelope above). When an agent's cached schema version (e.g., 1.2.0) doesn't match the CLI's current version (1.4.0), the agent knows to re-discover before re-planning.\n\nSchema introspection responses should declare which CLI version produced them, when each method was introduced, and whether any field is deprecated:\n\n```json\n{\n  \"method\": \"sleep.list\",\n  \"since\": \"1.2.0\",\n  \"deprecated\": false,\n  \"params\": {\n    \"startDate\": { \"type\": \"string\", \"format\": \"date\", \"required\": true },\n    \"endDate\":   { \"type\": \"string\", \"format\": \"date\", \"required\": true },\n    \"page_size\": { \"type\": \"integer\", \"default\": 20, \"max\": 100, \"deprecated\": true, \"replaced_by\": \"pageSize\", \"removed_in\": \"1.5.0\" }\n  }\n}\n```\n\nThis gives agents:\n\n- **Drift detection** — cached version mismatch triggers re-discovery before re-planning.\n- **Deprecation awareness** — agents migrate off `page_size` to `pageSize` proactively, before removal.\n- **Non-breaking updates** — adding a new optional field or method makes cached schemas incomplete but not incorrect; existing calls keep working.\n- **Token efficiency** — agents don't waste tokens on removed methods or obsolete fields; drift is corrected in one round-trip.\n\nAPI stability is a contract you owe the agent: a CLI that renames flags between point releases forces every dependent agent to re-discover and re-plan. Treat the schema as a versioned, append-mostly surface.\n\nSee `references/examples.md` Example 8 for the full request/response pair.\n\n---\n\n## Exit code model\n\n| Code | Meaning |\n|------|---------|\n| `0` | Success |\n| `1` | Runtime / API error |\n| `2` | Auth error |\n| `3` | Validation error |\n\nExact codes may vary — the mapping must be documented and deterministic. (`sysexits.h` defines a richer pre-agent vocabulary — `EX_USAGE=64`, `EX_DATAERR=65`, `EX_NOPERM=77`, `EX_CONFIG=78` — and a CLI is free to use it. The 0–3 mapping above is a deliberate simplification for agent routing; what matters is that codes are documented, stable, and distinct per failure class.)\n\n---\n\n## Idempotency and retry\n\nAgents retry. Networks fail, processes get killed, exit codes get misread. A CLI that is not idempotent forces every agent that uses it to write special-case retry logic — and most agents will get it wrong.\n\nEvery mutating command should accept `--idempotency-key <string>`. A retried call carrying the same key must be a safe no-op that returns the original result, in the same `ok` / `error` shape. The CLI is responsible for storing the key↔result mapping for long enough that legitimate retries can find it (typically minutes; the upper bound belongs to the underlying API).\n\nThis pairs with the `retryable` field in the error envelope: `retryable: true` tells the agent it is *safe* to call again with the same idempotency key, and that doing so will eventually converge; `retryable: false` tells the agent that retrying will not change the outcome.\n\n> \"Agents retry. Networks fail. Commands get interrupted. If your `create` command fails on the second run because the resource already exists, the agent has to write special-case retry logic.\" — Ugo Enyioha, *Writing CLI Tools That AI Agents Actually Want to Use* (Feb 2025)\n\n---\n\n## Non-interactive operation\n\nAn agent cannot answer a confirmation prompt, cannot type a password into a TTY, and cannot navigate a curses-style menu. A CLI that hangs waiting for stdin when stdin is not a TTY is, from the agent's perspective, broken.\n\nRules:\n\n- **Never prompt when stdin is not a TTY.** Detect at startup; if a confirmation would normally fire but `isatty(stdin) == false`, return a structured error instead of blocking:\n  ```json\n  { \"ok\": false, \"error\": { \"code\": \"confirmation_required\", \"message\": \"Pass --yes to confirm.\", \"retryable\": false } }\n  ```\n- **Always support `--yes` / `--no-input` / `--force`** (or equivalent) on every command that would otherwise prompt, so a human running interactively can opt out and an agent always runs without prompts.\n- **Never read secrets from interactive prompts in agent contexts.** Secrets come from environment variables or pre-set config (see Principle 7).\n- **Pagers off when stdout is not a TTY.** Detect at startup; never invoke `less` / `more` style pagers that would block on a non-TTY consumer.\n\n> \"An agent cannot type 'y' at a confirmation prompt. If your CLI hangs waiting for input, the agent's workflow is dead.\" — Ugo Enyioha (Feb 2025)\n\n---\n\n## Long-running commands and streaming\n\nA command that takes minutes to finish is a hazard for an agent: the agent does not know whether the CLI is making progress, stuck, or dead, and it cannot afford to wait blind on a single JSON envelope at the end. Two patterns work:\n\n**Structured progress on stderr, final JSON on stdout.** The agent reads stderr for liveness and stdout for the result. Progress events are themselves structured (one JSON object per line) so the orchestrator can parse them, but they never pollute the stdout JSON envelope. See `references/examples.md` Example 7 for a full transcript.\n\n**NDJSON streaming for long lists.** When a list command might return thousands of items, offer a `--stream` (or `--ndjson`) mode that emits one JSON object per line on stdout, with a final summary line. Agents can process the stream incrementally and stop when they have enough:\n\n```\n$ healthkit sleep list --since 2024-01-01 --stream\n{ \"ok\": true, \"data\": { \"id\": \"sl_001\", \"date\": \"2024-01-01\", \"minutes\": 412 } }\n{ \"ok\": true, \"data\": { \"id\": \"sl_002\", \"date\": \"2024-01-02\", \"minutes\": 388 } }\n...\n{ \"ok\": true, \"summary\": { \"count\": 730, \"has_more\": false } }\n```\n\nEither way: the agent must be able to tell, from output alone, whether the command is *making progress*, *finished*, or *failed*. Silent multi-minute waits are an availability bug, not a UX preference.\n\n---\n\n## Reducing agent round-trips\n\nThe cost of every CLI invocation an agent makes is paid twice: once in latency, once in context tokens. A CLI that takes three calls to surface what an agent needs in order to plan its next step is, for the agent, *worse* than one that takes one call — even if every individual call is faster. Optimize for round-trip count, not just per-call performance.\n\nConcrete tactics:\n\n- **Pre-compute aggregates in list responses.** A `list` that returns `{ \"ok\": true, \"data\": [...], \"count\": 7, \"has_more\": false }` saves a follow-up `count` call.\n- **Definitive empty states.** Return `{ \"ok\": true, \"data\": [], \"count\": 0 }`, never `null`. The agent should never have to disambiguate \"no results\" from \"missing field.\"\n- **Field selection on the response side.** Borrow from `gh --json title,number,state`: let callers ask for only the fields they need so list responses stay small and a follow-up \"give me more detail on item N\" is the exception, not the rule.\n- **Compact default, `--full` escape hatch.** List items should carry 3–4 fields by default; agents that need more pass an opt-in flag rather than parsing huge default payloads on every call.\n- **Next-step hints in success responses.** When the agent's likely next action is predictable, include a `next` slot: `{ \"ok\": true, \"data\": {...}, \"next\": [\"healthkit sleep summary --start-date 2026-01-01 --end-date 2026-01-07\"] }`. The agent saves a discovery turn.\n- **Cursor pagination in the envelope.** `{ \"ok\": true, \"data\": [...], \"page\": { \"next_cursor\": \"...\", \"has_more\": true } }` so the agent can decide whether to continue without parsing prose pagination markers.\n\n---\n\n## Help design\n\nProgressive, not monolithic: capability overview → resource → action → schema → examples → dry-run. A CLI with hundreds of commands should not dump its full schema into the agent's context on the first call. Top-level `--help` should be small enough to fit in a few hundred tokens; deeper detail is loaded on demand only when the agent has narrowed its target.\n\nAnthropic's *Code execution with MCP* (Nov 2025) reports the same insight from the MCP world: in one Google-Drive→Salesforce case, lazy-loading tool definitions reduced token usage from 150,000 to 2,000 — a 98.7% saving. The CLI equivalent is the layered help tree plus response-side field selection (`gh pr list --json number,title,state`).\n\n---\n\n## Safety design\n\nRead actions: easy to discover. Write actions: clearly marked. Destructive actions: hidden, gated, or separately enabled. Dry-run: everywhere feasible.\n\nTier table:\n\n| Tier | Commands | Exposure |\n|------|----------|----------|\n| preview | all commands | dry-run available everywhere |\n| open | list / get / search | full docs, easy to discover |\n| warned | create / update / send | explicit warning in help and skills |\n| hidden | delete / purge / empty | excluded from skills, gated separately |\n\nTiers are necessary but not sufficient. Graduated visibility is a prompt-side defense — it works only when the agent reads the warning and respects it, and approval fatigue degrades that defense quickly. Anthropic's *Beyond permission prompts* (Oct 2025) reports that OS-level sandboxing \"safely reduces permission prompts by 84%.\" An agent-native CLI should assume the agent runtime will additionally sandbox it at the OS level (filesystem, network, processes), and design destructive commands to fail closed inside that sandbox rather than relying on a single layer of warnings.\n\n---\n\n## Auth design\n\nHuman/system-managed token acquisition. Environment/config-based delegation. No agent involvement in browser auth flows. Separation between auth bootstrap and agent execution. See `references/examples.md` Example 3 for the canonical pattern.\n\n---\n\n## Locale, time, and determinism\n\nAgent behavior breaks subtly when CLI output depends on the host's locale or timezone. Pin determinism at the CLI boundary so the agent never has to second-guess what `2026-04-11` means or whether `1,234.56` is one number or two.\n\n- **All timestamps are UTC ISO-8601** with explicit timezone (`2026-04-11T14:30:00Z`), not local time and not Unix epoch unless explicitly requested.\n- **All dates are ISO-8601** (`2026-04-11`), never `04/11/2026` or `11/04/2026`.\n- **Numeric formats are locale-independent.** Decimal point `.`, no thousands separators in JSON output. (`1234.56`, never `1,234.56` or `1.234,56`.)\n- **Internal subprocess calls run under `LC_ALL=C`** (or equivalent), so any tool the CLI shells out to — `date`, `sort`, `awk` — produces the same bytes on every host.\n- **Sort orders are documented and stable.** Default sort is byte-wise unless the schema says otherwise.\n\nFile v1.3.5:references/examples.md\n\n# Examples and Non-Examples\n\nConcrete patterns and anti-patterns referenced from `SKILL.md`. Read this when you need to show the user what a good envelope, error, dry-run, or batch response actually looks like — or when you need to make an anti-pattern visible.\n\n---\n\n## Good examples\n\n### Example 1 — Structured error with routing fields\n\n```json\n{\n  \"ok\": false,\n  \"error\": {\n    \"code\": \"auth_expired\",\n    \"message\": \"Token expired. Re-authenticate to continue.\",\n    \"retryable\": true,\n    \"retry_after_auth\": true\n  }\n}\n```\n\nThe agent can read `retry_after_auth: true` and escalate to re-authentication without parsing prose.\n\n### Example 2 — Layered self-description\n\n```bash\n$ healthkit --help\nUsage: healthkit <resource> <action> [options]\n\nResources:\n  sleep       Sleep records and stages\n  steps       Step count and activity\n  heart       Heart rate and HRV\n\n$ healthkit sleep --help\nActions:\n  list        List sleep records by date range\n  summary     Aggregate sleep statistics\n\n$ healthkit sleep list --help\nFlags:\n  --start-date  ISO date (required)\n  --end-date    ISO date (required)\n  --format      json|table (default: json)\n  --dry-run     Preview request, do not execute\n```\n\nAn agent can traverse this tree to discover valid commands without reading external docs.\n\n### Example 3 — Delegated auth with env trust boundary\n\n```bash\n# Human / system runs once, out of band (shell profile, systemd unit, supervisor):\nhealthkit auth login                              # browser OAuth2 flow, stores token in keychain\nexport HEALTHKIT_TOKEN=\"$(healthkit auth token)\"  # token is injected into the agent's environment\n\n# Agent's own commands — it never invokes `auth login` or `auth token`:\nhealthkit sleep list --start-date 2026-01-01 --end-date 2026-01-07\n```\n\nThe agent inherits `HEALTHKIT_TOKEN` from an environment it did not build. It never runs the login subcommand, never runs the `auth token` retrieval subcommand, and never handles refresh — those belong to the human or the orchestrator. The env var is the trust boundary; the agent consumes credentials, it does not fetch them.\n\n### Example 4 — Dry-run preview before execution\n\n```bash\n$ healthkit sleep list --start-date 2026-01-01 --end-date 2026-01-07 --dry-run\n{\n  \"ok\": true,\n  \"dry_run\": true,\n  \"would_request\": {\n    \"method\": \"GET\",\n    \"url\": \"https://health.api/v1/sleep\",\n    \"params\": { \"startDate\": \"2026-01-01\", \"endDate\": \"2026-01-07\" }\n  }\n}\n```\n\nThe agent can verify the request shape before committing to execution.\n\n### Example 5 — Schema introspection\n\n```bash\n$ healthkit schema sleep.list\n{\n  \"method\": \"sleep.list\",\n  \"params\": {\n    \"startDate\": { \"type\": \"string\", \"format\": \"date\", \"required\": true },\n    \"endDate\":   { \"type\": \"string\", \"format\": \"date\", \"required\": true },\n    \"pageSize\":  { \"type\": \"integer\", \"default\": 20, \"max\": 100 }\n  }\n}\n```\n\n### Example 6 — Idempotent batch with partial success, next hint, and meta\n\n```bash\n$ healthkit alerts send-bulk \\\n    --recipients \"user1,user2,user3\" \\\n    --message \"Reminder: log today's sleep\" \\\n    --idempotency-key \"alert-batch-2026-04-11-am\"\n{\n  \"ok\": \"partial\",\n  \"data\": {\n    \"succeeded\": [\n      { \"recipient\": \"user1\", \"alert_id\": \"alrt_abc\" },\n      { \"recipient\": \"user2\", \"alert_id\": \"alrt_def\" }\n    ],\n    \"failed\": [\n      {\n        \"recipient\": \"user3\",\n        \"error\": {\n          \"code\": \"rate_limited\",\n          \"message\": \"Per-recipient rate limit exceeded\",\n          \"retryable\": true,\n          \"retry_after_seconds\": 60\n        }\n      }\n    ]\n  },\n  \"next\": [\n    \"healthkit alerts send-bulk --recipients user3 --message 'Reminder: log today\\\\'s sleep' --idempotency-key alert-batch-2026-04-11-am\"\n  ],\n  \"meta\": {\n    \"request_id\": \"req_8fa9c1\",\n    \"latency_ms\": 412,\n    \"schema_version\": \"1.4.0\"\n  }\n}\n```\n\n`ok: \"partial\"` plus per-item `error.retryable` plus the `next` slot plus a stable idempotency key give the agent everything it needs to recover in one round-trip — it does not need to re-process `user1` / `user2`, it knows exactly which item to retry, and re-running with the same `--idempotency-key` is safe.\n\n### Example 7 — Long-running export with structured stderr progress\n\n```bash\n$ healthkit export run --dataset sleep --since 2024-01-01 --format parquet > result.json\n```\n\nstderr — one JSON object per line, agent reads for liveness without blocking on stdout:\n\n```\n{ \"event\": \"start\",    \"command\": \"export.run\", \"request_id\": \"req_abc123\" }\n{ \"event\": \"progress\", \"phase\": \"fetch\", \"done\": 240, \"total\": 730, \"elapsed_ms\": 18421 }\n{ \"event\": \"progress\", \"phase\": \"fetch\", \"done\": 730, \"total\": 730, \"elapsed_ms\": 54017 }\n{ \"event\": \"progress\", \"phase\": \"write\", \"done\": 730, \"total\": 730, \"elapsed_ms\": 56103 }\n{ \"event\": \"complete\", \"request_id\": \"req_abc123\", \"elapsed_ms\": 56234 }\n```\n\nstdout — single envelope at the end:\n\n```json\n{\n  \"ok\": true,\n  \"data\": { \"rows\": 730, \"path\": \"/tmp/sleep_export.parquet\", \"size_bytes\": 142336 },\n  \"meta\": { \"request_id\": \"req_abc123\", \"latency_ms\": 56234 }\n}\n```\n\nWhat this gives the agent: liveness via stderr without polluting the stdout envelope; phase visibility (`fetch` vs `write`) so it can localize bottlenecks; one clean result object captured by the redirect; and a single `request_id` correlating stderr progress, stdout result, and upstream service logs.\n\n### Example 8 — Schema versioning with deprecation signals\n\nAn agent that calls the CLI in a loop may have cached an older view of the schema. Versioning in the response envelope lets the agent detect drift without failing silently.\n\n```bash\n$ healthkit sleep list --start-date 2026-01-01 --end-date 2026-01-07\n{\n  \"ok\": true,\n  \"data\": [{ \"id\": \"sl_001\", \"date\": \"2026-01-01\", \"minutes\": 412 }],\n  \"meta\": {\n    \"schema_version\": \"1.4.0\",\n    \"deprecated_fields\": [\"page_size\"],\n    \"introduced_in\": \"1.2.0\",\n    \"request_id\": \"req_abc123\"\n  }\n}\n```\n\nSchema introspection route:\n\n```bash\n$ healthkit schema sleep.list\n{\n  \"method\": \"sleep.list\",\n  \"introduced_in\": \"1.2.0\",\n  \"schema_version\": \"1.4.0\",\n  \"params\": {\n    \"startDate\": { \"type\": \"string\", \"format\": \"date\", \"required\": true },\n    \"endDate\":   { \"type\": \"string\", \"format\": \"date\", \"required\": true },\n    \"pageSize\":  { \"type\": \"integer\", \"default\": 20, \"max\": 100 },\n    \"page_size\": { \"type\": \"integer\", \"deprecated\": true, \"replaced_by\": \"pageSize\", \"removed_in\": \"1.5.0\" }\n  }\n}\n```\n\nThe agent compares its cached `schema_version` against the response's; on mismatch, it re-discovers before re-planning. Deprecation signals (`deprecated_fields`, `replaced_by`, `removed_in`) let it migrate proactively, and non-breaking additions don't invalidate cached calls — they just leave the cache incomplete.\n\n---\n\n## Non-examples\n\n### Non-Example 1 — Prose-only error\n\n```\nError: something went wrong with your request. Please check your input and try again.\n```\n\nThe agent cannot determine: what went wrong, whether to retry, what to fix, which field failed. It must guess or give up.\n\n### Non-Example 2 — Mixed stdout\n\n```\nFetching sleep records...\nFound 7 records.\n{\"records\": [...]}\nDone.\n```\n\nThe agent cannot reliably parse JSON because stdout contains prose mixed with data.\n\n### Non-Example 3 — No self-description\n\n```bash\n$ mytool --help\nUsage: mytool [OPTIONS] COMMAND [ARGS]...\n\nOptions:\n  --help  Show this message and exit.\n```\n\nNo resources, no actions, no schema. An agent must guess or hallucinate command names.\n\n### Non-Example 4 — Auth via agent-supplied argument\n\n```bash\nmytool --token $AGENT_GENERATED_TOKEN delete --id abc123\n```\n\nThe agent controls the token. A compromised agent can use any token it manufactures, bypassing human trust boundaries.\n\n### Non-Example 5 — Destructive commands fully exposed\n\n```bash\n$ mytool --help\nCommands:\n  list    List records\n  get     Get a record\n  delete  Delete a record        ← appears at same level as read commands\n  purge   Purge all records      ← no warning, no gate\n```\n\nAn agent browsing help can trivially discover and invoke destructive commands.\n\n### Non-Example 6 — Ambiguous exit codes\n\n```bash\n$ mytool list; echo $?\n# Returns 1 on API error\n# Returns 1 on validation error\n# Returns 1 on auth error\n# Returns 1 on network error\n```\n\nExit code 1 means everything. The orchestrator cannot route failures deterministically.\n\nFile v1.3.5:references/hybrid-mcp-cli.md\n\n# When CLI vs MCP vs Both\n\nThis skill teaches CLI design — but the largest production agents (Claude Code, Cursor, Gemini CLI, CircleCI) use both CLI and MCP, not one or the other. This file is the single home for the CLI-vs-MCP discussion and the benchmark data behind it; SKILL.md and the checklists point here.\n\n---\n\n## The hybrid pattern\n\n**State changes happen through the CLI. System understanding happens through MCP.**\n\n- **Use CLI for:** local/scriptable tasks, composable automation, state-changing operations, dev/infrastructure workflows\n- **Use MCP for:** multi-tenant SaaS, per-user authentication, stateful workflows, audit logs, fine-grained access control\n- **Use both:** most production agents that orchestrate infrastructure (Vercel CLI + MCP for SaaS integrations; GitHub CLI + MCP for enterprise GitHub instances)\n\n---\n\n## Decision matrix\n\n| Scenario | CLI | MCP | Notes |\n|----------|-----|-----|-------|\n| Single-user dev tool on same machine | ✅ | | Process model is cheap; auth is local; composable with Unix pipes |\n| Large multi-tenant SaaS with per-user OAuth | | ✅ | Centralized auth; per-user scoping; network-attachable; no binary shipping required |\n| Hundreds of tools where schema size matters | ✅ | ⚠️ | CLI wins: eager MCP schema dumps consume 55K–80K tokens upfront. CLI lazy-loads via progressive help. |\n| Orchestration + infrastructure changes | ✅ | | State changes favor process-model CLIs |\n| Complex permission models, audit requirements | | ✅ | MCP's structured audit logs and per-user attribution |\n| Hybrid: local infra + cloud SaaS | ✅✅ | ✅ | CLI for infrastructure, MCP for SaaS. Both in same agent. |\n\n---\n\n## Benchmark data\n\nThese are the numbers behind the \"CLI is more efficient than eager-loaded MCP\" claim. Cite this section when SKILL.md or the checklists need backing.\n\n- **Task completion:** CLI-based agents achieve **28% higher task completion** vs. MCP-only agents with the same token budget (Reinhard 2026).\n- **Token efficiency:** **33% advantage** measured by Token Efficiency Score (CLI: 202, MCP: 152).\n- **Per-task overhead:** ~4,150 tokens (CLI) vs ~145,000 tokens (MCP) for an identical browser-automation task — a 35× reduction (Reinhard 2026).\n- **Schema dump cost:** MCP servers that load all tool schemas upfront consume **55K–80K tokens** just for discovery. An agent running 10 sequential operations sees this overhead on every orchestration handoff.\n- **Lazy-loading wins, in either world:** Anthropic's *Code execution with MCP* (Nov 2025) reports that presenting MCP tools as code on a filesystem reduced one Google-Drive→Salesforce workflow from 150,000 tokens to 2,000 — a 98.7% saving. The same logic produces CLI's structural advantage: progressive `--help` is lazy-loading by default.\n\nThese numbers are why a CLI's progressive `--help` → resource help → `schema <resource.action>` pattern matters: the agent only pays for the parts it queries, not for everything the tool could do.\n\n---\n\n## When to stick with CLI alone\n\nMario Zechner's empirical benchmark (Aug 2025) of MCP vs CLI for coding agents lands on a one-line conclusion that's worth taking seriously: *\"Just like a lot of meetings could have been emails, a lot of MCPs could have been CLI invocations.\"* That doesn't make MCP wrong; it means the default has been wrong. For the workflows this skill targets — developer tools, infrastructure CLIs, single-user data and research workflows — CLI is the lighter, more inspectable, more composable choice.\n\n## When to switch to MCP or hybrid\n\nIf you reach a design where you'd be fighting the CLI process model (per-request user context, fine-grained per-call authorization, network-attached without local install, multi-tenant data isolation), that's the signal to add MCP to the mix, not to bend this skill out of shape. Consult the decision matrix above; if you need features from the MCP column, embrace the hybrid approach that production agents use.\n\nFile v1.3.5:references/rubric.md\n\n# Rubric\n\n14 criteria, scored 0–2 each. Every criterion is tagged with the principle it backs, so scoring the rubric is the same act as auditing the seven principles. Use this when producing the score component of a CLI review.\n\n| Criterion | Principle | 0 — Fail | 1 — Partial | 2 — Pass |\n|-----------|-----------|----------|-------------|----------|\n| **Three-audience support** | P0 | Designed for only one audience (human-only, or agent-only) | Serves two audiences well; the third is an afterthought or broken | Deliberately designed for human + agent + system with documented trade-offs |\n| **Stdout contract** | P1 | Prose or mixed output | JSON sometimes, not always | Always parseable JSON with stable envelope when stdout is not a TTY (agent context); human-readable by default only under TTY or explicit `--format table` |\n| **Stderr separation** | P1 | Diagnostics mixed into stdout | Some separation | Diagnostics always on stderr |\n| **Exit code semantics** | P1/P2 | All errors map to same code | Some codes defined | Documented, stable, distinct codes per failure class |\n| **Self-description (help)** | P3 | No `--help` or single flat page | Layered help exists but incomplete | Full progressive help: top → resource → action → schema |\n| **Schema introspection** | P6 | Not available | Partial or undocumented | `tool schema <resource.action>` returns full typed schema |\n| **Dry-run** | P3/P4 | Not available | Available for some commands | Available for all mutating commands |\n| **Idempotent retries** | P1 | Mutating commands have no idempotency story; retries create duplicates | `--idempotency-key` exists on some commands, or retry semantics are inconsistent | Every mutating command accepts `--idempotency-key`; retried calls return the original result; the `retryable` flag in the error envelope is meaningful and correct |\n| **Non-interactive operation** | P0 | CLI prompts on confirmation or password input regardless of TTY state; no `--yes`/`--force` flags | Some commands support `--yes` but TTY detection is incomplete or pagers still block | CLI never prompts when stdin is not a TTY; `--yes` / `--no-input` supported on every confirmation; pagers disabled when stdout is not a TTY; structured `confirmation_required` error returned instead of blocking |\n| **Safety tiers** | P4 | Destructive ops at same level as reads | Some warning on destructive ops | Read/write/destructive clearly tiered; destructive hidden from skills |\n| **Boundary validation** | P5 | Validation scattered across internal functions, or missing | Boundary validation exists but internal code still re-validates or accepts raw input | All input validated once at the CLI entry point; internal code operates on typed, trusted structures; validators are centralized and tested for pass and reject cases |\n| **Auth delegation** | P7 | Agent manages token lifecycle, runs login or token-retrieval subcommands | Token via env var but refreshed by the agent | Human/system manages token acquisition and refresh; agent receives a pre-fetched credential and never invokes the auth retrieval path |\n| **Error recoverability** | P1 | No error fields | `code` + `message` only | `code` + `message` + `retryable` + context fields |\n| **Trust boundary** | P2 | CLI args used for auth/config | Mixed | Env vars / config set by human; agent supplies only runtime params |\n\n**Scoring guide (max 28):**\n\n- 26–28: Agent-native\n- 17–25: Partially agent-native — specific gaps, actionable fixes\n- 0–16: Not yet agent-native — structural redesign needed\n\nFile v1.3.5:references/testing.md\n\n# Testing an Agent-Native CLI\n\nA CLI's \"agent-native\" claim is empirically falsifiable. Every contract this skill teaches — envelope shape, exit code semantics, idempotency, TTY behavior, dry-run safety, schema stability — is something an agent will discover at runtime. The cheaper, kinder discovery path is a CI suite that finds the breakage first.\n\nThis file mirrors `design-patterns.md` topic by topic. Each section: the rule, how to verify it, a minimal harness, and what a passing-into-broken regression looks like. Examples use bash because CLIs are language-agnostic and bash is the universal harness; the same checks port to `bats`, `pytest` + `subprocess`, or whatever your repo already has.\n\n---\n\n## Envelope contracts (P1, P6)\n\n**Rule.** Every response on stdout is valid JSON in the documented envelope shape (`ok: true|false|\"partial\"`, `data` or `error`, optional `meta`).\n\n**How to verify.** Parse stdout as JSON; validate against a JSON Schema kept in the repo; snapshot a representative success and failure per command.\n\n```bash\n# Smoke: every documented command must produce parseable JSON.\nfor cmd in $(tool _internal list-commands); do\n  out=$(eval \"$cmd --dry-run\" 2>/dev/null) || true\n  echo \"$out\" | jq -e 'has(\"ok\")' > /dev/null \\\n    || { echo \"FAIL: $cmd produced non-JSON or missing 'ok'\"; exit 1; }\ndone\n\n# Schema: validate each envelope against the published contract.\necho \"$out\" | jq . | ajv validate -s schemas/envelope.schema.json -d -\n```\n\n**Regression looks like:** a release that adds a new field at the top level instead of inside `data` or `meta`; a command that returns a bare array instead of an envelope; pretty-print whitespace creeping in and breaking byte-equal snapshots.\n\n---\n\n## Stdout/stderr separation (P1)\n\n**Rule.** Stdout carries the JSON envelope only. Stderr carries human diagnostics or NDJSON progress events — never the final envelope.\n\n**How to verify.** Capture both streams independently, then assert stdout parses as exactly one JSON document and stderr does not.\n\n```bash\nout=$(mktemp); err=$(mktemp)\ntool sleep list --start-date 2026-01-01 --end-date 2026-01-07 >\"$out\" 2>\"$err\"\n\njq -e . \"$out\" > /dev/null || { echo \"FAIL: stdout is not a single JSON doc\"; exit 1; }\nif jq -e 'has(\"ok\")' \"$err\" 2>/dev/null; then\n  echo \"FAIL: envelope leaked into stderr\"; exit 1\nfi\n```\n\n**Regression looks like:** a `printf \"fetching...\"` left in by a refactor; a debug `console.log` going to stdout instead of stderr; progress text appearing inline with the result envelope.\n\n---\n\n## Exit code semantics (P1/P2)\n\n**Rule.** Each failure class maps to a stable, documented exit code. Success is `0`; one code per failure family thereafter.\n\n**How to verify.** Synthesize each failure class deliberately and assert the code. Keep the assertions parameterized so adding a new class is one line.\n\n```bash\ndeclare -A cases=(\n  [success]=\"tool sleep list --start-date 2026-01-01 --end-date 2026-01-07|0\"\n  [validation]=\"tool sleep list --start-date not-a-date|3\"\n  [auth]=\"HEALTHKIT_TOKEN=invalid tool sleep list --start-date 2026-01-01 --end-date 2026-01-07|2\"\n  [runtime]=\"HEALTHKIT_BASE_URL=http://127.0.0.1:1 tool sleep list --start-date 2026-01-01 --end-date 2026-01-07|1\"\n)\nfor name in \"${!cases[@]}\"; do\n  IFS='|' read -r cmd want <<< \"${cases[$name]}\"\n  eval \"$cmd\" >/dev/null 2>&1; got=$?\n  [[ \"$got\" == \"$want\" ]] || { echo \"FAIL: $name expected $want, got $got\"; exit 1; }\ndone\n```\n\n**Regression looks like:** a refactor that collapses validation and runtime errors into the same `1`; a try/catch that swallows the auth code and re-emits it as `1`.\n\n---\n\n## Idempotency replay (P1)\n\n**Rule.** A mutating command invoked twice with the same `--idempotency-key` returns the same envelope, and the side effect happens only once.\n\n**How to verify.** Run twice, diff the bytes, then check the backing store. Both must be identical / unchanged on the second call.\n\n```bash\nkey=\"test-$(uuidgen)\"\nr1=$(tool alerts send --recipient u1 --message hi --idempotency-key \"$key\")\nr2=$(tool alerts send --recipient u1 --message hi --idempotency-key \"$key\")\n\ndiff <(echo \"$r1\") <(echo \"$r2\") || { echo \"FAIL: replay returned different envelope\"; exit 1; }\n\ncount=$(tool _internal alert-rows-for-key \"$key\")\n[[ \"$count\" == \"1\" ]] || { echo \"FAIL: side-effect ran $count times, expected 1\"; exit 1; }\n```\n\n**Regression looks like:** the second call returns a fresh `id` / `created_at`; a duplicate row appears in the audit table; the `next` slot in the original response no longer matches on replay.\n\n---\n\n## Non-interactive / TTY behavior (P0)\n\n**Rule.** When stdin is not a TTY the CLI never prompts; when stdout is not a TTY it never invokes a pager. Confirmations return a structured `confirmation_required` error instead of blocking.\n\n**How to verify.** Run under a closed stdin with a wall-clock timeout. The process must exit cleanly within the timeout and produce a structured envelope, not a hang.\n\n```bash\n# Stdin not a TTY: must not block.\nout=$(echo | timeout 5 tool danger purge --id abc)\ngot=$?\n[[ \"$got\" != \"124\" ]] || { echo \"FAIL: command hung (timed out)\"; exit 1; }\necho \"$out\" | jq -e '.error.code == \"confirmation_required\"' > /dev/null \\\n  || { echo \"FAIL: expected confirmation_required envelope, got: $out\"; exit 1; }\n\n# --yes path: same command must succeed without prompting.\necho | timeout 5 tool danger purge --id abc --yes >/dev/null \\\n  || { echo \"FAIL: --yes did not bypass the prompt\"; exit 1; }\n\n# Stdout not a TTY: pager must not be invoked.\nPAGER='/bin/false' tool sleep list --start-date 2026-01-01 --end-date 2026-01-07 \\\n  | jq -e . > /dev/null || { echo \"FAIL: pager invoked under non-TTY stdout\"; exit 1; }\n```\n\n**Regression looks like:** an interactive `read -p` slipped in for a new flag; a library upgrade that re-enables pager autodetect; a `Press Enter to continue` left in a long-running command.\n\n---\n\n## Self-description and schema drift (P3, P6)\n\n**Rule.** `--help` is small enough to live in an agent's context (a few hundred tokens at the top, growing only as the agent narrows its target). The `schema` subcommand returns a typed contract that includes `schema_version`. Cached agents detect drift through the version, not by hitting a removed method.\n\n**How to verify.** Snapshot help-output byte size as an upper bound; parse `schema <method>` and assert it includes the documented fields; on every release, diff the schema snapshot and fail loudly on incompatible changes.\n\n```bash\n# Token budget on top-level help (rough proxy: bytes / 4 ~ tokens).\nlimit=2000\nsize=$(tool --help | wc -c)\n[[ \"$size\" -le \"$limit\" ]] || { echo \"FAIL: top-level --help is $size bytes, limit $limit\"; exit 1; }\n\n# Schema endpoint is real and includes versioning.\ntool schema sleep.list | jq -e '.schema_version and .params and .introduced_in' > /dev/null \\\n  || { echo \"FAIL: schema response missing required fields\"; exit 1; }\n\n# Schema regression gate: snapshot diff blocks accidental breaking changes.\ntool schema sleep.list > /tmp/schema.new.json\ndiff -u tests/snapshots/schema.sleep.list.json /tmp/schema.new.json \\\n  || { echo \"REVIEW: schema changed — confirm version bump and deprecation signals\"; exit 1; }\n```\n\n**Regression looks like:** a help generator that starts emitting the full schema in `--help`; a schema response that drops `schema_version` after a refactor; a field renamed between versions without a `deprecated` / `replaced_by` entry on the old name.\n\n---\n\n## Dry-run safety (P3, P4)\n\n**Rule.** Every mutating command supports `--dry-run`. The dry-run output describes the would-be request. No side effect occurs.\n\n**How to verify.** Drive every mutating command in dry-run mode, assert the envelope shape, then assert the backing store is unchanged.\n\n```bash\nbefore=$(tool _internal store-fingerprint)\n\nfor cmd in $(tool _internal list-mutating-commands); do\n  out=$(eval \"$cmd --dry-run\")\n  echo \"$out\" | jq -e '.dry_run == true and .would_request' > /dev/null \\\n    || { echo \"FAIL: $cmd --dry-run is missing dry_run/would_request\"; exit 1; }\ndone\n\nafter=$(tool _internal store-fingerprint)\n[[ \"$before\" == \"$after\" ]] || { echo \"FAIL: dry-run mutated state\"; exit 1; }\n```\n\n**Regression looks like:** a new mutating command shipped without `--dry-run` wired up; a dry-run path that still issues the upstream request; an envelope that drops the `dry_run: true` flag.\n\n---\n\n## Auth delegation (P7)\n\n**Rule.** The agent never invokes the auth lifecycle (`auth login`, `auth token`, browser flows). It receives a credential from a human-managed env var and consumes it.\n\n**How to verify.** Grep the generated agent-facing artifacts (skills, MCP tool manifests, examples) for the auth subcommands. Any reference is a violation.\n\n```bash\nforbidden='auth login|auth token|auth refresh'\n# Grep generated agent-facing artifacts only (skill manifests, MCP tool defs, codex sidecars).\n# Human-facing docs and examples that document the human bootstrap path\n# (e.g. examples.md showing `healthkit auth login`) are expected and excluded here.\nhits=$(grep -RInE \"$forbidden\" agents/ 2>/dev/null | grep -v '^//' || true)\n[[ -z \"$hits\" ]] || { echo \"FAIL: agent surface references auth lifecycle:\"; echo \"$hits\"; exit 1; }\n```\n\n**Regression looks like:** a worked example in the skill that pipes `auth token` into the next command; a generated MCP tool that exposes `auth.login`; an \"easy onboarding\" addition that has the agent call `auth login` itself.\n\n---\n\n## Locale and time determinism\n\n**Rule.** Output bytes are identical regardless of host locale or timezone. Timestamps are UTC ISO-8601, dates are ISO-8601, numbers use `.` and no thousands separator.\n\n**How to verify.** Run the same invocation under two divergent locales and one shifted timezone. The captured bytes must match exactly.\n\n```bash\nfixed=(--start-date 2026-01-01 --end-date 2026-01-07)\n\na=$(LC_ALL=C       TZ=UTC          tool sleep list \"${fixed[@]}\")\nb=$(LC_ALL=de_DE.UTF-8 TZ=Asia/Tokyo tool sleep list \"${fixed[@]}\")\n\ndiff <(echo \"$a\") <(echo \"$b\") || { echo \"FAIL: output drifted with locale/TZ\"; exit 1; }\n```\n\n**Regression looks like:** `1.234,56` appearing instead of `1234.56`; a timestamp printed as `Mo., 12. Jan. 2026` instead of `2026-01-12T00:00:00Z`; a sort order that rearranges itself when the host's `LANG` changes.\n\n---\n\n## Long-running output and streaming\n\n**Rule.** Long-running commands emit structured progress on stderr (one JSON object per line) and a single envelope on stdout. NDJSON list commands emit one envelope per line plus a final summary.\n\n**How to verify.** Run with stderr redirected to a file, assert each line parses as JSON with an `event` field; for `--stream`, assert the last line carries a `summary`.\n\n```bash\ntool export run --dataset sleep --since 2024-01-01 > /tmp/result.json 2> /tmp/progress.log\njq -e . /tmp/result.json > /dev/null || { echo \"FAIL: stdout not a single envelope\"; exit 1; }\nwhile IFS= read -r line; do\n  echo \"$line\" | jq -e 'has(\"event\")' > /dev/null \\\n    || { echo \"FAIL: stderr line is not structured: $line\"; exit 1; }\ndone < /tmp/progress.log\n\ntool sleep list --since 2024-01-01 --stream | tail -1 \\\n  | jq -e '.summary' > /dev/null || { echo \"FAIL: stream missing final summary\"; exit 1; }\n```\n\n**Regression looks like:** progress events going to stdout and breaking the redirect-to-file pattern; an extra blank line at the end of `--stream` that breaks `tail -1`; a refactor that turns the per-line JSON into a single big array at the end.\n\n---\n\n## Wiring it into CI\n\nThese checks are cheap — most are seconds — and they belong on every PR. Three things make them durable:\n\n1. **Treat the schema snapshot like a public ABI.** A diff is a release decision, not a code-review nitpick. Pair it with a CHANGELOG entry and a version bump.\n2. **Run the locale and TTY tests in their own job.** They depend on environment shape, not on the build, and they fail fast when a maintainer's local laptop is the only thing the CLI was ever tested under.\n3. **Keep the failure messages structured.** A test suite for an agent-native CLI is itself an agent surface — the messages above are written so an LLM driving CI can read them and route the fix without grepping logs.\n\nFile v1.3.5:skill-card.md\n\n## Description:\n\nUse when designing, reviewing, or refactoring a CLI that must serve AI agents alongside humans, or when converting an API or SDK into an agent-usable CLI interface.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[agents365-ai](https://clawhub.ai/user/agents365-ai)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and engineers use this skill to evaluate, design, and refactor command-line interfaces for reliable use by humans, AI agents, and orchestration systems. It supports reviews of output contracts, schema introspection, dry-run behavior, safety boundaries, delegated authentication, and CLI-to-agent integration patterns.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Testing guidance includes eval-based shell harness examples that could execute command strings supplied by the CLI under review.\n\nMitigation: Treat the snippets as design guidance, replace eval loops with structured command manifests or argv arrays, and run CLI review tests in a sandbox with disposable credentials and non-production data.\n\nRisk: The security verdict is suspicious because unsafe shell examples appear in a reference file.\n\nMitigation: Review generated or copied test harnesses before execution and scan the skill before deployment.\n\n## Reference(s):\n\n- [ClawHub Skill Page](https://clawhub.ai/agents365-ai/skills/agent-native-design)\n- [Design Patterns](references/design-patterns.md)\n- [Rubric](references/rubric.md)\n- [Review Checklists](references/checklists.md)\n- [Examples and Non-Examples](references/examples.md)\n- [Testing an Agent-Native CLI](references/testing.md)\n- [When CLI vs MCP vs Both](references/hybrid-mcp-cli.md)\n- [Citations](references/citations.md)\n- [Anthropic: Code execution with MCP](https://www.anthropic.com/engineering/code-execution-with-mcp)\n- [Anthropic: Beyond permission prompts](https://www.anthropic.com/engineering/claude-code-sandboxing)\n- [Writing CLI Tools That AI Agents Actually Want to Use](https://dev.to/uenyioha/writing-cli-tools-that-ai-agents-actually-want-to-use-39no)\n- [Building a CLI That Works for Humans and Machines](https://www.openstatus.dev/blog/building-cli-for-human-and-agents)\n- [MCP vs CLI: Benchmarking Tools for Coding Agents](https://mariozechner.at/posts/2025-08-15-mcp-vs-cli/)\n- [CLI Tools vs MCP: Better AI Agents With Less Context](https://jannikreinhard.com/2026/02/22/why-cli-tools-are-beating-mcp-for-ai-agents/)\n\n## Skill Output:\n\n**Output Type(s):** [Analysis, Guidance, Markdown, Code, Shell commands, Configuration]\n\n**Output Format:** [Markdown with structured review sections, examples, and inline code or shell command blocks]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May include rubric scores, risk summaries, prioritized refactor plans, JSON contract examples, and CLI design checklists.]\n\n## Skill Version(s):\n\n1.3.5 (source: server release evidence and skill frontmatter metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.3.5:agents/openai.yaml\n\ninterface:\n  display_name: \"Agent Native Design\"\n  short_description: \"Design, review, and refactor CLIs to serve humans, AI agents, and orchestration systems simultaneously — structured output, schema introspection, dry-run, safety tiers, and delegated authentication\"\n  brand_color: \"#1A3A5C\"\n\npolicy:\n  allow_implicit_invocation: true\n\ncapabilities:\n  - Evaluate whether a CLI is agent-native across 7 design principles\n  - Design stdout JSON contracts with stable error envelopes\n  - Define exit code semantics per failure class (success, runtime, auth, validation)\n  - Design layered --help, schema introspection, and dry-run previews\n  - Define safety tiers: open (read), warned (write), hidden (destructive)\n  - Design delegated authentication — human owns auth lifecycle, agent receives token\n  - Apply directional trust model — env vars trusted, CLI args untrusted\n  - Convert a REST API or SDK into an agent-native CLI command tree\n  - Score a CLI on the 14-criterion rubric and summarize the seven design principles\n  - Produce a prioritized P0/P1/P2 refactor plan with concrete examples\n\nprerequisites: []\n\nArchive v1.3.4: 12 files, 31719 bytes\n\nFiles: agents/openai.yaml (1124b), references/checklists.md (3696b), references/citations.md (3748b), references/design-patterns.md (12673b), references/examples.md (8337b), references/hybrid-mcp-cli.md (3986b), references/rubric.md (3566b), references/testing.md (12206b), scripts/validate-metadata.sh (2722b), skill-card.md (2949b), SKILL.md (12880b), _meta.json (138b)\n\nFile v1.3.4:SKILL.md\n\n---\nname: agent-native-design\ndescription: Use when designing, reviewing, or refactoring a CLI that must serve AI agents alongside humans, or when converting an API or SDK into an agent-usable CLI interface.\nlicense: CC-BY-NC-4.0\nhomepage: https://github.com/Agents365-ai/agent-native-design\ncompatibility: Includes sidecar metadata for OpenClaw, Hermes, pi-mono, and OpenAI Codex; the core SKILL.md is portable to any agent runtime that supports Agent Skills-style instructions.\nplatforms: [macos, linux, windows]\nmetadata: {\"openclaw\":{\"requires\":{},\"emoji\":\"⌨️\",\"os\":[\"darwin\",\"linux\",\"win32\"]},\"hermes\":{\"tags\":[\"cli\",\"agent-native\",\"interface-design\",\"structured-output\",\"schema-driven\"],\"category\":\"engineering\",\"requires_tools\":[],\"related_skills\":[]},\"pimo\":{\"category\":\"engineering\",\"tags\":[\"cli\",\"agent-native\",\"interface-design\",\"structured-output\",\"schema-driven\"]},\"author\":\"Agents365-ai\",\"version\":\"1.3.5\"}\n---\n\n# agent-native-design\n\n## Purpose\n\nThis skill helps analyze, design, and refactor command-line tools so they can reliably serve **humans**, **AI agents**, and **orchestration systems** at the same time.\n\nIt is not a skill for merely *using* a CLI. It is a skill for designing and reviewing a CLI as an **agent-native interface**.\n\nThe skill focuses on four goals:\n\n1. Make CLI behavior predictable for AI agents.\n2. Make CLI output readable and recoverable for humans.\n3. Make CLI execution manageable for systems and orchestrators.\n4. Define a complete interaction loop from authentication to error routing.\n\n---\n\n## When to use this skill\n\nUse this skill when the user wants to:\n\n* evaluate whether an existing CLI is agent-friendly\n* redesign a CLI to better support AI agents\n* convert an API or SDK into an agent-native CLI\n* review help output, schema design, exit codes, or JSON contracts\n* design dry-run, auth delegation, or safety boundaries\n* generate CLI skills, docs, or interface conventions from schema\n* refactor a human-oriented CLI into a machine-friendly one\n* define how a CLI should interact with an agent runtime\n\nTypical prompts include:\n\n* \"Review this CLI and tell me whether it is agent-native.\"\n* \"Design a CLI for this API that an AI agent can use reliably.\"\n* \"Refactor this tool so stdout is machine-readable and safer for agents.\"\n* \"Help me define schema introspection, dry-run, and exit code semantics.\"\n\n## When not to use this skill\n\nDo not use this skill when the user only wants:\n\n* help running a specific command\n* installation help for a CLI\n* shell troubleshooting unrelated to interface design\n* generic Linux or terminal tutorials\n* agent planning or memory design unrelated to tools\n* API business logic review without any CLI/tooling layer\n\n---\n\n## Core model\n\nAn agent-native CLI must simultaneously serve three audiences.\n\n### 1. Human\n\nNeeds: readable output, friendly error messages, onboarding guidance.\nChannels: `stderr`, optional `--format table`, interactive TUI when appropriate.\n\n### 2. AI Agent\n\nNeeds: structured data, stable contracts, self-description.\nChannels: `stdout` as JSON, stable exit codes, schema introspection, dry-run previews, generated skills/docs.\n\n### 3. System / Orchestrator\n\nNeeds: delegated authentication, process management, deterministic error routing.\nChannels: environment variables, exit codes, dry-run mode, stable command semantics.\n\n### Foundational contract\n\n| Channel | Primary audience |\n|---------|-----------------|\n| `stdout` | Machines and agents |\n| `stderr` | Humans |\n| `exit codes` | Systems and orchestrators |\n\nThis skill teaches how to make CLI a first-class interface for agents. Production agents (Claude Code, Cursor, Gemini CLI) often pair a CLI with an MCP server — CLI for state changes and local/scriptable work, MCP for multi-tenant SaaS and per-user auth. When a design needs the MCP side as well, see `references/hybrid-mcp-cli.md` for the decision matrix and the benchmark data behind the CLI/MCP tradeoff.\n\n---\n\n## The complete interaction loop\n\n| Phase | Step | Description |\n|-------|------|-------------|\n| 0. Bootstrap | 1 | Human/system obtains auth token or credentials |\n| 0. Bootstrap | 2 | Set trusted env vars: token, profile, safety mode |\n| 1. Discovery | 3 | Agent loads skills or command summaries |\n| 1. Discovery | 4 | Agent queries schema/help for parameters |\n| 2. Planning | 5 | Agent uses `--dry-run` to preview request shape |\n| 3. Execution | 6 | Agent executes with validated inputs |\n| 4. Interpretation | 7 | Agent parses structured result |\n| 5. Recovery | 8 | Agent uses exit code + error object to retry, re-auth, repair, or escalate |\n\nA CLI that does not support every phase is incomplete from the agent's perspective.\n\n---\n\n## Seven principles\n\nThese are load-bearing. Each principle has at least one rubric criterion and at least one example backing it.\n\n### Principle 0. One CLI, Three Audiences\n\nThe CLI must serve human, agent, and system simultaneously. A design that serves only one audience is incomplete.\n\n### Principle 1. Structured Output Is the Interface\n\n`stdout` should always be parseable and stable. Both success and failure are structured JSON. The CLI must decide for itself which audience is reading: detect at startup whether stdout is a TTY, default to JSON when it is not, default to human-readable when it is. `NO_COLOR` and an explicit `--format json|table` flag override the auto-detection. Agents should never have to remember to pass `--format json` — if they have to, they will forget, and the run will silently produce un-parseable prose. Envelope and error contract: `references/design-patterns.md#output-envelopes`.\n\n### Principle 2. Trust Is Directional\n\nCLI arguments are not inherently trusted — they may come from a hallucinating or prompt-injected agent. Environment-level configuration set by the human or system is more trusted. The agent chooses *what to do* within a bounded surface; the human defines *where and how it is allowed to operate*.\n\n### Principle 3. The CLI Must Describe Itself\n\nThe CLI must be self-describing enough that an agent can use it without reading external README files. Self-description must be **progressive**, not eager: top-level `--help` lists resources; resource help lists actions; action help lists flags; a separate `schema <resource.action>` returns the full typed schema. A CLI with hundreds of commands that dumps everything into the first `--help` pays that token cost on every agent invocation. See `references/design-patterns.md#help-design` and `references/examples.md` Examples 2 and 5.\n\n### Principle 4. Safety Through Graduated Visibility\n\nRead commands are easy to discover; mutating commands carry explicit warnings; destructive commands are hidden from skills or gated separately. Tier table and rationale: `references/design-patterns.md#safety-design`. Tiers are necessary but not sufficient — they are a prompt-side defense and approval fatigue degrades them quickly. Assume the agent runtime will additionally sandbox the CLI at the OS level (filesystem, network, processes), and design destructive commands to fail closed inside that sandbox.\n\n### Principle 5. Validate at the Boundary, Not in the Middle\n\nInputs are validated once at the CLI entry point. Internal code operates on validated, typed, trusted structures. Validation functions are centralized and tested for both pass and reject cases.\n\n### Principle 6. The Schema Is the Source of Truth\n\nIf a schema exists, everything derives from it: CLI command structure, validation rules, help text, generated docs, generated skills, type definitions, dry-run contracts. The schema is never manually duplicated. The schema must also carry its own version and deprecation signals, surfaced in the `meta` block of every response, so agents that have cached an older view can detect drift and re-discover rather than silently calling a removed method. Full versioning contract and example: `references/design-patterns.md#schema-versioning`.\n\n### Principle 7. Authentication Must Be Delegatable\n\nAuthentication is obtained and refreshed by human/system-managed flows. The agent uses credentials; it never owns the auth lifecycle. Preferred mechanisms: environment variables, config files, OS keychain integration, externally refreshed tokens. Canonical pattern: `references/examples.md` Example 3.\n\n---\n\n## Standard review workflow\n\n### Step 1. Classify the input\n\nDecide whether the user is providing: an existing CLI, an API to be wrapped, a conceptual design, a partial interface, or a failure case.\n\n### Step 2. Map the three audiences\n\n**Human:** Is there readable output? Are errors understandable? Is onboarding supported?\n\n**Agent:** Is stdout stable JSON? Can the CLI describe itself? Is there schema introspection and dry-run?\n\n**System:** Is auth delegatable? Are exit codes stable? Can failures be routed deterministically?\n\n### Step 3. Review the interaction loop\n\nCheck whether the CLI supports: bootstrap, discovery, parameter understanding, preview, execution, parsing, recovery.\n\n### Step 4. Score the CLI with the rubric, then map back to principles\n\nUse the 14-criterion rubric to score the CLI. The full rubric lives in `references/rubric.md`. Every one of the seven principles has at least one rubric row backing it, so the score-to-principle mapping is total: P0 → Three-audience support, Non-interactive operation; P1 → Stdout contract, Stderr separation, Idempotent retries, Error recoverability; P2 → Trust boundary; P3 → Self-description (help), Dry-run; P4 → Safety tiers; P5 → Boundary validation; P6 → Schema introspection; P7 → Auth delegation. Then summarize per principle with evidence, risk, and recommendation. The full review checklists live in `references/checklists.md`.\n\n### Step 5. Produce a refactor plan\n\n- **P0** must fix\n- **P1** should improve\n- **P2** long-term enhancements\n\n---\n\n## Default output format\n\n### 1. Overall verdict\n\nState whether the CLI is **agent-native**, **partially agent-native**, or **not yet agent-native**.\n\n### 2. Three-audience contract review\n\nAssess support for human, agent, system.\n\n### 3. Interaction loop coverage\n\nAssess each phase: auth bootstrap → env setup → skill/help discovery → schema introspection → dry-run → execution → parsing and recovery.\n\n### 4. Rubric score + seven-principle review\n\nReport the 14-criterion rubric score first, then summarize the seven principles as: status · evidence · issue · recommendation.\n\n### 5. Key risks\n\nSummarize design failures: human-only output, unstable JSON, no schema introspection, destructive commands overexposed, auth coupled to agent, ambiguous exit codes.\n\n### 6. Refactor plan\n\nPrioritized recommendations with examples drawn from `references/examples.md`.\n\n---\n\n## Things this skill should avoid recommending\n\n* Human-readable prose as the only output contract\n* README required for basic command discovery\n* Schema and validation that drift apart\n* Auth supplied primarily via agent-generated arguments\n* Destructive actions exposed by default\n* CLI behavior that depends on undocumented conventions\n* Errors that are only textual and not machine-routable\n* Mutating commands that are not idempotent under retry\n* Confirmation prompts with no `--yes` escape and no TTY-aware fallback\n* Eager schema dumps in top-level `--help` — agents that call the CLI in loops pay this cost on every invocation; use progressive disclosure instead. The token-cost rationale lives in `references/hybrid-mcp-cli.md`.\n\n---\n\n## Reference files\n\nLoad on demand — these are not in the agent's context until needed:\n\n| File | Read when |\n|------|-----------|\n| `references/examples.md` | Showing the user a good envelope, error, dry-run, batch response, or anti-pattern |\n| `references/rubric.md` | Producing the score component of a CLI review |\n| `references/checklists.md` | Walking through a CLI auditing list with the user, or sanity-checking a new design |\n| `references/design-patterns.md` | Writing the contract for envelopes, exit codes, idempotency, non-interactive mode, long-running commands, schema versioning, locale/time |\n| `references/hybrid-mcp-cli.md` | Deciding CLI vs. MCP vs. both, or citing benchmark numbers behind the CLI efficiency claim |\n| `references/testing.md` | Showing the user how to verify their CLI actually upholds the contract (envelope shape, idempotency replay, TTY behavior, schema drift, dry-run safety, locale determinism) — load this when the design review converges on \"how do we keep it agent-native over time?\" |\n| `references/citations.md` | Citing the primary sources behind a recommendation |\n\n---\n\n## One-sentence summary\n\nThis skill helps turn a CLI into a trustworthy execution interface for **humans, AI agents, and systems** through **structured output, self-description, delegated authentication, safety boundaries, and a complete interaction loop**.\n\nFile v1.3.4:_meta.json\n\n{\n  \"ownerId\": \"kn74y2dmpaszt959h16p11k53983ht10\",\n  \"slug\": \"agent-native-design\",\n  \"version\": \"1.3.4\",\n  \"publishedAt\": 1783532096325\n}\n\nFile v1.3.4:references/checklists.md\n\n# Review Checklists\n\nUse this when evaluating a CLI for agent readiness, or when sanity-checking a new design before shipping.\n\n## Output\n\n- [ ] `stdout` is valid JSON when stdout is not a TTY or when `--format json` is passed (success and failure)\n- [ ] `stderr` carries human-readable diagnostics only\n- [ ] JSON envelope is stable: `{ \"ok\": bool, \"data\": ... }` or `{ \"ok\": false, \"error\": ... }`\n- [ ] Error object includes: `code`, `message`, `retryable`\n- [ ] No prose mixed into `stdout`\n\n## Exit codes\n\n- [ ] Exit codes are documented\n- [ ] Exit codes are stable across versions\n- [ ] Distinct codes for: success (0), runtime error, auth error, validation error\n- [ ] Exit code mapping is available via `--help` or `schema`\n\n## Retry and interaction mode\n\n- [ ] Every mutating command accepts `--idempotency-key`\n- [ ] Retried calls with the same idempotency key return the original result\n- [ ] `retryable` field in error envelope is meaningful and correct\n- [ ] CLI never prompts for input when stdin is not a TTY\n- [ ] `--yes` / `--no-input` / `--force` supported on every command that would otherwise prompt\n- [ ] Returns structured `confirmation_required` error instead of blocking on missing confirmation\n- [ ] stdout defaults to JSON when stdout is not a TTY (no `--format json` required)\n- [ ] Pagers (`less`, `more`) disabled when stdout is not a TTY\n\n## Self-description\n\n- [ ] Top-level `--help` lists all resources/commands\n- [ ] Resource-level `--help` lists actions\n- [ ] Action-level `--help` lists all flags with types\n- [ ] Schema introspection command available (`tool schema <resource.action>`)\n- [ ] Dry-run available for all mutating commands\n\n## Safety\n\n- [ ] Read commands clearly discoverable\n- [ ] Write/mutating commands carry explicit warning in help\n- [ ] Destructive commands (delete/purge) hidden from skills or gated\n- [ ] Dry-run covers all write operations\n\n## Auth\n\n- [ ] Human/system manages token acquisition (browser flow, keychain)\n- [ ] Agent receives credential via env var or pre-fetched token\n- [ ] Agent never navigates OAuth2 or browser flows\n- [ ] Token refresh handled outside agent runtime\n\n## Trust\n\n- [ ] CLI args treated as untrusted (validated at boundary)\n- [ ] Environment variables used for config/safety settings (human-set)\n- [ ] Agent cannot escalate its own privileges via CLI args\n\n## Schema\n\n- [ ] Schema is the single source of truth\n- [ ] CLI command structure derives from schema\n- [ ] Validation derives from schema\n- [ ] Help text derives from schema\n- [ ] Generated skills derive from schema (if applicable)\n- [ ] Schema version included in every response's `meta` block\n- [ ] Deprecation signals included in schema responses (`deprecated_fields`, `replaced_by`, `removed_in`)\n- [ ] Schema introspection is incremental (not eager): `--help` is small; full schema via `schema` subcommand only\n\n## Token efficiency\n\nFor agents that call the CLI in loops (orchestration, multi-turn planning), context cost matters. These items help CLIs realize the per-call token advantage over eager-loaded MCP servers (see `hybrid-mcp-cli.md` for the benchmark data this is based on).\n\n- [ ] Top-level `--help` response is under 500 tokens\n- [ ] Full schema is not dumped in top-level `--help`; accessed via `schema <resource.action>` instead\n- [ ] Field selection supported on list responses (`--json field1,field2,...` or similar)\n- [ ] Default list responses are compact (3–5 fields); full detail via `--full` flag\n- [ ] Requests that would normally require two CLI calls are collapsed (e.g., `count` + `list` → return count in the envelope)\n- [ ] Schema versioning allows agents to cache and avoid re-discovery on every invocation\n\nFile v1.3.4:references/citations.md\n\n# Citations\n\nSources actually quoted or directly relied on in SKILL.md and the other reference files. The broader landscape (clig.dev, the gh / aws / kubectl design corpus) is implicit background.\n\n## Anthropic engineering\n\n- *Code execution with MCP: Building more efficient agents* — https://www.anthropic.com/engineering/code-execution-with-mcp (Nov 4, 2025). Progressive disclosure of tool definitions; the 150K → 2K token reduction case study cited under Principle 3 / `design-patterns.md#help-design`.\n- *Beyond permission prompts: making Claude Code more secure and autonomous* — https://www.anthropic.com/engineering/claude-code-sandboxing (Oct 20, 2025). Approval fatigue and the 84% prompt-reduction figure cited under Principle 4 / `design-patterns.md#safety-design`.\n\n## CLI-for-agents writing\n\n- Ugo Enyioha, *Writing CLI Tools That AI Agents Actually Want to Use* — https://dev.to/uenyioha/writing-cli-tools-that-ai-agents-actually-want-to-use-39no (Feb 27, 2025). Idempotency-on-retry and \"agents cannot type 'y'\" framings cited in the Idempotency and Non-interactive sections of `design-patterns.md`.\n- Thibault Le Ouay Ducasse / openstatus, *Building a CLI That Works for Humans and Machines* — https://www.openstatus.dev/blog/building-cli-for-human-and-agents (Apr 2, 2026). TTY detection as the human/machine switch cited under Principle 1.\n- Mario Zechner, *MCP vs CLI: Benchmarking Tools for Coding Agents* — https://mariozechner.at/posts/2025-08-15-mcp-vs-cli/ (Aug 15, 2025). Empirical case that many MCP servers could be CLI invocations.\n- Armin Ronacher, *Skills vs Dynamic MCP Loadouts* — https://lucumr.pocoo.org/2025/12/13/skills-vs-mcp/ (Dec 13, 2025). Schema/API stability as a first-class concern cited under Principle 6.\n\n## CLI-vs-MCP benchmarks (2026)\n\n- Jannik Reinhard, *CLI Tools vs MCP: Better AI Agents With Less Context* — https://jannikreinhard.com/2026/02/22/why-cli-tools-are-beating-mcp-for-ai-agents/ (Feb 22, 2026). Source for the 28% / 33% / 55K / 35× numbers in `hybrid-mcp-cli.md`.\n- Manveer Chawla, *MCP vs. CLI for AI agents: When to Use Each (A Practical Decision Framework for 2026)* — https://manveerc.substack.com/p/mcp-vs-cli-ai-agents (Mar 8, 2026). Per-integration decision framework; production examples (Claude Code, Cowork) using both transports.\n- Soumyadeb Mitra / RudderStack, *CLI or MCP or both? The design pattern for AI agents managing your data stack* — https://www.rudderstack.com/blog/ai-agents-cli-mcp-design-pattern/ (Mar 18, 2026). The \"writes via CLI, reads via MCP\" split underpinning `hybrid-mcp-cli.md`.\n\n## Pre-agent baseline\n\n- *Scripting with GitHub CLI* — https://github.blog/engineering/engineering-principles/scripting-with-github-cli/ (Mar 11, 2021). `gh api --jq` and structured JSON output as a first-class CLI pattern. The post predates the `gh <resource> --json field1,field2` field-selection flag (cli/cli#1089) but lays out the design philosophy that flag is built on.\n- *Command Line Interface Guidelines* — https://clig.dev/. The pre-agent baseline for human-first CLI design. This skill extends it; it does not replace it.\n- `sysexits.h` — https://manpages.ubuntu.com/manpages/noble/man3/sysexits.h.3head.html. The BSD exit-code vocabulary that `design-patterns.md#exit-code-model` deliberately simplifies away from.\n\n---\n\n## A note on the \"agent-native CLI\" framing\n\nThe term used throughout this skill is one of several competing framings in current writing. *Agent-first CLI* (Propel, Keyboards Down) is more common; *CLI for humans and machines* (openstatus, Linearis) is the most descriptive. \"Native\" is chosen here to emphasize that agent support is a first-class design goal, not a retrofit on top of a human-only CLI.\n\nFile v1.3.4:references/design-patterns.md\n\n# Design Patterns\n\nReference details for specific design areas. SKILL.md states the principles; this file holds the concrete contracts and rules.\n\n---\n\n## Output envelopes\n\nSuccess:\n\n```json\n{ \"ok\": true, \"data\": {} }\n```\n\nFailure:\n\n```json\n{\n  \"ok\": false,\n  \"error\": {\n    \"code\": \"validation_error\",\n    \"message\": \"Missing required field: email\",\n    \"field\": \"email\",\n    \"retryable\": false\n  }\n}\n```\n\nPartial success (batch operations):\n\n```json\n{\n  \"ok\": \"partial\",\n  \"data\": {\n    \"succeeded\": [\n      { \"id\": \"msg_001\", \"status\": \"sent\" },\n      { \"id\": \"msg_002\", \"status\": \"sent\" }\n    ],\n    \"failed\": [\n      {\n        \"id\": \"msg_003\",\n        \"error\": { \"code\": \"rate_limited\", \"message\": \"Rate limit exceeded for recipient\", \"retryable\": true }\n      }\n    ]\n  }\n}\n```\n\nA batch command that collapses any per-item failure into a top-level `ok: false` forces every agent to re-process its successful items on retry. AWS SQS's `ReportBatchItemFailures` and similar APIs settled on this per-item shape for the same reason. Pair partial success with an idempotency key on the batch as a whole so that re-running with the same key only re-issues the *failed* items.\n\nOptional `meta` slot for observability:\n\n```json\n{\n  \"ok\": true,\n  \"data\": { \"id\": \"abc123\" },\n  \"meta\": {\n    \"request_id\": \"req_8fa9c1\",\n    \"latency_ms\": 412,\n    \"schema_version\": \"1.4.0\"\n  }\n}\n```\n\n`meta` is a freeform slot for telemetry the orchestrator may want without polluting `data`: `request_id` for log correlation, `latency_ms` for SLO tracking, `schema_version` so an agent can detect drift against a cached schema. When the underlying call has a token / quota cost the CLI knows about, surface it here too — `tokens_used`, `quota_remaining`. Optional and additive: agents that don't need it ignore it; agents that do gain observability without an extra round-trip. Aligns with where the OpenTelemetry GenAI semantic conventions are heading for tool-call traces.\n\n---\n\n## Schema versioning\n\nThis is the contract that lets agents cache schemas safely across calls.\n\nEvery response should carry `schema_version` in the optional `meta` block (see envelope above). When an agent's cached schema version (e.g., 1.2.0) doesn't match the CLI's current version (1.4.0), the agent knows to re-discover before re-planning.\n\nSchema introspection responses should declare which CLI version produced them, when each method was introduced, and whether any field is deprecated:\n\n```json\n{\n  \"method\": \"sleep.list\",\n  \"since\": \"1.2.0\",\n  \"deprecated\": false,\n  \"params\": {\n    \"startDate\": { \"type\": \"string\", \"format\": \"date\", \"required\": true },\n    \"endDate\":   { \"type\": \"string\", \"format\": \"date\", \"required\": true },\n    \"page_size\": { \"type\": \"integer\", \"default\": 20, \"max\": 100, \"deprecated\": true, \"replaced_by\": \"pageSize\", \"removed_in\": \"1.5.0\" }\n  }\n}\n```\n\nThis gives agents:\n\n- **Drift detection** — cached version mismatch triggers re-discovery before re-planning.\n- **Deprecation awareness** — agents migrate off `page_size` to `pageSize` proactively, before removal.\n- **Non-breaking updates** — adding a new optional field or method makes cached schemas incomplete but not incorrect; existing calls keep working.\n- **Token efficiency** — agents don't waste tokens on removed methods or obsolete fields; drift is corrected in one round-trip.\n\nAPI stability is a contract you owe the agent: a CLI that renames flags between point releases forces every dependent agent to re-discover and re-plan. Treat the schema as a versioned, append-mostly surface.\n\nSee `references/examples.md` Example 8 for the full request/response pair.\n\n---\n\n## Exit code model\n\n| Code | Meaning |\n|------|---------|\n| `0` | Success |\n| `1` | Runtime / API error |\n| `2` | Auth error |\n| `3` | Validation error |\n\nExact codes may vary — the mapping must be documented and deterministic. (`sysexits.h` defines a richer pre-agent vocabulary — `EX_USAGE=64`, `EX_DATAERR=65`, `EX_NOPERM=77`, `EX_CONFIG=78` — and a CLI is free to use it. The 0–3 mapping above is a deliberate simplification for agent routing; what matters is that codes are documented, stable, and distinct per failure class.)\n\n---\n\n## Idempotency and retry\n\nAgents retry. Networks fail, processes get killed, exit codes get misread. A CLI that is not idempotent forces every agent that uses it to write special-case retry logic — and most agents will get it wrong.\n\nEvery mutating command should accept `--idempotency-key <string>`. A retried call carrying the same key must be a safe no-op that returns the original result, in the same `ok` / `error` shape. The CLI is responsible for storing the key↔result mapping for long enough that legitimate retries can find it (typically minutes; the upper bound belongs to the underlying API).\n\nThis pairs with the `retryable` field in the error envelope: `retryable: true` tells the agent it is *safe* to call again with the same idempotency key, and that doing so will eventually converge; `retryable: false` tells the agent that retrying will not change the outcome.\n\n> \"Agents retry. Networks fail. Commands get interrupted. If your `create` command fails on the second run because the resource already exists, the agent has to write special-case retry logic.\" — Ugo Enyioha, *Writing CLI Tools That AI Agents Actually Want to Use* (Feb 2025)\n\n---\n\n## Non-interactive operation\n\nAn agent cannot answer a confirmation prompt, cannot type a password into a TTY, and cannot navigate a curses-style menu. A CLI that hangs waiting for stdin when stdin is not a TTY is, from the agent's perspective, broken.\n\nRules:\n\n- **Never prompt when stdin is not a TTY.** Detect at startup; if a confirmation would normally fire but `isatty(stdin) == false`, return a structured error instead of blocking:\n  ```json\n  { \"ok\": false, \"error\": { \"code\": \"confirmation_required\", \"message\": \"Pass --yes to confirm.\", \"retryable\": false } }\n  ```\n- **Always support `--yes` / `--no-input` / `--force`** (or equivalent) on every command that would otherwise prompt, so a human running interactively can opt out and an agent always runs without prompts.\n- **Never read secrets from interactive prompts in agent contexts.** Secrets come from environment variables or pre-set config (see Principle 7).\n- **Pagers off when stdout is not a TTY.** Detect at startup; never invoke `less` / `more` style pagers that would block on a non-TTY consumer.\n\n> \"An agent cannot type 'y' at a confirmation prompt. If your CLI hangs waiting for input, the agent's workflow is dead.\" — Ugo Enyioha (Feb 2025)\n\n---\n\n## Long-running commands and streaming\n\nA command that takes minutes to finish is a hazard for an agent: the agent does not know whether the CLI is making progress, stuck, or dead, and it cannot afford to wait blind on a single JSON envelope at the end. Two patterns work:\n\n**Structured progress on stderr, final JSON on stdout.** The agent reads stderr for liveness and stdout for the result. Progress events are themselves structured (one JSON object per line) so the orchestrator can parse them, but they never pollute the stdout JSON envelope. See `references/examples.md` Example 7 for a full transcript.\n\n**NDJSON streaming for long lists.** When a list command might return thousands of items, offer a `--stream` (or `--ndjson`) mode that emits one JSON object per line on stdout, with a final summary line. Agents can process the stream incrementally and stop when they have enough:\n\n```\n$ healthkit sleep list --since 2024-01-01 --stream\n{ \"ok\": true, \"data\": { \"id\": \"sl_001\", \"date\": \"2024-01-01\", \"minutes\": 412 } }\n{ \"ok\": true, \"data\": { \"id\": \"sl_002\", \"date\": \"2024-01-02\", \"minutes\": 388 } }\n...\n{ \"ok\": true, \"summary\": { \"count\": 730, \"has_more\": false } }\n```\n\nEither way: the agent must be able to tell, from output alone, whether the command is *making progress*, *finished*, or *failed*. Silent multi-minute waits are an availability bug, not a UX preference.\n\n---\n\n## Reducing agent round-trips\n\nThe cost of every CLI invocation an agent makes is paid twice: once in latency, once in context tokens. A CLI that takes three calls to surface what an agent needs in order to plan its next step is, for the agent, *worse* than one that takes one call — even if every individual call is faster. Optimize for round-trip count, not just per-call performance.\n\nConcrete tactics:\n\n- **Pre-compute aggregates in list responses.** A `list` that returns `{ \"ok\": true, \"data\": [...], \"count\": 7, \"has_more\": false }` saves a follow-up `count` call.\n- **Definitive empty states.** Return `{ \"ok\": true, \"data\": [], \"count\": 0 }`, never `null`. The agent should never have to disambiguate \"no results\" from \"missing field.\"\n- **Field selection on the response side.** Borrow from `gh --json title,number,state`: let callers ask for only the fields they need so list responses stay small and a follow-up \"give me more detail on item N\" is the exception, not the rule.\n- **Compact default, `--full` escape hatch.** List items should carry 3–4 fields by default; agents that need more pass an opt-in flag rather than parsing huge default payloads on every call.\n- **Next-step hints in success responses.** When the agent's likely next action is predictable, include a `next` slot: `{ \"ok\": true, \"data\": {...}, \"next\": [\"healthkit sleep summary --start-date 2026-01-01 --end-date 2026-01-07\"] }`. The agent saves a discovery turn.\n- **Cursor pagination in the envelope.** `{ \"ok\": true, \"data\": [...], \"page\": { \"next_cursor\": \"...\", \"has_more\": true } }` so the agent can decide whether to continue without parsing prose pagination markers.\n\n---\n\n## Help design\n\nProgressive, not monolithic: capability overview → resource → action → schema → examples → dry-run. A CLI with hundreds of commands should not dump its full schema into the agent's context on the first call. Top-level `--help` should be small enough to fit in a few hundred tokens; deeper detail is loaded on demand only when the agent has narrowed its target.\n\nAnthropic's *Code execution with MCP* (Nov 2025) reports the same insight from the MCP world: in one Google-Drive→Salesforce case, lazy-loading tool definitions reduced token usage from 150,000 to 2,000 — a 98.7% saving. The CLI equivalent is the layered help tree plus response-side field selection (`gh pr list --json number,title,state`).\n\n---\n\n## Safety design\n\nRead actions: easy to discover. Write actions: clearly marked. Destructive actions: hidden, gated, or separately enabled. Dry-run: everywhere feasible.\n\nTier table:\n\n| Tier | Commands | Exposure |\n|------|----------|----------|\n| preview | all commands | dry-run available everywhere |\n| open | list / get / search | full docs, easy to discover |\n| warned | create / update / send | explicit warning in help and skills |\n| hidden | delete / purge / empty | excluded from skills, gated separately |\n\nTiers are necessary but not sufficient. Graduated visibility is a prompt-side defense — it works only when the agent reads the warning and respects it, and approval fatigue degrades that defense quickly. Anthropic's *Beyond permission prompts* (Oct 2025) reports that OS-level sandboxing \"safely reduces permission prompts by 84%.\" An agent-native CLI should assume the agent runtime will additionally sandbox it at the OS level (filesystem, network, processes), and design destructive commands to fail closed inside that sandbox rather than relying on a single layer of warnings.\n\n---\n\n## Auth design\n\nHuman/system-managed token acquisition. Environment/config-based delegation. No agent involvement in browser auth flows. Separation between auth bootstrap and agent execution. See `references/examples.md` Example 3 for the canonical pattern.\n\n---\n\n## Locale, time, and determinism\n\nAgent behavior breaks subtly when CLI output depends on the host's locale or timezone. Pin determinism at the CLI boundary so the agent never has to second-guess what `2026-04-11` means or whether `1,234.56` is one number or two.\n\n- **All timestamps are UTC ISO-8601** with explicit timezone (`2026-04-11T14:30:00Z`), not local time and not Unix epoch unless explicitly requested.\n- **All dates are ISO-8601** (`2026-04-11`), never `04/11/2026` or `11/04/2026`.\n- **Numeric formats are locale-independent.** Decimal point `.`, no thousands separators in JSON output. (`1234.56`, never `1,234.56` or `1.234,56`.)\n- **Internal subprocess calls run under `LC_ALL=C`** (or equivalent), so any tool the CLI shells out to — `date`, `sort`, `awk` — produces the same bytes on every host.\n- **Sort orders are documented and stable.** Default sort is byte-wise unless the schema says otherwise.\n\nFile v1.3.4:references/examples.md\n\n# Examples and Non-Examples\n\nConcrete patterns and anti-patterns referenced from `SKILL.md`. Read this when you need to show the user what a good envelope, error, dry-run, or batch response actually looks like — or when you need to make an anti-pattern visible.\n\n---\n\n## Good examples\n\n### Example 1 — Structured error with routing fields\n\n```json\n{\n  \"ok\": false,\n  \"error\": {\n    \"code\": \"auth_expired\",\n    \"message\": \"Token expired. Re-authenticate to continue.\",\n    \"retryable\": true,\n    \"retry_after_auth\": true\n  }\n}\n```\n\nThe agent can read `retry_after_auth: true` and escalate to re-authentication without parsing prose.\n\n### Example 2 — Layered self-description\n\n```bash\n$ healthkit --help\nUsage: healthkit <resource> <action> [options]\n\nResources:\n  sleep       Sleep records and stages\n  steps       Step count and activity\n  heart       Heart rate and HRV\n\n$ healthkit sleep --help\nActions:\n  list        List sleep records by date range\n  summary     Aggregate sleep statistics\n\n$ healthkit sleep list --help\nFlags:\n  --start-date  ISO date (required)\n  --end-date    ISO date (required)\n  --format      json|table (default: json)\n  --dry-run     Preview request, do not execute\n```\n\nAn agent can traverse this tree to discover valid commands without reading external docs.\n\n### Example 3 — Delegated auth with env trust boundary\n\n```bash\n# Human / system runs once, out of band (shell profile, systemd unit, supervisor):\nhealthkit auth login                              # browser OAuth2 flow, stores token in keychain\nexport HEALTHKIT_TOKEN=\"$(healthkit auth token)\"  # token is injected into the agent's environment\n\n# Agent's own commands — it never invokes `auth login` or `auth token`:\nhealthkit sleep list --start-date 2026-01-01 --end-date 2026-01-07\n```\n\nThe agent inherits `HEALTHKIT_TOKEN` from an environment it did not build. It never runs the login subcommand, never runs the `auth token` retrieval subcommand, and never handles refresh — those belong to the human or the orchestrator. The env var is the trust boundary; the agent consumes credentials, it does not fetch them.\n\n### Example 4 — Dry-run preview before execution\n\n```bash\n$ healthkit sleep list --start-date 2026-01-01 --end-date 2026-01-07 --dry-run\n{\n  \"ok\": true,\n  \"dry_run\": true,\n  \"would_request\": {\n    \"method\": \"GET\",\n    \"url\": \"https://health.api/v1/sleep\",\n    \"params\": { \"startDate\": \"2026-01-01\", \"endDate\": \"2026-01-07\" }\n  }\n}\n```\n\nThe agent can verify the request shape before committing to execution.\n\n### Example 5 — Schema introspection\n\n```bash\n$ healthkit schema sleep.list\n{\n  \"method\": \"sleep.list\",\n  \"params\": {\n    \"startDate\": { \"type\": \"string\", \"format\": \"date\", \"required\": true },\n    \"endDate\":   { \"type\": \"string\", \"format\": \"date\", \"required\": true },\n    \"pageSize\":  { \"type\": \"integer\", \"default\": 20, \"max\": 100 }\n  }\n}\n```\n\n### Example 6 — Idempotent batch with partial success, next hint, and meta\n\n```bash\n$ healthkit alerts send-bulk \\\n    --recipients \"user1,user2,user3\" \\\n    --message \"Reminder: log today's sleep\" \\\n    --idempotency-key \"alert-batch-2026-04-11-am\"\n{\n  \"ok\": \"partial\",\n  \"data\": {\n    \"succeeded\": [\n      { \"recipient\": \"user1\", \"alert_id\": \"alrt_abc\" },\n      { \"recipient\": \"user2\", \"alert_id\": \"alrt_def\" }\n    ],\n    \"failed\": [\n      {\n        \"recipient\": \"user3\",\n        \"error\": {\n          \"code\": \"rate_limited\",\n          \"message\": \"Per-recipient rate limit exceeded\",\n          \"retryable\": true,\n          \"retry_after_seconds\": 60\n        }\n      }\n    ]\n  },\n  \"next\": [\n    \"healthkit alerts send-bulk --recipients user3 --message 'Reminder: log today\\\\'s sleep' --idempotency-key alert-batch-2026-04-11-am\"\n  ],\n  \"meta\": {\n    \"request_id\": \"req_8fa9c1\",\n    \"latency_ms\": 412,\n    \"schema_version\": \"1.4.0\"\n  }\n}\n```\n\n`ok: \"partial\"` plus per-item `error.retryable` plus the `next` slot plus a stable idempotency key give the agent everything it needs to recover in one round-trip — it does not need to re-process `user1` / `user2`, it knows exactly which item to retry, and re-running with the same `--idempotency-key` is safe.\n\n### Example 7 — Long-running export with structured stderr progress\n\n```bash\n$ healthkit export run --dataset sleep --since 2024-01-01 --format parquet > result.json\n```\n\nstderr — one JSON object per line, agent reads for liveness without blocking on stdout:\n\n```\n{ \"event\": \"start\",    \"command\": \"export.run\", \"request_id\": \"req_abc123\" }\n{ \"event\": \"progress\", \"phase\": \"fetch\", \"done\": 240, \"total\": 730, \"elapsed_ms\": 18421 }\n{ \"event\": \"progress\", \"phase\": \"fetch\", \"done\": 730, \"total\": 730, \"elapsed_ms\": 54017 }\n{ \"event\": \"progress\", \"phase\": \"write\", \"done\": 730, \"total\": 730, \"elapsed_ms\": 56103 }\n{ \"event\": \"complete\", \"request_id\": \"req_abc123\", \"elapsed_ms\": 56234 }\n```\n\nstdout — single envelope at the end:\n\n```json\n{\n  \"ok\": true,\n  \"data\": { \"rows\": 730, \"path\": \"/tmp/sleep_export.parquet\", \"size_bytes\": 142336 },\n  \"meta\": { \"request_id\": \"req_abc123\", \"latency_ms\": 56234 }\n}\n```\n\nWhat this gives the agent: liveness via stderr without polluting the stdout envelope; phase visibility (`fetch` vs `write`) so it can localize bottlenecks; one clean result object captured by the redirect; and a single `request_id` correlating stderr progress, stdout result, and upstream service logs.\n\n### Example 8 — Schema versioning with deprecation signals\n\nAn agent that calls the CLI in a loop may have cached an older view of the schema. Versioning in the response envelope lets the agent detect drift without failing silently.\n\n```bash\n$ healthkit sleep list --start-date 2026-01-01 --end-date 2026-01-07\n{\n  \"ok\": true,\n  \"data\": [{ \"id\": \"sl_001\", \"date\": \"2026-01-01\", \"minutes\": 412 }],\n  \"meta\": {\n    \"schema_version\": \"1.4.0\",\n    \"deprecated_fields\": [\"page_size\"],\n    \"introduced_in\": \"1.2.0\",\n    \"request_id\": \"req_abc123\"\n  }\n}\n```\n\nSchema introspection route:\n\n```bash\n$ healthkit schema sleep.list\n{\n  \"method\": \"sleep.list\",\n  \"introduced_in\": \"1.2.0\",\n  \"schema_version\": \"1.4.0\",\n  \"params\": {\n    \"startDate\": { \"type\": \"string\", \"format\": \"date\", \"required\": true },\n    \"endDate\":   { \"type\": \"string\", \"format\": \"date\", \"required\": true },\n    \"pageSize\":  { \"type\": \"integer\", \"default\": 20, \"max\": 100 },\n    \"page_size\": { \"type\": \"integer\", \"deprecated\": true, \"replaced_by\": \"pageSize\", \"removed_in\": \"1.5.0\" }\n  }\n}\n```\n\nThe agent compares its cached `schema_version` against the response's; on mismatch, it re-discovers before re-planning. Deprecation signals (`deprecated_fields`, `replaced_by`, `removed_in`) let it migrate proactively, and non-breaking additions don't invalidate cached calls — they just leave the cache incomplete.\n\n---\n\n## Non-examples\n\n### Non-Example 1 — Prose-only error\n\n```\nError: something went wrong with your request. Please check your input and try again.\n```\n\nThe agent cannot determine: what went wrong, whether to retry, what to fix, which field failed. It must guess or give up.\n\n### Non-Example 2 — Mixed stdout\n\n```\nFetching sleep records...\nFound 7 records.\n{\"records\": [...]}\nDone.\n```\n\nThe agent cannot reliably parse JSON because stdout contains prose mixed with data.\n\n### Non-Example 3 — No self-description\n\n```bash\n$ mytool --help\nUsage: mytool [OPTIONS] COMMAND [ARGS]...\n\nOptions:\n  --help  Show this message and exit.\n```\n\nNo resources, no actions, no schema. An agent must guess or hallucinate command names.\n\n### Non-Example 4 — Auth via agent-supplied argument\n\n```bash\nmytool --token $AGENT_GENERATED_TOKEN delete --id abc123\n```\n\nThe agent controls the token. A compromised agent can use any token it manufactures, bypassing human trust boundaries.\n\n### Non-Example 5 — Destructive commands fully exposed\n\n```bash\n$ mytool --help\nCommands:\n  list    List records\n  get     Get a record\n  delete  Delete a record        ← appears at same level as read commands\n  purge   Purge all records      ← no warning, no gate\n```\n\nAn agent browsing help can trivially discover and invoke destructive commands.\n\n### Non-Example 6 — Ambiguous exit codes\n\n```bash\n$ mytool list; echo $?\n# Returns 1 on API error\n# Returns 1 on validation error\n# Returns 1 on auth error\n# Returns 1 on network error\n```\n\nExit code 1 means everything. The orchestrator cannot route failures deterministically.\n\nFile v1.3.4:references/hybrid-mcp-cli.md\n\n# When CLI vs MCP vs Both\n\nThis skill teaches CLI design — but the largest production agents (Claude Code, Cursor, Gemini CLI, CircleCI) use both CLI and MCP, not one or the other. This file is the single home for the CLI-vs-MCP discussion and the benchmark data behind it; SKILL.md and the checklists point here.\n\n---\n\n## The hybrid pattern\n\n**State changes happen through the CLI. System understanding happens through MCP.**\n\n- **Use CLI for:** local/scriptable tasks, composable automation, state-changing operations, dev/infrastructure workflows\n- **Use MCP for:** multi-tenant SaaS, per-user authentication, stateful workflows, audit logs, fine-grained access control\n- **Use both:** most production agents that orchestrate infrastructure (Vercel CLI + MCP for SaaS integrations; GitHub CLI + MCP for enterprise GitHub instances)\n\n---\n\n## Decision matrix\n\n| Scenario | CLI | MCP | Notes |\n|----------|-----|-----|-------|\n| Single-user dev tool on same machine | ✅ | | Process model is cheap; auth is local; composable with Unix pipes |\n| Large multi-tenant SaaS with per-user OAuth | | ✅ | Centralized auth; per-user scoping; network-attachable; no binary shipping required |\n| Hundreds of tools where schema size matters | ✅ | ⚠️ | CLI wins: eager MCP schema dumps consume 55K–80K tokens upfront. CLI lazy-loads via progressive help. |\n| Orchestration + infrastructure changes | ✅ | | State changes favor process-model CLIs |\n| Complex permission models, audit requirements | | ✅ | MCP's structured audit logs and per-user attribution |\n| Hybrid: local infra + cloud SaaS | ✅✅ | ✅ | CLI for infrastructure, MCP for SaaS. Both in same agent. |\n\n---\n\n## Benchmark data\n\nThese are the numbers behind the \"CLI is more efficient than eager-loaded MCP\" claim. Cite this section when SKILL.md or the checklists need backing.\n\n- **Task completion:** CLI-based agents achieve **28% higher task completion** vs. MCP-only agents with the same token budget (Reinhard 2026).\n- **Token efficiency:** **33% advantage** measured by Token Efficiency Score (CLI: 202, MCP: 152).\n- **Per-task overhead:** ~4,150 tokens (CLI) vs ~145,000 tokens (MCP) for an identical browser-automation task — a 35× reduction (Reinhard 2026).\n- **Schema dump cost:** MCP servers that load all tool schemas upfront consume **55K–80K tokens** just for discovery. An agent running 10 sequential operations sees this overhead on every orchestration handoff.\n- **Lazy-loading wins, in either world:** Anthropic's *Code execution with MCP* (Nov 2025) reports that presenting MCP tools as code on a filesystem reduced one Google-Drive→Salesforce workflow from 150,000 tokens to 2,000 — a 98.7% saving. The same logic produces CLI's structural advantage: progressive `--help` is lazy-loading by default.\n\nThese numbers are why a CLI's progressive `--help` → resource help → `schema <resource.action>` pattern matters: the agent only pays for the parts it queries, not for everything the tool could do.\n\n---\n\n## When to stick with CLI alone\n\nMario Zechner's empirical benchmark (Aug 2025) of MCP vs CLI for coding agents lands on a one-line conclusion that's worth taking seriously: *\"Just like a lot of meetings could have been emails, a lot of MCPs could have been CLI invocations.\"* That doesn't make MCP wrong; it means the default has been wrong. For the workflows this skill targets — developer tools, infrastructure CLIs, single-user data and research workflows — CLI is the lighter, more inspectable, more composable choice.\n\n## When to switch to MCP or hybrid\n\nIf you reach a design where you'd be fighting the CLI process model (per-request user context, fine-grained per-call authorization, network-attached without local install, multi-tenant data isolation), that's the signal to add MCP to the mix, not to bend this skill out of shape. Consult the decision matrix above; if you need features from the MCP column, embrace the hybrid approach that production agents use.\n\nFile v1.3.4:references/rubric.md\n\n# Rubric\n\n14 criteria, scored 0–2 each. Every criterion is tagged with the principle it backs, so scoring the rubric is the same act as auditing the seven principles. Use this when producing the score component of a CLI review.\n\n| Criterion | Principle | 0 — Fail | 1 — Partial | 2 — Pass |\n|-----------|-----------|----------|-------------|----------|\n| **Three-audience support** | P0 | Designed for only one audience (human-only, or agent-only) | Serves two audiences well; the third is an afterthought or broken | Deliberately designed for human + agent + system with documented trade-offs |\n| **Stdout contract** | P1 | Prose or mixed output | JSON sometimes, not always | Always parseable JSON with stable envelope when stdout is not a TTY (agent context); human-readable by default only under TTY or explicit `--format table` |\n| **Stderr separation** | P1 | Diagnostics mixed into stdout | Some separation | Diagnostics always on stderr |\n| **Exit code semantics** | P1/P2 | All errors map to same code | Some codes defined | Documented, stable, distinct codes per failure class |\n| **Self-description (help)** | P3 | No `--help` or single flat page | Layered help exists but incomplete | Full progressive help: top → resource → action → schema |\n| **Schema introspection** | P6 | Not available | Partial or undocumented | `tool schema <resource.action>` returns full typed schema |\n| **Dry-run** | P3/P4 | Not available | Available for some commands | Available for all mutating commands |\n| **Idempotent retries** | P1 | Mutating commands have no idempotency story; retries create duplicates | `--idempotency-key` exists on some commands, or retry semantics are inconsistent | Every mutating command accepts `--idempotency-key`; retried calls return the original result; the `retryable` flag in the error envelope is meaningful and correct |\n| **Non-interactive operation** | P0 | CLI prompts on confirmation or password input regardless of TTY state; no `--yes`/`--force` flags | Some commands support `--yes` but TTY detection is incomplete or pagers still block | CLI never prompts when stdin is not a TTY; `--yes` / `--no-input` supported on every confirmation; pagers disabled when stdout is not a TTY; structured `confirmation_required` error returned instead of blocking |\n| **Safety tiers** | P4 | Destructive ops at same level as reads | Some warning on destructive ops | Read/write/destructive clearly tiered; destructive hidden from skills |\n| **Boundary validation** | P5 | Validation scattered across internal functions, or missing | Boundary validation exists but internal code still re-validates or accepts raw input | All input validated once at the CLI entry point; internal code operates on typed, trusted structures; validators are centralized and tested for pass and reject cases |\n| **Auth delegation** | P7 | Agent manages token lifecycle, runs login or token-retrieval subcommands | Token via env var but refreshed by the agent | Human/system manages token acquisition and refresh; agent receives a pre-fetched credential and never invokes the auth retrieval path |\n| **Error recoverability** | P1 | No error fields | `code` + `message` only | `code` + `message` + `retryable` + context fields |\n| **Trust boundary** | P2 | CLI args used for auth/config | Mixed | Env vars / config set by human; agent supplies only runtime params |\n\n**Scoring guide (max 28):**\n\n- 26–28: Agent-native\n- 17–25: Partially agent-native — specific gaps, actionable fixes\n- 0–16: Not yet agent-native — structural redesign needed\n\nFile v1.3.4:references/testing.md\n\n# Testing an Agent-Native CLI\n\nA CLI's \"agent-native\" claim is empirically falsifiable. Every contract this skill teaches — envelope shape, exit code semantics, idempotency, TTY behavior, dry-run safety, schema stability — is something an agent will discover at runtime. The cheaper, kinder discovery path is a CI suite that finds the breakage first.\n\nThis file mirrors `design-patterns.md` topic by topic. Each section: the rule, how to verify it, a minimal harness, and what a passing-into-broken regression looks like. Examples use bash because CLIs are language-agnostic and bash is the universal harness; the same checks port to `bats`, `pytest` + `subprocess`, or whatever your repo already has.\n\n---\n\n## Envelope contracts (P1, P6)\n\n**Rule.** Every response on stdout is valid JSON in the documented envelope shape (`ok: true|false|\"partial\"`, `data` or `error`, optional `meta`).\n\n**How to verify.** Parse stdout as JSON; validate against a JSON Schema kept in the repo; snapshot a representative success and failure per command.\n\n```bash\n# Smoke: every documented command must produce parseable JSON.\nfor cmd in $(tool _internal list-commands); do\n  out=$(eval \"$cmd --dry-run\" 2>/dev/null) || true\n  echo \"$out\" | jq -e 'has(\"ok\")' > /dev/null \\\n    || { echo \"FAIL: $cmd produced non-JSON or missing 'ok'\"; exit 1; }\ndone\n\n# Schema: validate each envelope against the published contract.\necho \"$out\" | jq . | ajv validate -s schemas/envelope.schema.json -d -\n```\n\n**Regression looks like:** a release that adds a new field at the top level instead of inside `data` or `meta`; a command that returns a bare array instead of an envelope; pretty-print whitespace creeping in and breaking byte-equal snapshots.\n\n---\n\n## Stdout/stderr separation (P1)\n\n**Rule.** Stdout carries the JSON envelope only. Stderr carries human diagnostics or NDJSON progress events — never the final envelope.\n\n**How to verify.** Capture both streams independently, then assert stdout parses as exactly one JSON document and stderr does not.\n\n```bash\nout=$(mktemp); err=$(mktemp)\ntool sleep list --start-date 2026-01-01 --end-date 2026-01-07 >\"$out\" 2>\"$err\"\n\njq -e . \"$out\" > /dev/null || { echo \"FAIL: stdout is not a single JSON doc\"; exit 1; }\nif jq -e 'has(\"ok\")' \"$err\" 2>/dev/null; then\n  echo \"FAIL: envelope leaked into stderr\"; exit 1\nfi\n```\n\n**Regression looks like:** a `printf \"fetching...\"` left in by a refactor; a debug `console.log` going to stdout instead of stderr; progress text appearing inline with the result envelope.\n\n---\n\n## Exit code semantics (P1/P2)\n\n**Rule.** Each failure class maps to a stable, documented exit code. Success is `0`; one code per failure family thereafter.\n\n**How to verify.** Synthesize each failure class deliberately and assert the code. Keep the assertions parameterized so adding a new class is one line.\n\n```bash\ndeclare -A cases=(\n  [success]=\"tool sleep list --start-date 2026-01-01 --end-date 2026-01-07|0\"\n  [validation]=\"tool sleep list --start-date not-a-date|3\"\n  [auth]=\"HEALTHKIT_TOKEN=invalid tool sleep list --start-date 2026-01-01 --end-date 2026-01-07|2\"\n  [runtime]=\"HEALTHKIT_BASE_URL=http://127.0.0.1:1 tool sleep list --start-date 2026-01-01 --end-date 2026-01-07|1\"\n)\nfor name in \"${!cases[@]}\"; do\n  IFS='|' read -r cmd want <<< \"${cases[$name]}\"\n  eval \"$cmd\" >/dev/null 2>&1; got=$?\n  [[ \"$got\" == \"$want\" ]] || { echo \"FAIL: $name expected $want, got $got\"; exit 1; }\ndone\n```\n\n**Regression looks like:** a refactor that collapses validation and runtime errors into the same `1`; a try/catch that swallows the auth code and re-emits it as `1`.\n\n---\n\n## Idempotency replay (P1)\n\n**Rule.** A mutating command invoked twice with the same `--idempotency-key` returns the same envelope, and the side effect happens only once.\n\n**How to verify.** Run twice, diff the bytes, then check the backing store. Both must be identical / unchanged on the second call.\n\n```bash\nkey=\"test-$(uuidgen)\"\nr1=$(tool alerts send --recipient u1 --message hi --idempotency-key \"$key\")\nr2=$(tool alerts send --recipient u1 --message hi --idempotency-key \"$key\")\n\ndiff <(echo \"$r1\") <(echo \"$r2\") || { echo \"FAIL: replay returned different envelope\"; exit 1; }\n\ncount=$(tool _internal alert-rows-for-key \"$key\")\n[[ \"$count\" == \"1\" ]] || { echo \"FAIL: side-effect ran $count times, expected 1\"; exit 1; }\n```\n\n**Regression looks like:** the second call returns a fresh `id` / `created_at`; a duplicate row appears in the audit table; the `next` slot in the original response no longer matches on replay.\n\n---\n\n## Non-interactive / TTY behavior (P0)\n\n**Rule.** When stdin is not a TTY the CLI never prompts; when stdout is not a TTY it never invokes a pager. Confirmations return a structured `confirmation_required` error instead of blocking.\n\n**How to verify.** Run under a closed stdin with a wall-clock timeout. The process must exit cleanly within the timeout and produce a structured envelope, not a hang.\n\n```bash\n# Stdin not a TTY: must not block.\nout=$(echo | timeout 5 tool danger purge --id abc)\ngot=$?\n[[ \"$got\" != \"124\" ]] || { echo \"FAIL: command hung (timed out)\"; exit 1; }\necho \"$out\" | jq -e '.error.code == \"confirmation_required\"' > /dev/null \\\n  || { echo \"FAIL: expected confirmation_required envelope, got: $out\"; exit 1; }\n\n# --yes path: same command must succeed without prompting.\necho | timeout 5 tool danger purge --id abc --yes >/dev/null \\\n  || { echo \"FAIL: --yes did not bypass the prompt\"; exit 1; }\n\n# Stdout not a TTY: pager must not be invoked.\nPAGER='/bin/false' tool sleep list --start-date 2026-01-01 --end-date 2026-01-07 \\\n  | jq -e . > /dev/null || { echo \"FAIL: pager invoked under non-TTY stdout\"; exit 1; }\n```\n\n**Regression looks like:** an interactive `read -p` slipped in for a new flag; a library upgrade that re-enables pager autodetect; a `Press Enter to continue` left in a long-running command.\n\n---\n\n## Self-description and schema drift (P3, P6)\n\n**Rule.** `--help` is small enough to live in an agent's context (a few hundred tokens at the top, growing only as the agent narrows its target). The `schema` subcommand returns a typed contract that includes `schema_version`. Cached agents detect drift through the version, not by hitting a removed method.\n\n**How to verify.** Snapshot help-output byte size as an upper bound; parse `schema <method>` and assert it includes the documented fields; on every release, diff the schema snapshot and fail loudly on incompatible changes.\n\n```bash\n# Token budget on top-level help (rough proxy: bytes / 4 ~ tokens).\nlimit=2000\nsize=$(tool --help | wc -c)\n[[ \"$size\" -le \"$limit\" ]] || { echo \"FAIL: top-level --help is $size bytes, limit $limit\"; exit 1; }\n\n# Schema endpoint is real and includes versioning.\ntool schema sleep.list | jq -e '.schema_version and .params and .introduced_in' > /dev/null \\\n  || { echo \"FAIL: schema response missing required fields\"; exit 1; }\n\n# Schema regression gate: snapshot diff blocks accidental breaking changes.\ntool schema sleep.list > /tmp/schema.new.json\ndiff -u tests/snapshots/schema.sleep.list.json /tmp/schema.new.json \\\n  || { echo \"REVIEW: schema changed — confirm version bump and deprecation signals\"; exit 1; }\n```\n\n**Regression looks like:** a help generator that starts emitting the full schema in `--help`; a schema response that drops `schema_version` after a refactor; a field renamed between versions without a `deprecated` / `replaced_by` entry on the old name.\n\n---\n\n## Dry-run safety (P3, P4)\n\n**Rule.** Every mutating command supports `--dry-run`. The dry-run output describes the would-be request. No side effect occurs.\n\n**How to verify.** Drive every mutating command in dry-run mode, assert the envelope shape, then assert the backing store is unchanged.\n\n```bash\nbefore=$(tool _internal store-fingerprint)\n\nfor cmd in $(tool _internal list-mutating-commands); do\n  out=$(eval \"$cmd --dry-run\")\n  echo \"$out\" | jq -e '.dry_run == true and .would_request' > /dev/null \\\n    || { echo \"FAIL: $cmd --dry-run is missing dry_run/would_request\"; exit 1; }\ndone\n\nafter=$(tool _internal store-fingerprint)\n[[ \"$before\" == \"$after\" ]] || { echo \"FAIL: dry-run mutated state\"; exit 1; }\n```\n\n**Regression looks like:** a new mutating command shipped without `--dry-run` wired up; a dry-run path that still issues the upstream request; an envelope that drops the `dry_run: true` flag.\n\n---\n\n## Auth delegation (P7)\n\n**Rule.** The agent never invokes the auth lifecycle (`auth login`, `auth token`, browser flows). It receives a credential from a human-managed env var and consumes it.\n\n**How to verify.** Grep the generated agent-facing artifacts (skills, MCP tool manifests, examples) for the auth subcommands. Any reference is a violation.\n\n```bash\nforbidden='auth login|auth token|auth refresh'\n# Grep generated agent-facing artifacts only (skill manifests, MCP tool defs, codex sidecars).\n# Human-facing docs and examples that document the human bootstrap path\n# (e.g. examples.md showing `healthkit auth login`) are expected and excluded here.\nhits=$(grep -RInE \"$forbidden\" agents/ 2>/dev/null | grep -v '^//' || true)\n[[ -z \"$hits\" ]] || { echo \"FAIL: agent surface references auth lifecycle:\"; echo \"$hits\"; exit 1; }\n```\n\n**Regression looks like:** a worked example in the skill that pipes `auth token` into the next command; a generated MCP tool that exposes `auth.login`; an \"easy onboarding\" addition that has the agent call `auth login` itself.\n\n---\n\n## Locale and time determinism\n\n**Rule.** Output bytes are identical regardless of host locale or timezone. Timestamps are UTC ISO-8601, dates are ISO-8601, numbers use `.` and no thousands separator.\n\n**How to verify.** Run the same invocation under two divergent locales and one shifted timezone. The captured bytes must match exactly.\n\n```bash\nfixed=(--start-date 2026-01-01 --end-date 2026-01-07)\n\na=$(LC_ALL=C       TZ=UTC          tool sleep list \"${fixed[@]}\")\nb=$(LC_ALL=de_DE.UTF-8 TZ=Asia/Tokyo tool sleep list \"${fixed[@]}\")\n\ndiff <(echo \"$a\") <(echo \"$b\") || { echo \"FAIL: output drifted with locale/TZ\"; exit 1; }\n```\n\n**Regression looks like:** `1.234,56` appearing instead of `1234.56`; a timestamp printed as `Mo., 12. Jan. 2026` instead of `2026-01-12T00:00:00Z`; a sort order that rearranges itself when the host's `LANG` changes.\n\n---\n\n## Long-running output and streaming\n\n**Rule.** Long-running commands emit structured progress on stderr (one JSON object per line) and a single envelope on stdout. NDJSON list commands emit one envelope per line plus a final summary.\n\n**How to verify.** Run with stderr redirected to a file, assert each line parses as JSON with an `event` field; for `--stream`, assert the last line carries a `summary`.\n\n```bash\ntool export run --dataset sleep --since 2024-01-01 > /tmp/result.json 2> /tmp/progress.log\njq -e . /tmp/result.json > /dev/null || { echo \"FAIL: stdout not a single envelope\"; exit 1; }\nwhile IFS= read -r line; do\n  echo \"$line\" | jq -e 'has(\"event\")' > /dev/null \\\n    || { echo \"FAIL: stderr line is not structured: $line\"; exit 1; }\ndone < /tmp/progress.log\n\ntool sleep list --since 2024-01-01 --stream | tail -1 \\\n  | jq -e '.summary' > /dev/null || { echo \"FAIL: stream missing final summary\"; exit 1; }\n```\n\n**Regression looks like:** progress events going to stdout and breaking the redirect-to-file pattern; an extra blank line at the end of `--stream` that breaks `tail -1`; a refactor that turns the per-line JSON into a single big array at the end.\n\n---\n\n## Wiring it into CI\n\nThese checks are cheap — most are seconds — and they belong on every PR. Three things make them durable:\n\n1. **Treat the schema snapshot like a public ABI.** A diff is a release decision, not a code-review nitpick. Pair it with a CHANGELOG entry and a version bump.\n2. **Run the locale and TTY tests in their own job.** They depend on environment shape, not on the build, and they fail fast when a maintainer's local laptop is the only thing the CLI was ever tested under.\n3. **Keep the failure messages structured.** A test suite for an agent-native CLI is itself an agent surface — the messages above are written so an LLM driving CI can read them and route the fix without grepping logs.\n\nFile v1.3.4:skill-card.md\n\n## Description: <br>\nUse when designing, reviewing, or refactoring a CLI that must serve AI agents alongside humans, or when converting an API or SDK into an agent-usable CLI interface. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[agents365-ai](https://clawhub.ai/user/agents365-ai) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and engineers use this skill to evaluate, design, and refactor command-line interfaces so they provide predictable structured output, self-description, dry-run previews, delegated authentication, and safer execution boundaries for humans, AI agents, and orchestration systems. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill may be invoked automatically for CLI-design topics and can recommend changes involving credentials or destructive operations. <br>\nMitigation: Review generated CLI designs and implementations before granting real credentials, write authority, delete authority, or production access. <br>\nRisk: Design guidance or refactor plans could be incomplete or misleading for a specific CLI's operational constraints. <br>\nMitigation: Validate recommendations against the target CLI's tests, security model, and release process before deployment. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/agents365-ai/skills/agent-native-design) <br>\n- [Review Checklists](references/checklists.md) <br>\n- [Rubric](references/rubric.md) <br>\n- [Design Patterns](references/design-patterns.md) <br>\n- [Examples and Non-Examples](references/examples.md) <br>\n- [When CLI vs MCP vs Both](references/hybrid-mcp-cli.md) <br>\n- [Testing an Agent-Native CLI](references/testing.md) <br>\n- [Citations](references/citations.md) <br>\n- [Code execution with MCP: Building more efficient agents](https://www.anthropic.com/engineering/code-execution-with-mcp) <br>\n- [Beyond permission prompts: making Claude Code more secure and autonomous](https://www.anthropic.com/engineering/claude-code-sandboxing) <br>\n- [Scripting with GitHub CLI](https://github.blog/engineering/engineering-principles/scripting-with-github-cli/) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, code, shell commands, configuration, guidance] <br>\n**Output Format:** [Markdown with structured review sections and inline code or shell blocks] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [May include rubric scores, seven-principle reviews, risk summaries, and prioritized refactor plans.] <br>\n\n## Skill Version(s): <br>\n1.3.4 (source: server release metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.3.4:agents/openai.yaml\n\ninterface:\n  display_name: \"Agent Native Design\"\n  short_description: \"Design, review, and refactor CLIs to serve humans, AI agents, and orchestration systems simultaneously — structured output, schema introspection, dry-run, safety tiers, and delegated authentication\"\n  brand_color: \"#1A3A5C\"\n\npolicy:\n  allow_implicit_invocation: true\n\ncapabilities:\n  - Evaluate whether a CLI is agent-native across 7 design principles\n  - Design stdout JSON contracts with stable error envelopes\n  - Define exit code semantics per failure class (success, runtime, auth, validation)\n  - Design layered --help, schema introspection, and dry-run previews\n  - Define safety tiers: open (read), warned (write), hidden (destructive)\n  - Design delegated authentication — human owns auth lifecycle, agent receives token\n  - Apply directional trust model — env vars trusted, CLI args untrusted\n  - Convert a REST API or SDK into an agent-native CLI command tree\n  - Score a CLI on the 14-criterion rubric and summarize the seven design principles\n  - Produce a prioritized P0/P1/P2 refactor plan with concrete examples\n\nprerequisites: []\n\nArchive v1.3.3: 18 files, 62882 bytes\n\nFiles: agents/openai.yaml (1121b), docs/index.html (25313b), docs/maintainers/IMPROVEMENTS_APPLIED.md (7683b), docs/maintainers/REVIEW_2026.md (14872b), docs/zh.html (25213b), README_CN.md (6611b), README.md (11144b), references/checklists.md (3644b), references/citations.md (3748b), references/design-patterns.md (12640b), references/examples.md (8337b), references/hybrid-mcp-cli.md (3986b), references/rubric.md (3454b), references/testing.md (11979b), scripts/validate-metadata.sh (2525b), skill-card.md (3522b), SKILL.md (14233b), _meta.json (138b)\n\nFile v1.3.3:SKILL.md\n\n---\nname: agent-native-design\ndescription: Use when designing, reviewing, or refactoring a CLI that must serve AI agents alongside humans, or when converting an API or SDK into an agent-usable CLI interface.\nlicense: MIT\nhomepage: https://github.com/Agents365-ai/agent-native-design\ncompatibility: Includes sidecar metadata for OpenClaw, Hermes, pi-mono, and OpenAI Codex; the core SKILL.md is portable to any agent runtime that supports Agent Skills-style instructions.\nplatforms: [macos, linux, windows]\nmetadata: {\"openclaw\":{\"requires\":{},\"emoji\":\"⌨️\",\"os\":[\"darwin\",\"linux\",\"win32\"]},\"hermes\":{\"tags\":[\"cli\",\"agent-native\",\"interface-design\",\"structured-output\",\"schema-driven\"],\"category\":\"engineering\",\"requires_tools\":[],\"related_skills\":[]},\"pimo\":{\"category\":\"engineering\",\"tags\":[\"cli\",\"agent-native\",\"interface-design\",\"structured-output\",\"schema-driven\"]},\"author\":\"Agents365-ai\",\"version\":\"1.3.3\"}\n---\n\n# agent-native-design\n\n## Purpose\n\nThis skill helps analyze, design, and refactor command-line tools so they can reliably serve **humans**, **AI agents**, and **orchestration systems** at the same time.\n\nIt is not a skill for merely *using* a CLI. It is a skill for designing and reviewing a CLI as an **agent-native interface**.\n\nThe skill focuses on four goals:\n\n1. Make CLI behavior predictable for AI agents.\n2. Make CLI output readable and recoverable for humans.\n3. Make CLI execution manageable for systems and orchestrators.\n4. Define a complete interaction loop from authentication to error routing.\n\n---\n\n## When to use this skill\n\nUse this skill when the user wants to:\n\n* evaluate whether an existing CLI is agent-friendly\n* redesign a CLI to better support AI agents\n* convert an API or SDK into an agent-native CLI\n* review help output, schema design, exit codes, or JSON contracts\n* design dry-run, auth delegation, or safety boundaries\n* generate CLI skills, docs, or interface conventions from schema\n* refactor a human-oriented CLI into a machine-friendly one\n* define how a CLI should interact with an agent runtime\n\nTypical prompts include:\n\n* \"Review this CLI and tell me whether it is agent-native.\"\n* \"Design a CLI for this API that an AI agent can use reliably.\"\n* \"Refactor this tool so stdout is machine-readable and safer for agents.\"\n* \"Help me define schema introspection, dry-run, and exit code semantics.\"\n\n## When not to use this skill\n\nDo not use this skill when the user only wants:\n\n* help running a specific command\n* installation help for a CLI\n* shell troubleshooting unrelated to interface design\n* generic Linux or terminal tutorials\n* agent planning or memory design unrelated to tools\n* API business logic review without any CLI/tooling layer\n\n---\n\n## Core model\n\nAn agent-native CLI must simultaneously serve three audiences.\n\n### 1. Human\n\nNeeds: readable output, friendly error messages, onboarding guidance.\nChannels: `stderr`, optional `--format table`, interactive TUI when appropriate.\n\n### 2. AI Agent\n\nNeeds: structured data, stable contracts, self-description.\nChannels: `stdout` as JSON, stable exit codes, schema introspection, dry-run previews, generated skills/docs.\n\n### 3. System / Orchestrator\n\nNeeds: delegated authentication, process management, deterministic error routing.\nChannels: environment variables, exit codes, dry-run mode, stable command semantics.\n\n### Foundational contract\n\n| Channel | Primary audience |\n|---------|-----------------|\n| `stdout` | Machines and agents |\n| `stderr` | Humans |\n| `exit codes` | Systems and orchestrators |\n\nThis skill teaches how to make CLI a first-class interface for agents. Production agents (Claude Code, Cursor, Gemini CLI) often pair a CLI with an MCP server — CLI for state changes and local/scriptable work, MCP for multi-tenant SaaS and per-user auth. When a design needs the MCP side as well, see `references/hybrid-mcp-cli.md` for the decision matrix and the benchmark data behind the CLI/MCP tradeoff.\n\n---\n\n## The complete interaction loop\n\n| Phase | Step | Description |\n|-------|------|-------------|\n| 0. Bootstrap | 1 | Human/system obtains auth token or credentials |\n| 0. Bootstrap | 2 | Set trusted env vars: token, profile, safety mode |\n| 1. Discovery | 3 | Agent loads skills or command summaries |\n| 1. Discovery | 4 | Agent queries schema/help for parameters |\n| 2. Planning | 5 | Agent uses `--dry-run` to preview request shape |\n| 3. Execution | 6 | Agent executes with validated inputs |\n| 4. Interpretation | 7 | Agent parses structured result |\n| 5. Recovery | 8 | Agent uses exit code + error object to retry, re-auth, repair, or escalate |\n\nA CLI that does not support every phase is incomplete from the agent's perspective.\n\n---\n\n## Seven principles\n\nThese are load-bearing. Each principle has at least one rubric criterion and at least one example backing it.\n\n### Principle 0. One CLI, Three Audiences\n\nThe CLI must serve human, agent, and system simultaneously. A design that serves only one audience is incomplete.\n\n### Principle 1. Structured Output Is the Interface\n\n`stdout` should always be parseable and stable. Both success and failure are structured JSON. The CLI must decide for itself which audience is reading: detect at startup whether stdout is a TTY, default to JSON when it is not, default to human-readable when it is. `NO_COLOR` and an explicit `--format json|table` flag override the auto-detection. Agents should never have to remember to pass `--format json` — if they have to, they will forget, and the run will silently produce un-parseable prose. Envelope and error contract: `references/design-patterns.md#output-envelopes`.\n\n### Principle 2. Trust Is Directional\n\nCLI arguments are not inherently trusted — they may come from a hallucinating or prompt-injected agent. Environment-level configuration set by the human or system is more trusted. The agent chooses *what to do* within a bounded surface; the human defines *where and how it is allowed to operate*.\n\n### Principle 3. The CLI Must Describe Itself\n\nThe CLI must be self-describing enough that an agent can use it without reading external README files. Self-description must be **progressive**, not eager: top-level `--help` lists resources; resource help lists actions; action help lists flags; a separate `schema <resource.action>` returns the full typed schema. A CLI with hundreds of commands that dumps everything into the first `--help` pays that token cost on every agent invocation. See `references/design-patterns.md#help-design` and `examples.md` Examples 2 and 5.\n\n### Principle 4. Safety Through Graduated Visibility\n\nRead commands are easy to discover; mutating commands carry explicit warnings; destructive commands are hidden from skills or gated separately. Tier table and rationale: `references/design-patterns.md#safety-design`. Tiers are necessary but not sufficient — they are a prompt-side defense and approval fatigue degrades them quickly. Assume the agent runtime will additionally sandbox the CLI at the OS level (filesystem, network, processes), and design destructive commands to fail closed inside that sandbox.\n\n### Principle 5. Validate at the Boundary, Not in the Middle\n\nInputs are validated once at the CLI entry point. Internal code operates on validated, typed, trusted structures. Validation functions are centralized and tested for both pass and reject cases.\n\n### Principle 6. The Schema Is the Source of Truth\n\nIf a schema exists, everything derives from it: CLI command structure, validation rules, help text, generated docs, generated skills, type definitions, dry-run contracts. The schema is never manually duplicated. The schema must also carry its own version and deprecation signals, surfaced in the `meta` block of every response, so agents that have cached an older view can detect drift and re-discover rather than silently calling a removed method. Full versioning contract and example: `references/design-patterns.md#schema-versioning`.\n\n### Principle 7. Authentication Must Be Delegatable\n\nAuthentication is obtained and refreshed by human/system-managed flows. The agent uses credentials; it never owns the auth lifecycle. Preferred mechanisms: environment variables, config files, OS keychain integration, externally refreshed tokens. Canonical pattern: `examples.md` Example 3.\n\n---\n\n## Standard review workflow\n\n### Step 0. Update check (notify, don't pull) — first use per conversation\n\nThrottle to one check per 24 hours per installation; never mutate the skill directory without explicit user consent.\n\n1. If `<this-skill-dir>/.last_update` exists and is less than 24 hours old, skip this step entirely.\n\n2. Otherwise, fetch the latest tag from upstream:\n\n   ```bash\n   git -C <this-skill-dir> ls-remote --tags origin 'v*' 2>/dev/null \\\n     | awk '{print $2}' | sed 's|refs/tags/||' \\\n     | sort -V | tail -1\n   ```\n\n3. Compare with this skill's `metadata.version` from the frontmatter. If the upstream tag is strictly newer (semver), tell the user one line and ask:\n\n   > \"A newer version of this skill is available: vX.Y.Z → vA.B.C. Want me to `git pull`?\"\n\n   If they say yes, run `git -C <this-skill-dir> pull --ff-only`. Refresh `.last_update` either way so the prompt doesn't repeat for 24 hours.\n\n4. If upstream is the same or older, refresh `.last_update` silently and continue.\n\n5. On any failure (offline, not a git checkout — e.g. ClawHub-installed copy, read-only path, no permission), swallow the error silently and continue with the user's task. Do not mention the failure.\n\nThis step is the only place this skill ever touches its own files. It notifies; it does not pull without permission. The user owns the updat\n\nArchive v1.3.2: 17 files, 60393 bytes\n\nFiles: agents/openai.yaml (1121b), docs/index.html (25313b), docs/maintainers/IMPROVEMENTS_APPLIED.md (7683b), docs/maintainers/REVIEW_2026.md (14872b), docs/zh.html (25213b), README_CN.md (6611b), README.md (10303b), references/checklists.md (3644b), references/citations.md (3748b), references/design-patterns.md (12640b), references/examples.md (8337b), references/hybrid-mcp-cli.md (3986b), references/rubric.md (3454b), references/testing.md (11979b), scripts/validate-metadata.sh (2525b), SKILL.md (13358b), _meta.json (138b)\n\nArchive v1.3.1: 17 files, 59896 bytes\n\nFiles: agents/openai.yaml (1121b), docs/index.html (25313b), docs/maintainers/IMPROVEMENTS_APPLIED.md (7683b), docs/maintainers/REVIEW_2026.md (14872b), docs/zh.html (25213b), README_CN.md (6611b), README.md (9818b), references/checklists.md (3644b), references/citations.md (3748b), references/design-patterns.md (12640b), references/examples.md (8337b), references/hybrid-mcp-cli.md (3986b), references/rubric.md (3454b), references/testing.md (11979b), scripts/validate-metadata.sh (2525b), SKILL.md (12849b), _meta.json (138b)","readmeExcerpt":"Skill: Agent Native Design Owner: agents365-ai Summary: Use when designing, reviewing, or refactoring a CLI that must serve AI agents alongside humans, or when converting an API or SDK into an agent-usable CLI int... Tags: agent-native:1.3.3, cli:1.3.3, interface-design:1.3.3, latest:1.3.5, schema-driven:1.3.3, structured-output:1.3.3 Version history: v1.3.5 | 2026-07-08T17:35:37.031Z | auto - No file changes detecte","codeSnippets":[],"executableExamples":[{"language":"json","snippet":"{ \"ok\": true, \"data\": {} }"},{"language":"json","snippet":"{\n  \"ok\": false,\n  \"error\": {\n    \"code\": \"validation_error\",\n    \"message\": \"Missing required field: email\",\n    \"field\": \"email\",\n    \"retryable\": false\n  }\n}"},{"language":"json","snippet":"{\n  \"ok\": \"partial\",\n  \"data\": {\n    \"succeeded\": [\n      { \"id\": \"msg_001\", \"status\": \"sent\" },\n      { \"id\": \"msg_002\", \"status\": \"sent\" }\n    ],\n    \"failed\": [\n      {\n        \"id\": \"msg_003\",\n        \"error\": { \"code\": \"rate_limited\", \"message\": \"Rate limit exceeded for recipient\", \"retryable\": true }\n      }\n    ]\n  }\n}"},{"language":"json","snippet":"{\n  \"ok\": true,\n  \"data\": { \"id\": \"abc123\" },\n  \"meta\": {\n    \"request_id\": \"req_8fa9c1\",\n    \"latency_ms\": 412,\n    \"schema_version\": \"1.4.0\"\n  }\n}"},{"language":"json","snippet":"{\n  \"method\": \"sleep.list\",\n  \"since\": \"1.2.0\",\n  \"deprecated\": false,\n  \"params\": {\n    \"startDate\": { \"type\": \"string\", \"format\": \"date\", \"required\": true },\n    \"endDate\":   { \"type\": \"string\", \"format\": \"date\", \"required\": true },\n    \"page_size\": { \"type\": \"integer\", \"default\": 20, \"max\": 100, \"deprecated\": true, \"replaced_by\": \"pageSize\", \"removed_in\": \"1.5.0\" }\n  }\n}"},{"language":"json","snippet":"{ \"ok\": false, \"error\": { \"code\": \"confirmation_required\", \"message\": \"Pass --yes to confirm.\", \"retryable\": false } }"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: agent-native-design\ndescription: Use when designing, reviewing, or refactoring a CLI that must serve AI agents alongside humans, or when converting an API or SDK into an agent-usable CLI interface.\nlicense: CC-BY-NC-4.0\nhomepage: https://github.com/Agents365-ai/agent-native-design\ncompatibility: Includes sidecar metadata for OpenClaw, Hermes, pi-mono, and OpenAI Codex; the core SKILL.md is portable to any agent runtime that supports Agent Skills-style instructions.\nplatforms: [macos, linux, windows]\nmetadata: {\"openclaw\":{\"requires\":{},\"emoji\":\"⌨️\",\"os\":[\"darwin\",\"linux\",\"win32\"]},\"hermes\":{\"tags\":[\"cli\",\"agent-native\",\"interface-design\",\"structured-output\",\"schema-driven\"],\"category\":\"engineering\",\"requires_tools\":[],\"related_skills\":[]},\"pimo\":{\"category\":\"engineering\",\"tags\":[\"cli\",\"agent-native\",\"interface-design\",\"structured-output\",\"schema-driven\"]},\"author\":\"Agents365-ai\",\"version\":\"1.3.5\"}\n---\n\n# agent-native-design\n\n## Purpose\n\nThis skill helps analyze, design, and refactor command-line tools so they can reliably serve **humans**, **AI agents**, and **orchestration systems** at the same time.\n\nIt is not a skill for merely *using* a CLI. It is a skill for designing and reviewing a CLI as an **agent-native interface**.\n\nThe skill focuses on four goals:\n\n1. Make CLI behavior predictable for AI agents.\n2. Make CLI output readable and recoverable for humans.\n3. Make CLI execution manageable for systems and orchestrators.\n4. Define a complete interaction loop from authentication to error routing.\n\n---\n\n## When to use this skill\n\nUse this skill when the user wants to:\n\n* evaluate whether an existing CLI is agent-friendly\n* redesign a CLI to better support AI agents\n* convert an API or SDK into an agent-native CLI\n* review help output, schema design, exit codes, or JSON contracts\n* design dry-run, auth delegation, or safety boundaries\n* generate CLI skills, docs, or interface conventions from schema\n* refactor a human-oriented CLI into a machine-friendly one\n* define how a CLI should interact with an agent runtime\n\nTypical prompts include:\n\n* \"Review this CLI and tell me whether it is agent-native.\"\n* \"Design a CLI for this API that an AI agent can use reliably.\"\n* \"Refactor this tool so stdout is machine-readable and safer for agents.\"\n* \"Help me define schema introspection, dry-run, and exit code semantics.\"\n\n## When not to use this skill\n\nDo not use this skill when the user only wants:\n\n* help running a specific command\n* installation help for a CLI\n* shell troubleshooting unrelated to interface design\n* generic Linux or terminal tutorials\n* agent planning or memory design unrelated to tools\n* API business logic review without any CLI/tooling layer\n\n---\n\n## Core model\n\nAn agent-native CLI must simultaneously serve three audiences.\n\n### 1. Human\n\nNeeds: readable output, friendly error messages, onboarding guidance.\nChannels: `stderr`, optional `--format table`, interactive TUI when appropriate.\n\n### 2. AI Agent\n\nNeeds: structured dat"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn74y2dmpaszt959h16p11k53983ht10\",\n  \"slug\": \"agent-native-design\",\n  \"version\": \"1.3.5\",\n  \"publishedAt\": 1783532137031\n}"},{"path":"references/checklists.md","content":"# Review Checklists\n\nUse this when evaluating a CLI for agent readiness, or when sanity-checking a new design before shipping.\n\n## Output\n\n- [ ] `stdout` is valid JSON when stdout is not a TTY or when `--format json` is passed (success and failure)\n- [ ] `stderr` carries human-readable diagnostics only\n- [ ] JSON envelope is stable: `{ \"ok\": bool, \"data\": ... }` or `{ \"ok\": false, \"error\": ... }`\n- [ ] Error object includes: `code`, `message`, `retryable`\n- [ ] No prose mixed into `stdout`\n\n## Exit codes\n\n- [ ] Exit codes are documented\n- [ ] Exit codes are stable across versions\n- [ ] Distinct codes for: success (0), runtime error, auth error, validation error\n- [ ] Exit code mapping is available via `--help` or `schema`\n\n## Retry and interaction mode\n\n- [ ] Every mutating command accepts `--idempotency-key`\n- [ ] Retried calls with the same idempotency key return the original result\n- [ ] `retryable` field in error envelope is meaningful and correct\n- [ ] CLI never prompts for input when stdin is not a TTY\n- [ ] `--yes` / `--no-input` / `--force` supported on every command that would otherwise prompt\n- [ ] Returns structured `confirmation_required` error instead of blocking on missing confirmation\n- [ ] stdout defaults to JSON when stdout is not a TTY (no `--format json` required)\n- [ ] Pagers (`less`, `more`) disabled when stdout is not a TTY\n\n## Self-description\n\n- [ ] Top-level `--help` lists all resources/commands\n- [ ] Resource-level `--help` lists actions\n- [ ] Action-level `--help` lists all flags with types\n- [ ] Schema introspection command available (`tool schema <resource.action>`)\n- [ ] Dry-run available for all mutating commands\n\n## Safety\n\n- [ ] Read commands clearly discoverable\n- [ ] Write/mutating commands carry explicit warning in help\n- [ ] Destructive commands (delete/purge) hidden from skills or gated\n- [ ] Dry-run covers all write operations\n\n## Auth\n\n- [ ] Human/system manages token acquisition (browser flow, keychain)\n- [ ] Agent receives credential via env var or pre-fetched token\n- [ ] Agent never navigates OAuth2 or browser flows\n- [ ] Token refresh handled outside agent runtime\n\n## Trust\n\n- [ ] CLI args treated as untrusted (validated at boundary)\n- [ ] Environment variables used for config/safety settings (human-set)\n- [ ] Agent cannot escalate its own privileges via CLI args\n\n## Schema\n\n- [ ] Schema is the single source of truth\n- [ ] CLI command structure derives from schema\n- [ ] Validation derives from schema\n- [ ] Help text derives from schema\n- [ ] Generated skills derive from schema (if applicable)\n- [ ] Schema version included in every response's `meta` block\n- [ ] Deprecation signals included in schema responses (`deprecated_fields`, `replaced_by`, `removed_in`)\n- [ ] Schema introspection is incremental (not eager): `--help` is small; full schema via `schema` subcommand only\n\n## Token efficiency\n\nFor agents that call the CLI in loops (orchestration, multi-turn planning), context cost matters. These items he"},{"path":"references/citations.md","content":"# Citations\n\nSources actually quoted or directly relied on in SKILL.md and the other reference files. The broader landscape (clig.dev, the gh / aws / kubectl design corpus) is implicit background.\n\n## Anthropic engineering\n\n- *Code execution with MCP: Building more efficient agents* — https://www.anthropic.com/engineering/code-execution-with-mcp (Nov 4, 2025). Progressive disclosure of tool definitions; the 150K → 2K token reduction case study cited under Principle 3 / `design-patterns.md#help-design`.\n- *Beyond permission prompts: making Claude Code more secure and autonomous* — https://www.anthropic.com/engineering/claude-code-sandboxing (Oct 20, 2025). Approval fatigue and the 84% prompt-reduction figure cited under Principle 4 / `design-patterns.md#safety-design`.\n\n## CLI-for-agents writing\n\n- Ugo Enyioha, *Writing CLI Tools That AI Agents Actually Want to Use* — https://dev.to/uenyioha/writing-cli-tools-that-ai-agents-actually-want-to-use-39no (Feb 27, 2025). Idempotency-on-retry and \"agents cannot type 'y'\" framings cited in the Idempotency and Non-interactive sections of `design-patterns.md`.\n- Thibault Le Ouay Ducasse / openstatus, *Building a CLI That Works for Humans and Machines* — https://www.openstatus.dev/blog/building-cli-for-human-and-agents (Apr 2, 2026). TTY detection as the human/machine switch cited under Principle 1.\n- Mario Zechner, *MCP vs CLI: Benchmarking Tools for Coding Agents* — https://mariozechner.at/posts/2025-08-15-mcp-vs-cli/ (Aug 15, 2025). Empirical case that many MCP servers could be CLI invocations.\n- Armin Ronacher, *Skills vs Dynamic MCP Loadouts* — https://lucumr.pocoo.org/2025/12/13/skills-vs-mcp/ (Dec 13, 2025). Schema/API stability as a first-class concern cited under Principle 6.\n\n## CLI-vs-MCP benchmarks (2026)\n\n- Jannik Reinhard, *CLI Tools vs MCP: Better AI Agents With Less Context* — https://jannikreinhard.com/2026/02/22/why-cli-tools-are-beating-mcp-for-ai-agents/ (Feb 22, 2026). Source for the 28% / 33% / 55K / 35× numbers in `hybrid-mcp-cli.md`.\n- Manveer Chawla, *MCP vs. CLI for AI agents: When to Use Each (A Practical Decision Framework for 2026)* — https://manveerc.substack.com/p/mcp-vs-cli-ai-agents (Mar 8, 2026). Per-integration decision framework; production examples (Claude Code, Cowork) using both transports.\n- Soumyadeb Mitra / RudderStack, *CLI or MCP or both? The design pattern for AI agents managing your data stack* — https://www.rudderstack.com/blog/ai-agents-cli-mcp-design-pattern/ (Mar 18, 2026). The \"writes via CLI, reads via MCP\" split underpinning `hybrid-mcp-cli.md`.\n\n## Pre-agent baseline\n\n- *Scripting with GitHub CLI* — https://github.blog/engineering/engineering-principles/scripting-with-github-cli/ (Mar 11, 2021). `gh api --jq` and structured JSON output as a first-class CLI pattern. The post predates the `gh <resource> --json field1,field2` field-selection flag (cli/cli#1089) but lays out the design philosophy that flag is built on.\n- *Command Line Interface Guidelines* — "},{"path":"references/design-patterns.md","content":"# Design Patterns\n\nReference details for specific design areas. SKILL.md states the principles; this file holds the concrete contracts and rules.\n\n---\n\n## Output envelopes\n\nSuccess:\n\n```json\n{ \"ok\": true, \"data\": {} }\n```\n\nFailure:\n\n```json\n{\n  \"ok\": false,\n  \"error\": {\n    \"code\": \"validation_error\",\n    \"message\": \"Missing required field: email\",\n    \"field\": \"email\",\n    \"retryable\": false\n  }\n}\n```\n\nPartial success (batch operations):\n\n```json\n{\n  \"ok\": \"partial\",\n  \"data\": {\n    \"succeeded\": [\n      { \"id\": \"msg_001\", \"status\": \"sent\" },\n      { \"id\": \"msg_002\", \"status\": \"sent\" }\n    ],\n    \"failed\": [\n      {\n        \"id\": \"msg_003\",\n        \"error\": { \"code\": \"rate_limited\", \"message\": \"Rate limit exceeded for recipient\", \"retryable\": true }\n      }\n    ]\n  }\n}\n```\n\nA batch command that collapses any per-item failure into a top-level `ok: false` forces every agent to re-process its successful items on retry. AWS SQS's `ReportBatchItemFailures` and similar APIs settled on this per-item shape for the same reason. Pair partial success with an idempotency key on the batch as a whole so that re-running with the same key only re-issues the *failed* items.\n\nOptional `meta` slot for observability:\n\n```json\n{\n  \"ok\": true,\n  \"data\": { \"id\": \"abc123\" },\n  \"meta\": {\n    \"request_id\": \"req_8fa9c1\",\n    \"latency_ms\": 412,\n    \"schema_version\": \"1.4.0\"\n  }\n}\n```\n\n`meta` is a freeform slot for telemetry the orchestrator may want without polluting `data`: `request_id` for log correlation, `latency_ms` for SLO tracking, `schema_version` so an agent can detect drift against a cached schema. When the underlying call has a token / quota cost the CLI knows about, surface it here too — `tokens_used`, `quota_remaining`. Optional and additive: agents that don't need it ignore it; agents that do gain observability without an extra round-trip. Aligns with where the OpenTelemetry GenAI semantic conventions are heading for tool-call traces.\n\n---\n\n## Schema versioning\n\nThis is the contract that lets agents cache schemas safely across calls.\n\nEvery response should carry `schema_version` in the optional `meta` block (see envelope above). When an agent's cached schema version (e.g., 1.2.0) doesn't match the CLI's current version (1.4.0), the agent knows to re-discover before re-planning.\n\nSchema introspection responses should declare which CLI version produced them, when each method was introduced, and whether any field is deprecated:\n\n```json\n{\n  \"method\": \"sleep.list\",\n  \"since\": \"1.2.0\",\n  \"deprecated\": false,\n  \"params\": {\n    \"startDate\": { \"type\": \"string\", \"format\": \"date\", \"required\": true },\n    \"endDate\":   { \"type\": \"string\", \"format\": \"date\", \"required\": true },\n    \"page_size\": { \"type\": \"integer\", \"default\": 20, \"max\": 100, \"deprecated\": true, \"replaced_by\": \"pageSize\", \"removed_in\": \"1.5.0\" }\n  }\n}\n```\n\nThis gives agents:\n\n- **Drift detection** — cached version mismatch triggers re-discovery before re-planning.\n- **Deprecation awareness** — agents migrate"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Use when designing, reviewing, or refactoring a CLI that must serve AI agents alongside humans, or when converting an API or SDK into an agent-usable CLI int... Skill: Agent Native Design Owner: agents365-ai Summary: Use when designing, reviewing, or refactoring a CLI that must serve AI agents alongside humans, or when converting an API or SDK into an agent-usable CLI int... Tags: agent-native:1.3.3, cli:1.3.3, interface-design:1.3.3, latest:1.3.5, schema-driven:1.3.3, structured-output:1.3.3 Version history: v1.3.5 | 2026-07-08T17:35:37.031Z | auto - No file changes detecte","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":2142,"uniquenessScore":46,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T01:06:57.128Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T01:06:57.128Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T03:55:06.544Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}