{"id":"91c0a060-8268-493b-a276-34d40a974d92","entityType":"agent","slug":"clawhub-sdk-team-alibabacloud-pai-eas-service-diagnose","name":"Alibabacloud Pai Eas Service Diagnose","canonicalUrl":"https://www.xpersona.co/agent/clawhub-sdk-team-alibabacloud-pai-eas-service-diagnose","canonicalPath":"/agent/clawhub-sdk-team-alibabacloud-pai-eas-service-diagnose","generatedAt":"2026-10-11T08:44:10.590Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T05:56:37.011Z","emptyReason":null},"description":"PAI-EAS service diagnosis and troubleshooting. Diagnose startup failures, error logs, slow responses, instance restarts, OOMKilled, ImagePullBackOff, CrashLo...","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.1K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s173swjet2yrebzqrp6hjkvmy583mxef:alibabacloud-pai-eas-service-diagnose","sourceUrl":"https://clawhub.ai/sdk-team/alibabacloud-pai-eas-service-diagnose","homepage":"https://clawhub.ai/sdk-team/skills/alibabacloud-pai-eas-service-diagnose","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/sdk-team/alibabacloud-pai-eas-service-diagnose","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/sdk-team/skills/alibabacloud-pai-eas-service-diagnose","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":61,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Alibabacloud Pai Eas Service Diagnose technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T05:56:37.011Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T05:56:37.011Z","emptyReason":null},"stars":null,"forks":null,"downloads":1140,"packageName":null,"latestVersion":"0.0.1","tractionLabel":"1.1K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T05:56:36.996Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T05:56:37.011Z","lastCrawledAt":"2026-10-11T05:56:36.996Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T05:56:36.996Z","lastVerifiedAt":null,"highlights":[{"version":"0.0.1","createdAt":"2026-07-13T08:18:20.514Z","changelog":"alibabacloud-pai-eas-service-diagnose v0.0.1 – Initial Release - Provides autonomous diagnosis and troubleshooting for Alibaba Cloud PAI-EAS service issues, including startup failures, error logs, OOMKilled, and service inaccessibility. - Adds strict workflow requirements: always proceed beyond service listing to full diagnosis; ensure every diagnosis ends with a health analysis and recommendations section. - Details installation steps with secure Aliyun CLI installation, plugin setup, and per-session user-agent tracing for observability. - Enforces RAM permission and authentication pre-checks before running diagnostics. - Removes the obsolete skill-card.md file; fully reworks documentation for procedural accuracy and compliance.","fileCount":12,"zipByteSize":29142},{"version":"0.0.1-beta.1","createdAt":"2026-04-22T02:15:27.668Z","changelog":"Initial beta release of PAI-EAS service diagnosis and troubleshooting skill. - Diagnose a wide range of PAI-EAS service issues: startup failures, error logs, slow responses, restarts, OOMKilled, ImagePullBackOff, CrashLoopBackOff, GPU errors, health check failures, and service inaccessibility. - Autonomous, step-by-step diagnostic workflow using the Aliyun CLI and EAS plugin – does not require manual data collection from the user. - Enforces strict security and permissions checks; never requests or prints sensitive credential data. - Comprehensive installation, authentication, and environment pre-check instructions included. - Not intended for deployment or service management operations; focused solely on diagnosis and troubleshooting.","fileCount":12,"zipByteSize":27316}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s173swjet2yrebzqrp6hjkvmy583mxef:alibabacloud-pai-eas-service-diagnose","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s173swjet2yrebzqrp6hjkvmy583mxef:alibabacloud-pai-eas-service-diagnose` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/sdk-team/alibabacloud-pai-eas-service-diagnose before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-sdk-team-alibabacloud-pai-eas-service-diagnose/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-sdk-team-alibabacloud-pai-eas-service-diagnose/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-sdk-team-alibabacloud-pai-eas-service-diagnose/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-sdk-team-alibabacloud-pai-eas-service-diagnose/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-sdk-team-alibabacloud-pai-eas-service-diagnose/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-sdk-team-alibabacloud-pai-eas-service-diagnose/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T08:44:10.588Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-sdk-team-alibabacloud-pai-eas-service-diagnose/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-sdk-team-alibabacloud-pai-eas-service-diagnose/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-sdk-team-alibabacloud-pai-eas-service-diagnose/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-sdk-team-alibabacloud-pai-eas-service-diagnose/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T05:56:37.011Z","emptyReason":null},"readme":"Skill: Alibabacloud Pai Eas Service Diagnose\n\nOwner: sdk-team\n\nSummary: PAI-EAS service diagnosis and troubleshooting. Diagnose startup failures, error logs, slow responses, instance restarts, OOMKilled, ImagePullBackOff, CrashLo...\n\nTags: latest:0.0.1\n\nVersion history:\n\nv0.0.1 | 2026-07-13T08:18:20.514Z | auto\n\nalibabacloud-pai-eas-service-diagnose v0.0.1 – Initial Release\n\n- Provides autonomous diagnosis and troubleshooting for Alibaba Cloud PAI-EAS service issues, including startup failures, error logs, OOMKilled, and service inaccessibility.\n- Adds strict workflow requirements: always proceed beyond service listing to full diagnosis; ensure every diagnosis ends with a health analysis and recommendations section.\n- Details installation steps with secure Aliyun CLI installation, plugin setup, and per-session user-agent tracing for observability.\n- Enforces RAM permission and authentication pre-checks before running diagnostics.\n- Removes the obsolete skill-card.md file; fully reworks documentation for procedural accuracy and compliance.\n\nv0.0.1-beta.1 | 2026-04-22T02:15:27.668Z | auto\n\nInitial beta release of PAI-EAS service diagnosis and troubleshooting skill.\n\n- Diagnose a wide range of PAI-EAS service issues: startup failures, error logs, slow responses, restarts, OOMKilled, ImagePullBackOff, CrashLoopBackOff, GPU errors, health check failures, and service inaccessibility.\n- Autonomous, step-by-step diagnostic workflow using the Aliyun CLI and EAS plugin – does not require manual data collection from the user.\n- Enforces strict security and permissions checks; never requests or prints sensitive credential data.\n- Comprehensive installation, authentication, and environment pre-check instructions included.\n- Not intended for deployment or service management operations; focused solely on diagnosis and troubleshooting.\n\nArchive index:\n\nArchive v0.0.1: 12 files, 29142 bytes\n\nFiles: references/acceptance-criteria.md (7291b), references/api-reference.md (9552b), references/cli-installation-guide.md (12305b), references/diagnosis-flow.md (11852b), references/error-codes.md (7605b), references/health-check.md (6750b), references/ram-policies.md (2809b), references/related-apis.md (6420b), references/verification-method.md (4561b), skill-card.md (2934b), SKILL.md (24331b), _meta.json (156b)\n\nFile v0.0.1:SKILL.md\n\n---\nname: alibabacloud-pai-eas-service-diagnose\ndescription: |\n  PAI-EAS service diagnosis and troubleshooting. Diagnose startup failures, error logs,\n  slow responses, instance restarts, OOMKilled, ImagePullBackOff, CrashLoopBackOff,\n  GPU errors, health check failures, liveness probe issues, service inaccessible.\n  \n  When to use: Diagnose EAS service issues - startup failures, logs, slow responses,\n  restarts, OOMKilled, ImagePullBackOff, CrashLoopBackOff, GPU errors, health checks,\n  service inaccessible, gateway issues, liveness probe failed.\n  Triggers: \"服务启动失败\", \"服务Failed\", \"看日志\", \"实例重启\", \"响应慢\",\n  \"OOMKilled\", \"ImagePullBackOff\", \"CrashLoopBackOff\", \"CUDA out of memory\",\n  \"GPU内存不足\", \"liveness probe\", \"服务访问不了\".\n  \n  Not for: deploying (use service-deploy), managing create/update/delete/stop/restart/scale\n  (use service-manage), listing services (use service-manage), DLC/DSW, non-EAS products.\nlicense: Apache-2.0\nmetadata:\n  version: \"1.0.0\"\n  domain: aiops\n  owner: pai-eas-team\n  contact: pai-eas-agent@alibaba-inc.com\n  tags:\n    - pai-eas\n    - diagnosis\n    - troubleshooting\n    - log-analysis\n    - service-health\n  required_tools:\n    - aliyun\n    - jq\n  required_permissions:\n    - \"eas:DescribeService\"\n    - \"eas:DescribeServiceLog\"\n    - \"eas:DescribeServiceEvent\"\n    - \"eas:DescribeServiceDiagnosis\"\n    - \"eas:DescribeServiceInstanceDiagnosis\"\n    - \"eas:ListServiceInstances\"\n    - \"eas:ListServiceContainers\"\n    - \"eas:ListServices\"\n    - \"eas:DescribeResource\"\n    - \"eas:DescribeGateway\"\n---\n\n# PAI-EAS Service Operations Diagnosis\n\nHelps users diagnose issues with running PAI-EAS services.\n\n> **🔴🔴 [MUST — READ BEFORE ANYTHING ELSE] Two failure modes that cause task failure:**\n>\n> 1. **Listing is NOT diagnosing.** After `ListServices`/`list-services` returns\n>    the target service, you MUST immediately continue and run the FULL diagnostic\n>    workflow (`DescribeService` → `DescribeServiceEvent` → `DescribeServiceLog`\n>    multi-keyword → `ListServiceInstances` → `ListServiceContainers` →\n>    `DescribeServiceDiagnosis`) on a chosen service. NEVER stop after listing.\n> 2. **Always finish with an analysis section.** Every diagnosis MUST end with a\n>    written health-analysis + recommendation section. If the user described a\n>    symptom (e.g. \"CUDA out of memory\") that the live data does NOT show, you\n>    MUST still add a secondary section explaining common causes and fixes for\n>    that described symptom. Never end the turn with only raw command output.\n\n---\n\n## Installation\n\nInstall the Aliyun CLI (>= 3.3.3, required baseline for plugin ecosystem\ncompatibility) using a trusted package manager. On macOS, use Homebrew:\n\n```bash\nbrew install aliyun-cli\naliyun version\n```\n\nOn other platforms, download the versioned release tarball directly from\n`https://aliyuncli.alicdn.com/`, verify its checksum, and extract it manually\nbefore use (see the [CLI Installation Guide](references/cli-installation-guide.md)).\nDo not pipe remote scripts directly into a shell.\n\nVerify CLI version >= 3.3.3, then enable automatic plugin installation and update plugins:\n\n```bash\naliyun configure set --auto-plugin-install true\naliyun plugin update\n```\n\n### User-Agent Setup\n\nBefore running any commands, generate the per-session `SESSION_ID` (see the\nObservability section) and build the unified user-agent string that embeds it.\nPass this string via the `--user-agent` parameter on every CLI command:\n\n```bash\n# Generate once per session (32-char lowercase hex), reuse for every call\nSESSION_ID=$(openssl rand -hex 16)\nUA=\"AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose/${SESSION_ID}\"\naliyun plugin update\n```\n\n> **Detailed Installation Guide**: For more installation options (Windows, ARM64, etc.), see [CLI Installation Guide](references/cli-installation-guide.md).\n\n---\n\n## Observability\n\nAll CLI, SDK, and Terraform calls made by this skill MUST carry a unified\nuser-agent (UA) so that requests can be traced back to this skill and to a\nsingle session.\n\n**UA template declaration (placeholder form):**\n\n```bash\n--user-agent AlibabaCloud-Agent-Skills/{SKILL_NAME}/{session-id}\n```\n\n- `{SKILL_NAME}` — the fixed skill identifier: `alibabacloud-pai-eas-service-diagnose`\n- `{session-id}` — a per-session trace id (see rule below)\n\n**session-id generation rule:**\n\n1. Generate the `session-id` **once per session**, before running any CLI/SDK/Terraform command.\n2. It MUST be a **32-character lowercase hexadecimal** string (128-bit), e.g. generated via:\n   ```bash\n   SESSION_ID=$(openssl rand -hex 16)   # 32 hex chars\n   ```\n3. Reuse the **same** `session-id` for **every** call in the session — CLI (`--user-agent`),\n   SDK (UA/client config), and Terraform (`ALIBABACLOUDSTACK_UA` / provider `ua`), so all\n   requests in one session share a single trace id.\n4. Do NOT regenerate the `session-id` between steps; do NOT hardcode it in the repo.\n\n**Applying the UA (pass `--user-agent` on every CLI command):**\n\n```bash\n# Generate once per session\nSESSION_ID=$(openssl rand -hex 16)\nUA=\"AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose/${SESSION_ID}\"\n\n# Pass it on every command, e.g.:\naliyun eas describe-service --cluster-id <region> --service-name <service> \\\n  --user-agent \"AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose/${SESSION_ID}\"\n```\n\nFor SDK calls, set the same `UA` string as the client user-agent; for Terraform,\nexport it via the provider user-agent configuration. The `{SKILL_NAME}` and\n`{session-id}` segments MUST be identical across all three surfaces.\n\n---\n\n## Environment Variables\n\nNo additional environment variables required. Alibaba Cloud credentials are managed via `aliyun configure`.\n\n---\n\n## Authentication\n\n> **Security Rules:**\n> - **NEVER** read, echo, or print AK/SK values\n> - **NEVER** ask the user to input AK/SK directly\n> - **NEVER** use `aliyun configure set` with literal credential values\n> - **ONLY** use `aliyun configure list` to check credential status\n\n```bash\naliyun configure list\n```\n\nCheck the output for a valid profile (AK, STS, or OAuth identity).\n**If no valid profile exists, STOP here.**\n\n---\n\n## RAM Policy\n\nThe following RAM permissions are required to execute this Skill:\n\n| RAM Action | Description |\n|------------|-------------|\n| `eas:DescribeService` | Query service details |\n| `eas:DescribeServiceLog` | Query service logs |\n| `eas:DescribeServiceEvent` | Query service events |\n| `eas:DescribeServiceDiagnosis` | Service diagnosis report |\n| `eas:DescribeServiceInstanceDiagnosis` | Instance diagnosis |\n| `eas:ListServiceInstances` | List instances |\n| `eas:ListServiceContainers` | List containers |\n| `eas:ListServices` | List services |\n| `eas:DescribeResource` | Resource group details |\n| `eas:DescribeGateway` | Gateway details |\n\n> **[MUST] RAM Permission Pre-check:** Before executing diagnostic commands, verify the user has the required permissions:\n> 1. Use `aliyun ram list-policies-for-user` or check with the user's admin to confirm required permissions\n> 2. Compare against [RAM Policies](references/ram-policies.md)\n> 3. If a command returns `Forbidden` or permission error, abort and prompt the user to grant the missing permission\n\n---\n\n## Autonomous Execution Rules\n\n> **[MUST] This skill is designed for autonomous diagnosis. Follow these rules:**\n>\n> 1. **Do NOT ask the user for information you can find yourself** — Use `list-services` to find services, `describe-service` to get details\n> 2. **If the user provides a region (e.g., \"cn-hangzhou\"), use it directly** — Do NOT ask for confirmation\n> 3. **If the user describes a symptom but doesn't specify a service name**, use `list-services` to find matching services by status\n> 4. **If a command times out or fails, retry once or try a different approach** — Do NOT ask the user to troubleshoot CLI issues\n> 5. **Execute commands directly** — Do NOT ask \"should I proceed?\" before each step\n> 6. **Provide the diagnosis results proactively** — Do NOT wait for the user to confirm each step\n\n---\n\n## CLI Environment Verification\n\n> **[MUST]** Before any diagnosis, verify EAS CLI plugin is installed and core diagnostic APIs are working:\n\n```bash\n# Step 1: Verify EAS plugin is installed\naliyun eas list-services --region cn-hangzhou --max-items 1 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n**If Step 1 fails** with errors like \"pai-eas is not a valid command\" or \"product not supported\":\n1. Run: `aliyun plugin update && aliyun plugin install eas`\n2. If still failing, STOP and inform user: \"EAS CLI plugin not available. Please install via: aliyun plugin install eas\"\n3. **Do NOT proceed with diagnosis until CLI is properly configured**\n4. **Do NOT use ECS/FC/EDAS APIs as workaround for EAS services**\n\n```bash\n# Step 2: Verify DescribeServiceLog API is available (use a known service for testing)\naliyun eas describe-service-log --cluster-id cn-hangzhou --service-name <any-service> --keyword \"error\" --limit 5 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose 2>&1 | grep -q \"can not find api\" && echo \"FATAL: DescribeServiceLog API not available\" || echo \"DescribeServiceLog API verified\"\n```\n\n**If Step 2 fails** with \"can not find api by path\":\n1. Reinstall the official EAS plugin so the latest API metadata is available:\n   `aliyun plugin update && aliyun plugin install eas`\n   The `eas` plugin is a first-party Alibaba Cloud CLI plugin. Only add the\n   `--force` flag if a stale cached version blocks reinstall, since `--force`\n   skips version/dependency validation and should be used sparingly.\n2. If still failing, STOP and inform user: \"DescribeServiceLog API not available in current EAS plugin version. Please update CLI.\"\n3. **Do NOT proceed with log-based diagnosis until API is verified**\n\n**If any command times out:**\n1. Retry once with `--read-timeout 60` flag\n2. If still timing out, try `--region cn-hangzhou --page-size 10` to reduce response size\n3. Do NOT ask the user to troubleshoot network issues — handle it yourself\n\n---\n\n## Product Verification\n\n> **[MUST] Before diagnosing any service, confirm it belongs to PAI-EAS:**\n>\n> This Skill ONLY handles PAI-EAS services. Do NOT use FC, ECS, EDAS, or other product APIs.\n> If the user does not specify a service name, use `list-services` to find the service first.\n\n```bash\n# Find the service in PAI-EAS\naliyun eas list-services --region cn-hangzhou --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | jq '.Services[] | select(.ServiceName == \"my-service\") | {ServiceName, Status}'\n```\n\nIf the service is NOT found in EAS list, STOP and inform the user this is not a PAI-EAS service.\n\n---\n\n## Handling User Description vs Actual Data Mismatch\n\n> If user reports specific error (e.g., \"CUDA out of memory\") but actual service data shows different errors:\n> 1. **Report the discrepancy clearly**: \"You mentioned X, but actual service shows Y\"\n> 2. **Diagnose the actual error found**: Provide analysis for the real error condition (PRIMARY)\n> 3. **Provide generic analysis for user-described issue**: Even if not present in current service, include a section explaining common causes and solutions for the issue user mentioned (SECONDARY)\n> 4. **Do NOT fabricate analysis** for errors that don't exist — but DO provide general troubleshooting guidance\n> 5. **Still complete the full diagnostic workflow**: Check status, events, logs, instances regardless\n\n---\n\n## Core Workflow\n\nWhen a user reports an issue, follow this workflow. **Each step is mandatory:**\n\n> **[MUST] Execution Rules:**\n> - You MUST execute each command directly — do NOT write scripts without executing them\n> - You MUST wait for each command's output before proceeding to the next step\n> - If a command fails or times out, retry once — do NOT ask the user to troubleshoot\n> - If a command still fails after retry, skip to the next diagnostic step and report the error at the end\n> - Do NOT ask the user \"should I proceed?\" or \"please confirm\" — just execute the diagnostic workflow\n\n```\n0. [MUST] CLI Environment Verification → Confirm EAS plugin AND DescribeServiceLog API are working\n1. [MUST] Check service status → DescribeService\n2. [MUST] Check event list → DescribeServiceEvent (NEVER skip this step regardless of issue type)\n   - If this command fails: Retry once with `--read-timeout 60`\n   - If still failing: Document the error in your diagnosis report and continue to next step\n   - NEVER skip this step silently — events are critical for understanding the timeline\n3. [MUST] Check error logs → DescribeServiceLog (MUST call multiple times with different keywords)\n   - MANDATORY keywords: error, oom, killed, exit (4 calls minimum)\n   - GPU issues: Add cuda, gpu keywords (6 calls total)\n   - Do NOT call without --keyword — each call must specify exactly one keyword\n4. [MUST] Check instance status → ListServiceInstances THEN ListServiceContainers\n   - MANDATORY: You MUST call ListServiceContainers even if RestartCount is available in ListServiceInstances\n   - ListServiceContainers provides container-level details (Image, RestartCount, Status) required for diagnosis\n5. [MUST] Run diagnosis → DescribeServiceDiagnosis\n```\n\n> **🔴 [MUST] Listing is NOT diagnosing.** Calling `ListServices` alone\n> does NOT complete the task. After you find the target service(s), you\n> MUST pick one and run the FULL workflow (Steps 1–5) on it. Never stop\n> after listing services.\n>\n> **🔴 [MUST] Self-Verify before finishing.** Before you produce the\n> final answer, confirm you actually called ALL of: DescribeService,\n> DescribeServiceEvent, DescribeServiceLog (multi-keyword),\n> ListServiceInstances, AND ListServiceContainers. If ANY is missing →\n> go back and run it NOW. Do NOT report results until every mandatory\n> API above has been called for the diagnosed service.\n\n### Forced Call Order for Instance & Container Queries\n\n> **[MUST]** Even if `list-service-instances` returns RestartCount, you MUST still call `list-service-containers`\n> to get container-level diagnostic information (Image, RestartCount, Status per container).\n> Do NOT skip this step. Skipping ListServiceContainers will cause evaluation failure.\n>\n> `list-service-containers` requires `--instance-name` parameter.\n> You MUST call `list-service-instances` first to get the instance name, then pass it to `list-service-containers`.\n\n```bash\n# Step 1: Get instance name (MANDATORY first step)\naliyun eas list-service-instances --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '.Instances[] | {InstanceId, InstanceName: .InstanceName, Status}'\n\n# Step 2: Use the instance name from Step 1 (MANDATORY — do NOT skip)\naliyun eas list-service-containers --cluster-id $CLUSTER_ID --service-name $SERVICE \\\n  --instance-name \"<InstanceName from Step 1>\" --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### Mandatory Multi-Keyword Log Queries\n\n> **[MUST]** `--keyword` only supports a single keyword per query. You MUST call `describe-service-log`\n> multiple times with different keywords to cover all relevant error patterns.\n>\n> **Minimum 4 calls required** for every diagnosis: `error`, `oom`, `killed`, `exit`\n>\n> **For GPU-related issues**, add these additional calls: `cuda`, `gpu`\n>\n> **NEVER call DescribeServiceLog without --keyword parameter** — unfiltered logs may miss critical errors.\n> Each call MUST specify exactly one keyword. Calling without --keyword is a violation of this rule.\n\n### One-Click Diagnostic Commands\n\n```bash\nSERVICE=\"my-service\"\nCLUSTER_ID=\"cn-hangzhou\"\n\n# 0. [MUST] Verify service exists in PAI-EAS\naliyun eas list-services --region cn-hangzhou --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | jq '.Services[] | select(.ServiceName == \"'$SERVICE'\") | {ServiceName, Status}'\n\n# 1. Service status\naliyun eas describe-service --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '{Status, RunningInstance, TotalInstance, Message}'\n\n# 2. Recent events (MANDATORY — retry if fails)\naliyun eas describe-service-event --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '.Events[-5:] | .[] | {Time, Type, Reason, Message}' || \\\n  (echo \"ERROR: Failed to retrieve events. Retrying...\" && \\\n   aliyun eas describe-service-event --cluster-id $CLUSTER_ID --service-name $SERVICE --read-timeout 60 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose)\n\n# 3. Error logs — MUST call multiple times with different keywords\naliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name $SERVICE \\\n  --keyword \"error\" --limit 30 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\naliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name $SERVICE \\\n  --keyword \"oom\" --limit 30 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\naliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name $SERVICE \\\n  --keyword \"killed\" --limit 30 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\naliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name $SERVICE \\\n  --keyword \"exit\" --limit 30 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n\n# 4. Instance status (MUST get instance name first, then query containers)\naliyun eas list-service-instances --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '.Instances[] | {InstanceId, InstanceName: .InstanceName, Status}'\n\n# 4b. Container details (requires --instance-name from step 4)\nINSTANCE_NAME=\"<InstanceName from step 4>\"\naliyun eas list-service-containers --cluster-id $CLUSTER_ID --service-name $SERVICE \\\n  --instance-name $INSTANCE_NAME --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n\n# 5. Diagnosis report\naliyun eas describe-service-diagnosis --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n> **Cross-region queries**: When querying services in a region different from your default, specify the `--cluster-id` parameter with the target region:\n> ```bash\n> aliyun eas describe-service --cluster-id cn-shanghai --service-name my-service --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n> ```\n\n### Quick Issue Locator\n\n| Scenario | Typical Symptoms | Detailed Diagnosis Flow |\n|----------|-----------------|------------------------|\n| Service startup failure | Status is Failed / Creating timeout | [Diagnosis Flow - Scenario 1](references/diagnosis-flow.md#scenario-1-service-startup-failure) |\n| Slow service response | Increased request latency, high CPU/memory usage | [Diagnosis Flow - Scenario 2](references/diagnosis-flow.md#scenario-2-slow-service-response) |\n| Frequent instance restarts | RestartCount keeps growing, OOMKilled | [Diagnosis Flow - Scenario 3](references/diagnosis-flow.md#scenario-3-abnormal-instance-restarts) |\n| Service inaccessible | Network unreachable, Token failure, gateway anomaly | [Diagnosis Flow - Scenario 4](references/diagnosis-flow.md#scenario-4-service-inaccessible) |\n| GPU-related issues | CUDA OOM, GPU driver errors | [Diagnosis Flow - Scenario 5](references/diagnosis-flow.md#scenario-5-gpu-related-issues) |\n\n---\n\n## Common Error Keywords\n\n| Keyword | Possible Cause | Reference |\n|---------|---------------|-----------|\n| `OOMKilled` | Out of memory | [Error Codes](references/error-codes.md) |\n| `ImagePullBackOff` | Image pull failure | [Error Codes](references/error-codes.md) |\n| `CrashLoopBackOff` | Container startup failure | [Error Codes](references/error-codes.md) |\n| `OutOfGPU` | Insufficient GPU resources | [Error Codes](references/error-codes.md) |\n| `liveness probe failed` | Health check failure | [Health Check](references/health-check.md) |\n\n---\n\n## Best Practices\n\n1. **[MUST] CLI Environment Pre-check**: Before diagnosis, verify `aliyun eas list-services --region cn-hangzhou --max-items 1` works. If it fails, install EAS plugin first\n2. **[MUST] Product Verification first**: Always confirm the service belongs to PAI-EAS using `list-services`. NEVER use FC, ECS, EDAS, or other product APIs to diagnose EAS services\n3. **[MUST] Check status first**: Get overall status and Message from `DescribeService`\n4. **[MUST] ALWAYS check events**: Use `DescribeServiceEvent` for EVERY diagnosis — regardless of whether the issue is GPU, startup, restart, or any other type. Events are critical for understanding the timeline\n5. **[MUST] Check logs with multiple keywords**: `--keyword` only supports a single keyword per query. You MUST call `DescribeServiceLog` multiple times with different keywords (e.g., `--keyword \"error\"`, `--keyword \"oom\"`, `--keyword \"killed\"`, `--keyword \"exit\"`)\n6. **[MUST] Instance → Container call chain**: `list-service-containers` requires `--instance-name`. You MUST call `list-service-instances` first, then use the returned instance name in `list-service-containers`\n7. **[MUST] Execute commands directly**: Do NOT write scripts without executing them. Do NOT ask the user \"should I proceed?\" — just execute the diagnostic workflow autonomously\n8. **[MUST] Handle data mismatch**: If user describes a specific error but actual service data shows different errors, diagnose the ACTUAL error found — do not fabricate analysis for non-existent errors\n9. **[MUST] Do NOT ask the user for information you can find yourself**: Use `list-services` to find services by status, `describe-service` to get details. Do NOT ask for ServiceName, Cluster ID, or other information that can be obtained programmatically\n\n---\n\n## API and Command Tables\n\n| API | CLI Command | Description |\n|-----|------------|-------------|\n| DescribeService | `aliyun eas describe-service --cluster-id <region> --service-name <name>` | Query service details |\n| DescribeServiceLog | `aliyun eas describe-service-log --cluster-id <region> --service-name <name>` | Query service logs |\n| DescribeServiceEvent | `aliyun eas describe-service-event --cluster-id <region> --service-name <name>` | Query service events |\n| DescribeServiceDiagnosis | `aliyun eas describe-service-diagnosis --cluster-id <region> --service-name <name>` | Service diagnosis report |\n| ListServiceInstances | `aliyun eas list-service-instances --cluster-id <region> --service-name <name>` | List instances |\n| ListServiceContainers | `aliyun eas list-service-containers --cluster-id <region> --service-name <name> --instance-name <instance>` | List containers (requires --instance-name) |\n| DescribeServiceEndpoints | `aliyun eas describe-service-endpoints --cluster-id <region> --service-name <name>` | Service endpoints |\n| DescribeResource | `aliyun eas describe-resource --cluster-id <region> --resource-id <id>` | Resource group details |\n| DescribeGateway | `aliyun eas describe-gateway --cluster-id <region> --gateway-id <id>` | Gateway details |\n\n**Detailed CLI command reference**: [Related APIs](references/related-apis.md)\n\n---\n\n## Reference Links\n\n| Document | Purpose |\n|----------|---------|\n| [CLI Installation Guide](references/cli-installation-guide.md) | CLI installation and configuration |\n| [API Reference](references/api-reference.md) | API fields, jq paths, parameter descriptions |\n| [Error Codes](references/error-codes.md) | Error codes, root cause analysis, solutions |\n| [Diagnosis Flow](references/diagnosis-flow.md) | Scenario-based diagnosis workflows |\n| [Health Check](references/health-check.md) | Health check configuration reference |\n| [Related APIs](references/related-apis.md) | API and CLI command list |\n| [RAM Policies](references/ram-policies.md) | Minimum permission policies |\n| [Verification Method](references/verification-method.md) | Diagnosis result verification |\n| [Acceptance Criteria](references/acceptance-criteria.md) | Skill test acceptance criteria |\n\nFile v0.0.1:_meta.json\n\n{\n  \"ownerId\": \"kn74p5w8ywv6prh40g0s82gmqh83nw54\",\n  \"slug\": \"alibabacloud-pai-eas-service-diagnose\",\n  \"version\": \"0.0.1\",\n  \"publishedAt\": 1783930700514\n}\n\nFile v0.0.1:references/acceptance-criteria.md\n\n# Acceptance Criteria: alibabacloud-pai-eas-service-diagnose\n\n**Scenario**: PAI-EAS Service Diagnosis\n**Purpose**: Skill test acceptance criteria\n\n---\n\n# Correct CLI Command Patterns\n\n## 1. EAS Service Diagnostic Operations\n\n### Correct: Query service status\n\n```bash\naliyun eas describe-service --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### Incorrect: Missing --user-agent\n\n```bash\naliyun eas describe-service --cluster-id cn-hangzhou --service-name my-service\n```\n\n### Incorrect: Using API format instead of plugin mode\n\n```bash\naliyun eas DescribeService --region cn-hangzhou --ServiceName my-service\n```\n\n## 2. Log Query\n\n### Correct: Keyword filtering\n\n```bash\naliyun eas describe-service-log --cluster-id cn-hangzhou --service-name my-service --keyword \"error\" --limit 20 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### Incorrect: Missing --service-name\n\n```bash\naliyun eas describe-service-log --cluster-id cn-hangzhou --keyword \"error\" --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n## 3. Event Query\n\n### Correct: Query service events\n\n```bash\naliyun eas describe-service-event --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### Correct: Filter Warning events\n\n```bash\naliyun eas describe-service-event --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | jq '.Events[] | select(.Type == \"Warning\")'\n```\n\n## 4. Instance Query\n\n### Correct: List instances\n\n```bash\naliyun eas list-service-instances --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### Correct: List containers (requires --instance-name)\n\n```bash\n# First get instance name\naliyun eas list-service-instances --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | jq -r '.Instances[0].InstanceName'\n\n# Then list containers with instance-name\naliyun eas list-service-containers --cluster-id cn-hangzhou --service-name my-service --instance-name my-service-xxxxx-yyyyy --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n## 5. Gateway Query\n\n### Correct: Query gateway details\n\n```bash\naliyun eas describe-gateway --cluster-id cn-hangzhou --gateway-id gw-xxx --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### Incorrect: Missing cluster-id\n\n```bash\naliyun eas describe-gateway --gateway-id gw-xxx --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n---\n\n# Diagnosis Flow Verification\n\n## 1. Service Startup Failure Diagnosis\n\n### Correct flow\n\n```bash\n# 1. Check service status\naliyun eas describe-service --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | jq '{Status, Message}'\n\n# 2. Check failure events\naliyun eas describe-service-event --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | jq '.Events[] | select(.Type == \"Warning\")'\n\n# 3. Check error logs\naliyun eas describe-service-log --cluster-id cn-hangzhou --service-name my-service --keyword \"error\" --limit 20 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### Incorrect: Skipping status check and jumping to logs\n\n```bash\n# Wrong: Not confirming service status first\naliyun eas describe-service-log --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n## 2. jq Filter Verification\n\n### Correct: Extract status information\n\n```bash\naliyun eas describe-service --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | jq '{Status, RunningInstance, TotalInstance, Message}'\n```\n\n### Correct: Filter restart events\n\n```bash\naliyun eas describe-service-event --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | jq '.Events[] | select(.Reason == \"Restarted\")'\n```\n\n### Incorrect: Wrong jq path\n\n```bash\n# Wrong: Field name case error\naliyun eas describe-service --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | jq '{status, runningInstance}'\n```\n\n---\n\n# Security Rule Verification\n\n## Correct: Credential check\n\n```bash\naliyun configure list\n```\n\n## Incorrect: Reading AK/SK\n\n```bash\n# Forbidden: Reading or outputting AK/SK\necho $ALIBABA_CLOUD_ACCESS_KEY_ID\necho $ALIBABA_CLOUD_ACCESS_KEY_SECRET\n```\n\n## Incorrect: Asking user to input AK/SK\n\n```bash\n# Forbidden: Interactive credential input\nread -p \"Enter AccessKey ID: \" AK\nread -p \"Enter AccessKey Secret: \" SK\n```\n\n---\n\n# Parameter Confirmation Requirements\n\n## Correct: All user parameters must be confirmed\n\nThe following parameters must be confirmed before diagnosis:\n- Region\n- ServiceName\n- InstanceId (if querying a specific instance)\n\n## Incorrect: Using default values without confirmation\n\n```bash\n# Wrong: Using default region without asking the user\naliyun eas describe-service --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n---\n\n# Error Keyword Identification\n\n## Correct: Identify common errors\n\n| Keyword | Correct Diagnosis Direction |\n|---------|-----------------------------|\n| `OOMKilled` | Out of memory, suggest increasing memory specification |\n| `ImagePullBackOff` | Image pull failure, check image address and permissions |\n| `CrashLoopBackOff` | Container startup failure, check startup command and configuration |\n| `OutOfGPU` | Insufficient GPU resources, check tp parameter or change specification |\n| `liveness probe failed` | Health check failure, adjust health check parameters |\n\n## Incorrect: Wrong diagnosis direction\n\n```markdown\n# Wrong: Diagnosing OOMKilled as a network issue\nOOMKilled → Check network configuration\n```\n\n---\n\n# Diagnosis Suggestion Verification\n\n## Correct: Provide actionable suggestions\n\n```markdown\nOOMKilled issue suggestions:\n1. Increase memory specification (e.g., upgrade from 16Gi to 32Gi)\n2. Check for memory leaks\n3. For large models, consider model quantization\n```\n\n## Incorrect: Provide vague suggestions\n\n```markdown\n# Wrong: Suggestions not specific\nOOMKilled issue suggestions:\n- Optimize memory usage\n- Check configuration\n```\n\n---\n\n# Boundary Verification\n\n## Correct: Identify Skill boundaries\n\nDiagnostic scenarios:\n- Service status check ✅\n- Log viewing and analysis ✅\n- Event analysis ✅\n- Instance status diagnosis ✅\n\nNon-diagnostic scenarios (should use other Skills):\n- Create/update/delete services → `alibabacloud-pai-eas-service-manage`\n- Start/stop/restart services → `alibabacloud-pai-eas-service-manage`\n- Auto-scaling configuration → `alibabacloud-pai-eas-service-manage`\n- Deploy new services → `alibabacloud-pai-eas-service-deploy`\n\nFile v0.0.1:references/api-reference.md\n\n# Diagnostic API Quick Reference\n\n**Table of Contents**\n\n- [Service Status API](#service-status-api)\n- [Log API](#log-api)\n- [Event API](#event-api)\n- [Instance API](#instance-api)\n- [Diagnosis API](#diagnosis-api)\n- [Resource Group API](#resource-group-api)\n- [Gateway API](#gateway-api)\n- [Quick Diagnostic Command Summary](#quick-diagnostic-command-summary)\n\n## Service Status API\n\n### describe-service (Service Details)\n\n```bash\naliyun eas describe-service --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n**Key fields**:\n\n| Field | Description | jq Path |\n|-------|-------------|---------|\n| Service name | Unique service identifier | `.ServiceName` |\n| Status | Current status | `.Status` |\n| Running instances | Normally running instances | `.RunningInstance` |\n| Total instances | Total instance count | `.TotalInstance` |\n| CPU | CPU cores | `.Cpu` |\n| Memory | Memory size | `.Memory` |\n| GPU | GPU count | `.GPU` |\n| Image | Container image | `.Image` |\n| Error message | Failure reason | `.Message` |\n\n**Status descriptions**:\n\n| Status | Description |\n|--------|-------------|\n| Creating | Being created |\n| Starting | Starting up |\n| Running | Running |\n| Updating | Being updated |\n| Stopping | Stopping |\n| Stopped | Stopped |\n| Failed | Failed |\n\n---\n\n### list-services (Service List)\n\n```bash\naliyun eas list-services --cluster-id $CLUSTER_ID --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n**Common filters**:\n\n```bash\n# Filter failed services\naliyun eas list-services --cluster-id $CLUSTER_ID --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '.Services[] | select(.Status == \"Failed\")'\n\n# Filter services by resource group\naliyun eas list-services --cluster-id $CLUSTER_ID --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '.Services[] | select(.ResourceId == \"eas-r-xxx\")'\n```\n\n---\n\n### describe-service-endpoints (Service Endpoints)\n\n```bash\naliyun eas describe-service-endpoints --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n**Response fields**:\n\n| Field | Description |\n|-------|-------------|\n| `.InternetEndpoint` | Public endpoint |\n| `.IntranetEndpoint` | Internal endpoint |\n| `.Token` | Access Token |\n\n---\n\n## Log API\n\n### describe-service-log (Service Logs)\n\n```bash\n# Basic query\naliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name $SERVICE --limit 100 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n\n# Keyword filtering\naliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name $SERVICE \\\n  --keyword \"error\" --limit 50 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n\n# Time range\naliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name $SERVICE \\\n  --start-time \"2026-03-19T00:00:00Z\" --end-time \"2026-03-19T23:59:59Z\" --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n\n# Specific instance\naliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name $SERVICE \\\n  --instance-id i-xxx --limit 50 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n**Parameter descriptions**:\n\n| Parameter | Description | Default |\n|-----------|-------------|---------|\n| `--limit` | Number of entries to return | 100 |\n| `--keyword` | Single keyword filter (does not support pipe-separated multiple keywords; run multiple queries for different keywords) | - |\n| `--start-time` | Start time | - |\n| `--end-time` | End time | - |\n| `--instance-id` | Specific instance | - |\n\n---\n\n## Event API\n\n### describe-service-event (Service Events)\n\n```bash\naliyun eas describe-service-event --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n**Key fields**:\n\n| Field | Description |\n|-------|-------------|\n| `.Time` | Event time |\n| `.Type` | Event type |\n| `.Reason` | Event reason |\n| `.Message` | Event details |\n\n**Event types**:\n\n| Type | Description |\n|------|-------------|\n| Normal | Normal event |\n| Warning | Warning event |\n\n**Common Reasons**:\n\n| Reason | Description |\n|--------|-------------|\n| Started | Started successfully |\n| Failed | Failed |\n| Scaled | Scaled up/down |\n| Restarted | Restarted |\n| Updated | Updated |\n| Unhealthy | Unhealthy |\n\n---\n\n## Instance API\n\n### list-service-instances (Instance List)\n\n```bash\naliyun eas list-service-instances --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n**Response fields**:\n\n| Field | Description |\n|-------|-------------|\n| `.InstanceId` | Instance ID |\n| `.Status` | Instance status |\n| `.IpAddress` | Instance IP |\n| `.CreateTime` | Creation time |\n| `.CpuUtilization` | CPU utilization |\n| `.MemoryUtilization` | Memory utilization |\n\n**Instance statuses**:\n\n| Status | Description |\n|--------|-------------|\n| Pending | Pending |\n| Creating | Being created |\n| Running | Running |\n| Failed | Failed |\n| Stopping | Stopping |\n| Stopped | Stopped |\n\n---\n\n### list-service-containers (Container List)\n\n> **Note**: `--instance-name` is a required parameter. Get it from `list-service-instances` first.\n\n```bash\n# Step 1: Get instance name\nINSTANCE_NAME=$(aliyun eas list-service-instances --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | jq -r '.Instances[0].InstanceName')\n\n# Step 2: List containers\naliyun eas list-service-containers --cluster-id $CLUSTER_ID --service-name $SERVICE --instance-name \"$INSTANCE_NAME\" --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n**Key fields**:\n\n| Field | Description |\n|-------|-------------|\n| `.ContainerId` | Container ID |\n| `.InstanceId` | Parent instance |\n| `.Status` | Container status |\n| `.RestartCount` | Restart count |\n\n---\n\n## Diagnosis API\n\n### describe-service-diagnosis (Service Diagnosis)\n\n```bash\naliyun eas describe-service-diagnosis --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n**Diagnosis items**:\n\n| Diagnosis Item | Description |\n|---------------|-------------|\n| Resource usage | CPU/memory utilization |\n| Network status | Network connectivity |\n| Health status | Health check status |\n| Storage status | Storage mount status |\n\n---\n\n### describe-service-instance-diagnosis (Instance Diagnosis)\n\n```bash\naliyun eas describe-service-instance-diagnosis --cluster-id $CLUSTER_ID \\\n  --service-name $SERVICE --instance-id i-xxx --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n---\n\n## Resource Group API\n\n### describe-resource (Resource Group Details)\n\n```bash\naliyun eas describe-resource --cluster-id $CLUSTER_ID --resource-id eas-r-xxx --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n**Key fields**:\n\n| Field | Description |\n|-------|-------------|\n| `.Name` | Resource group name |\n| `.Status` | Resource group status |\n| `.TotalNodes` | Total node count |\n| `.HealthyNodes` | Healthy node count |\n| `.Nodes[]` | Node list |\n\n---\n\n### list-resources (Resource Group List)\n\n```bash\naliyun eas list-resources --cluster-id $CLUSTER_ID --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n---\n\n## Gateway API\n\n### describe-gateway (Gateway Details)\n\n```bash\naliyun eas describe-gateway --cluster-id $CLUSTER_ID --gateway-id gw-xxx --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n# or\naliyun eas describe-gateway --cluster-id $CLUSTER_ID --gateway-name my-gateway --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n**Key fields**:\n\n| Field | Description |\n|-------|-------------|\n| `.Name` | Gateway name |\n| `.LoadBalancerList[0].Status` | Load balancer status |\n| `.LoadBalancerList[0].Address` | Load balancer address |\n\n---\n\n### list-gateways (Gateway List)\n\n```bash\naliyun eas list-gateways --cluster-id $CLUSTER_ID --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n---\n\n## Quick Diagnostic Command Summary\n\n```bash\nCLUSTER_ID=\"cn-hangzhou\"\nSERVICE=\"my-service\"\n\n# Quick view service status\nalias ds='aliyun eas describe-service --cluster-id $CLUSTER_ID --service-name'\n\n# Quick view logs\nalias dl='aliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name'\n\n# Quick view events\nalias de='aliyun eas describe-service-event --cluster-id $CLUSTER_ID --service-name'\n\n# Quick view instances\nalias di='aliyun eas list-service-instances --cluster-id $CLUSTER_ID --service-name'\n\n# Quick diagnose\nalias dd='aliyun eas describe-service-diagnosis --cluster-id $CLUSTER_ID --service-name'\n\n# Usage examples\nds $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | jq '{Status, RunningInstance, TotalInstance}'\ndl $SERVICE --keyword \"error\" --limit 20 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\nde $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | jq '.Events[-5:]'\ndi $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | jq '.Instances[].Status'\ndd $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | jq '.DiagnosisItems[]'\n```\n\nFile v0.0.1:references/cli-installation-guide.md\n\n# Aliyun CLI Installation & Configuration Guide\n\nComplete guide for installing and configuring Aliyun CLI.\n\n> **Aliyun CLI 3.3.3+**: Supports installing and using all published Alibaba Cloud product plugins. Make sure to upgrade to 3.3.3 or later for full plugin ecosystem coverage.\n\n## Installation\n\n> **Note on `sudo mv ... /usr/local/bin/`**: The binary-install steps below use\n> `sudo` only to place the `aliyun` executable into the system-wide `/usr/local/bin`\n> directory, which requires administrator privileges. This is the standard,\n> well-understood way to make a CLI available on `PATH` for all users. If you\n> prefer NOT to use elevated privileges, install into a user-local directory\n> instead and add it to your `PATH` — no `sudo` required:\n>\n> ```bash\n> mkdir -p \"$HOME/.local/bin\"\n> mv aliyun \"$HOME/.local/bin/\"\n> export PATH=\"$HOME/.local/bin:$PATH\"   # add to ~/.bashrc or ~/.zshrc to persist\n> ```\n\n### macOS\n\n**Using Homebrew (Recommended)**\n```bash\nbrew install aliyun-cli\n# Upgrade to latest\nbrew upgrade aliyun-cli\n\n# Verify version (>= 3.3.3)\naliyun version\n```\n\n**Using Binary**\n```bash\n# Download\nwget https://aliyuncli.alicdn.com/aliyun-cli-macosx-latest-amd64.tgz\n\n# Extract\ntar -xzf aliyun-cli-macosx-latest-amd64.tgz\n\n# Move to PATH\nsudo mv aliyun /usr/local/bin/\n\n# Verify\naliyun version\n```\n\n### Linux\n\n**Debian/Ubuntu**\n```bash\n# Download\nwget https://aliyuncli.alicdn.com/aliyun-cli-linux-latest-amd64.tgz\n\n# Extract and install\ntar -xzf aliyun-cli-linux-latest-amd64.tgz\nsudo mv aliyun /usr/local/bin/\n\n# Verify\naliyun version\n```\n\n**CentOS/RHEL**\n```bash\n# Download\nwget https://aliyuncli.alicdn.com/aliyun-cli-linux-latest-amd64.tgz\n\n# Extract and install\ntar -xzf aliyun-cli-linux-latest-amd64.tgz\nsudo mv aliyun /usr/local/bin/\n\n# Verify\naliyun version\n```\n\n**ARM64 Architecture**\n```bash\n# Download ARM64 version\nwget https://aliyuncli.alicdn.com/aliyun-cli-linux-latest-arm64.tgz\n\n# Extract and install\ntar -xzf aliyun-cli-linux-latest-arm64.tgz\nsudo mv aliyun /usr/local/bin/\n```\n\n### Windows\n\n**Using Binary**\n1. Download from: https://aliyuncli.alicdn.com/aliyun-cli-windows-latest-amd64.zip\n2. Extract the ZIP file\n3. Add the directory to your PATH environment variable\n4. Open new Command Prompt or PowerShell\n5. Verify: `aliyun version`\n\n**Using PowerShell**\n```powershell\n# Download\nInvoke-WebRequest -Uri \"https://aliyuncli.alicdn.com/aliyun-cli-windows-latest-amd64.zip\" -OutFile \"aliyun-cli.zip\"\n\n# Extract\nExpand-Archive -Path aliyun-cli.zip -DestinationPath C:\\aliyun-cli\n\n# Add to PATH (requires admin privileges)\n$env:Path += \";C:\\aliyun-cli\"\n[Environment]::SetEnvironmentVariable(\"Path\", $env:Path, [System.EnvironmentVariableTarget]::Machine)\n\n# Verify\naliyun version\n```\n\n## Configuration\n\n### Quick Start\n\n```bash\naliyun configure set \\\n  --mode AK \\\n  --access-key-id <your-access-key-id> \\\n  --access-key-secret <your-access-key-secret> \\\n  --region cn-hangzhou\n```\n\nAll `aliyun configure` commands support non-interactive flags, which is the recommended approach —\nit works in scripts, CI/CD pipelines, and agent-driven automation without hanging on stdin prompts.\n\n**Where to Get Access Keys**\n\n1. Log in to Aliyun Console: https://ram.console.aliyun.com/\n2. Navigate to: AccessKey Management\n3. Create a new AccessKey pair\n4. Save the secret immediately — it's only shown once\n\n### Configuration Modes\n\nAliyun CLI supports 6 authentication modes. All examples below use non-interactive flags.\n\n#### 1. AK Mode (Access Key)\n\nMost common mode for personal accounts and scripts.\n\n```bash\naliyun configure set \\\n  --mode AK \\\n  --access-key-id <your-access-key-id> \\\n  --access-key-secret <your-access-key-secret> \\\n  --region cn-hangzhou\n```\n\nConfiguration is stored in `~/.aliyun/config.json`:\n\n```json\n{\n  \"current\": \"default\",\n  \"profiles\": [\n    {\n      \"name\": \"default\",\n      \"mode\": \"AK\",\n      \"access_key_id\": \"<your-access-key-id>\",\n      \"access_key_secret\": \"<your-access-key-secret>\",\n      \"region_id\": \"cn-hangzhou\",\n      \"output_format\": \"json\",\n      \"language\": \"en\"\n    }\n  ]\n}\n```\n\n#### 2. StsToken Mode (Temporary Credentials)\n\nFor short-lived access (tokens expire in 1-12 hours).\n\n```bash\naliyun configure set \\\n  --mode StsToken \\\n  --access-key-id <your-access-key-id> \\\n  --access-key-secret <your-access-key-secret> \\\n  --sts-token v1.0:XXXXXXXXXXXXXXXX \\\n  --region cn-hangzhou\n```\n\nUse cases: CI/CD pipelines, temporary access for external contractors, cross-account access.\n\n#### 3. RamRoleArn Mode (Assume RAM Role)\n\nAssume a RAM role for elevated or cross-account access.\n\n```bash\naliyun configure set \\\n  --mode RamRoleArn \\\n  --access-key-id <your-access-key-id> \\\n  --access-key-secret <your-access-key-secret> \\\n  --ram-role-arn acs:ram::123456789012:role/AdminRole \\\n  --role-session-name my-session \\\n  --region cn-hangzhou\n```\n\nUse cases: cross-account resource access, temporary elevated privileges, role-based access control.\n\n#### 4. EcsRamRole Mode (ECS Instance RAM Role)\n\nUse the RAM role attached to an ECS instance — no credentials needed.\n\n```bash\naliyun configure set \\\n  --mode EcsRamRole \\\n  --ram-role-name MyEcsRole \\\n  --region cn-hangzhou\n```\n\nRequirements: must be running on an ECS instance with a RAM role attached.\n\nUse cases: scripts and automation running on ECS instances.\n\n#### 5. RsaKeyPair Mode (RSA Key Pair)\n\nUse RSA key pair for authentication (generate key pair in Aliyun Console first).\n\n```bash\naliyun configure set \\\n  --mode RsaKeyPair \\\n  --private-key /path/to/private-key.pem \\\n  --key-pair-name my-key-pair \\\n  --region cn-hangzhou\n```\n\n#### 6. RamRoleArnWithEcs Mode (ECS + RAM Role)\n\nCombine ECS instance role with RAM role assumption for cross-account access from ECS.\n\n```bash\naliyun configure set \\\n  --mode RamRoleArnWithEcs \\\n  --ram-role-name MyEcsRole \\\n  --ram-role-arn acs:ram::123456789012:role/TargetRole \\\n  --role-session-name my-session \\\n  --region cn-hangzhou\n```\n\n### Environment Variables\n\n**Highest priority** - overrides config file\n\n**Access Key Mode**\n```bash\nexport ALIBABA_CLOUD_ACCESS_KEY_ID=your_access_key_id\nexport ALIBABA_CLOUD_ACCESS_KEY_SECRET=your_access_key_secret\nexport ALIBABA_CLOUD_REGION_ID=cn-hangzhou\n```\n\n**STS Token Mode**\n```bash\nexport ALIBABA_CLOUD_ACCESS_KEY_ID=your_access_key_id\nexport ALIBABA_CLOUD_ACCESS_KEY_SECRET=your_access_key_secret\nexport ALIBABA_CLOUD_SECURITY_TOKEN=your_sts_token\nexport ALIBABA_CLOUD_REGION_ID=cn-hangzhou\n```\n\n**ECS RAM Role Mode**\n```bash\nexport ALIBABA_CLOUD_ECS_METADATA=role_name\n```\n\n**Use Case**:\n- CI/CD pipelines\n- Docker containers\n- Temporary credential override\n\n### Managing Multiple Profiles\n\n**Create Named Profiles**\n\n```bash\naliyun configure set --profile projectA \\\n  --mode AK \\\n  --access-key-id <your-projectA-access-key-id> \\\n  --access-key-secret <your-projectA-access-key-secret> \\\n  --region cn-hangzhou\n\naliyun configure set --profile projectB \\\n  --mode AK \\\n  --access-key-id <your-projectB-access-key-id> \\\n  --access-key-secret <your-projectB-access-key-secret> \\\n  --region cn-shanghai\n```\n\n**Use Specific Profile**\n\n```bash\naliyun ecs describe-instances --profile projectA\n\nexport ALIBABA_CLOUD_PROFILE=projectA\naliyun ecs describe-instances   # Uses projectA\n```\n\n**List and Switch Profiles**\n\n```bash\naliyun configure list                      # List all profiles\naliyun configure set --current projectA    # Switch default profile\n```\n\n### Credential Priority\n\nCredentials are loaded in this order (first found wins):\n\n1. **Command-line flag**: `--profile <name>`\n2. **Environment variable**: `ALIBABA_CLOUD_PROFILE`\n3. **Environment credentials**: `ALIBABA_CLOUD_ACCESS_KEY_ID`, etc.\n4. **Configuration file**: `~/.aliyun/config.json` (current profile)\n5. **ECS Instance RAM Role**: If running on ECS with attached role\n\n## Verification\n\n### Test Authentication\n\n```bash\n# Basic test - list regions\naliyun ecs describe-regions\n\n# Expected output: JSON array of regions\n```\n\n**If successful**, you'll see:\n```json\n{\n  \"Regions\": {\n    \"Region\": [\n      {\n        \"RegionId\": \"cn-hangzhou\",\n        \"RegionEndpoint\": \"ecs.cn-hangzhou.aliyuncs.com\",\n        \"LocalName\": \"East China 1 (Hangzhou)\"\n      },\n      ...\n    ]\n  },\n  \"RequestId\": \"...\"\n}\n```\n\n**If failed**, you'll see error messages:\n- `InvalidAccessKeyId.NotFound` - Wrong Access Key ID\n- `SignatureDoesNotMatch` - Wrong Access Key Secret\n- `InvalidSecurityToken.Expired` - STS token expired (for StsToken mode)\n- `Forbidden.RAM` - Insufficient permissions\n\n### Debug Configuration\n\n```bash\n# Show current configuration\naliyun configure get\n\n# Test with debug logging\naliyun ecs describe-regions --log-level=debug\n\n# Check credential provider\naliyun configure get mode\n```\n\n## Security Best Practices\n\n### 1. Use RAM Users (Not Root Account)\n\n❌ **Don't**: Use Aliyun root account credentials\n✅ **Do**: Create RAM users with specific permissions\n\n```bash\n# Create RAM user in console\n# Attach only necessary policies\n# Use RAM user's access keys\n```\n\n### 2. Principle of Least Privilege\n\nGrant only the minimum permissions needed:\n\n```bash\n# Example: Read-only ECS access\n# Attach policy: AliyunECSReadOnlyAccess\n```\n\n### 3. Rotate Access Keys Regularly\n\n```bash\n# Create new access key in RAM Console, then update configuration\naliyun configure set --access-key-id NEW_KEY --access-key-secret NEW_SECRET\n# Delete old access key from console\n```\n\n### 4. Use STS Tokens for Temporary Access\n\n```bash\naliyun configure set --mode StsToken \\\n  --access-key-id XXXX --access-key-secret XXXX \\\n  --sts-token XXXX --region cn-hangzhou\n```\n\n### 5. Use ECS RAM Roles When Possible\n\n```bash\naliyun configure set --mode EcsRamRole --ram-role-name MyRole --region cn-hangzhou\n```\n\n### 6. Never Commit Credentials\n\n```bash\n# Add to .gitignore\necho \"~/.aliyun/config.json\" >> .gitignore\n\n# Use environment variables in CI/CD instead\n```\n\n### 7. Secure Config File\n\n```bash\n# Restrict permissions\nchmod 600 ~/.aliyun/config.json\n```\n\n## Troubleshooting\n\n### Issue: Command Not Found\n\n```bash\n# Check installation\nwhich aliyun\n\n# Check PATH\necho $PATH\n\n# Reinstall or add to PATH\n```\n\n### Issue: Authentication Failed\n\n```bash\n# Verify configuration\naliyun configure get\n\n# Test with debug\naliyun ecs describe-regions --log-level=debug\n\n# Check credentials in console\n# Verify access key is active\n```\n\n### Issue: Permission Denied\n\n```bash\n# Error: Forbidden.RAM\n\n# Check RAM user permissions\n# Attach necessary policies in RAM console\n# Example: AliyunECSFullAccess for ECS operations\n```\n\n### Issue: STS Token Expired\n\n```bash\n# Error: InvalidSecurityToken.Expired\n\n# Reconfigure with new token\naliyun configure set --mode StsToken \\\n  --access-key-id XXXX --access-key-secret XXXX \\\n  --sts-token NEW_TOKEN --region cn-hangzhou\n```\n\n### Issue: Wrong Region\n\n```bash\n# Some resources may not exist in the specified region\n\n# Check available regions\naliyun ecs describe-regions\n\n# Update default region\naliyun configure set region cn-shanghai\n```\n\n## Advanced Configuration\n\n### Custom Endpoint\n\n```bash\n# Use custom or private endpoint\nexport ALIBABA_CLOUD_ECS_ENDPOINT=ecs-vpc.cn-hangzhou.aliyuncs.com\n```\n\n### Proxy Settings\n\n```bash\n# HTTP proxy\nexport HTTP_PROXY=http://proxy.example.com:8080\nexport HTTPS_PROXY=http://proxy.example.com:8080\n\n# No proxy for specific domains\nexport NO_PROXY=localhost,127.0.0.1,.aliyuncs.com\n```\n\n### Timeout Settings\n\n```bash\n# Connection timeout (default: 10s)\nexport ALIBABA_CLOUD_CONNECT_TIMEOUT=30\n\n# Read timeout (default: 10s)\nexport ALIBABA_CLOUD_READ_TIMEOUT=30\n```\n\n## Next Steps\n\nAfter installation and configuration:\n\n1. **Install plugins** for services you need (v3.3.3+ supports all published product plugins):\n   ```bash\n   aliyun plugin install --names ecs vpc rds\n\n   # List all available plugins\n   aliyun plugin list-remote\n   ```\n\n2. **Explore commands**:\n   ```bash\n   aliyun ecs --help\n   aliyun fc --help\n   ```\n\n3. **Read documentation**:\n   - [Command Syntax Guide](./command-syntax.md)\n   - [Global Flags Reference](./global-flags.md)\n   - [Common Scenarios](./common-scenarios.md)\n\n## References\n\n- Official Documentation: https://help.aliyun.com/zh/cli/\n- RAM Console: https://ram.console.aliyun.com/\n- Access Key Management: https://ram.console.aliyun.com/manage/ak\n- Plugin Repository: https://github.com/aliyun/aliyun-cli\n\nFile v0.0.1:references/diagnosis-flow.md\n\n# Diagnosis Flow Guide\n\n**Table of Contents**\n\n- [Quick Diagnosis Entry](#quick-diagnosis-entry)\n- [Scenario 1: Service Startup Failure](#scenario-1-service-startup-failure)\n- [Scenario 2: Slow Service Response](#scenario-2-slow-service-response)\n- [Scenario 3: Abnormal Instance Restarts](#scenario-3-abnormal-instance-restarts)\n- [Scenario 4: Service Inaccessible](#scenario-4-service-inaccessible)\n- [Scenario 5: GPU Related Issues](#scenario-5-gpu-related-issues)\n- [Diagnosis Report Generation](#diagnosis-report-generation)\n\n## Quick Diagnosis Entry\n\nWhen a user reports an issue, first confirm the service name and region, then follow this workflow:\n\n```\n1. Check service status → Confirm current state\n2. Check event list → Understand recent changes\n3. Check error logs → Locate specific errors\n4. Check instance status → Verify instance health\n5. Run diagnosis → Get diagnosis report\n```\n\n---\n\n## Scenario 1: Service Startup Failure\n\n### Diagnosis Flow\n\n```bash\nSERVICE=\"my-service\"\nREGION=\"cn-hangzhou\"\n\n# Step 1: Check service status\naliyun eas describe-service --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '{Status, Message, RunningInstance, TotalInstance}'\n\n# Step 2: View recent events\naliyun eas describe-service-event --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '.Events[-5:] | .[] | {Time, Type, Reason, Message}'\n\n# Step 3: View error logs\naliyun eas describe-service-log --cluster-id $REGION --service-name $SERVICE \\\n  --keyword \"error\" --limit 30 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n\n# Step 4: Check instance status\naliyun eas list-service-instances --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '.Instances[] | {InstanceId, Status, IpAddress}'\n\n# Step 5: Run diagnosis\naliyun eas describe-service-diagnosis --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### Common Causes and Solutions\n\n| Error Symptom | Possible Cause | Solution |\n|--------------|---------------|----------|\n| ImagePullBackOff | Image pull failure | Check image address and permissions |\n| CrashLoopBackOff | Container startup failure | Check startup command and logs |\n| OOMKilled | Out of memory | Increase memory specification |\n| OutOfGPU | Insufficient GPU resources | Check tp parameter or change specification |\n| Pending | Waiting for resources | Check resource group inventory |\n\n---\n\n## Scenario 2: Slow Service Response\n\n### Diagnosis Flow\n\n```bash\nSERVICE=\"my-service\"\nREGION=\"cn-hangzhou\"\n\n# Step 1: Check instance running status\naliyun eas describe-service --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '{RunningInstance, TotalInstance, Cpu, Memory}'\n\n# Step 2: Check instance resource usage\naliyun eas list-service-instances --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '.Instances[] | {InstanceId, Status, CpuUtilization, MemoryUtilization}'\n\n# Step 3: View slow query logs\naliyun eas describe-service-log --cluster-id $REGION --service-name $SERVICE \\\n  --keyword \"slow\" --limit 20 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n\n# Step 4: Check for error logs\naliyun eas describe-service-log --cluster-id $REGION --service-name $SERVICE \\\n  --keyword \"error\" --limit 20 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### Performance Issue Analysis\n\n| Symptom | Possible Cause | Solution |\n|---------|---------------|----------|\n| High CPU usage | Compute-intensive tasks | Increase CPU cores or scale out |\n| High memory usage | Model loading, data caching | Increase memory specification |\n| Low GPU utilization | Inference batch too small | Increase batch size |\n| Request queuing | Insufficient instances | Enable auto-scaling |\n\n### Auto-Scaling Configuration Check\n\n```bash\n# Check if auto-scaling is enabled\naliyun eas describe-service --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '.AutoScale'\n\n# Check current instance count\naliyun eas describe-service --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '{RunningInstance, DesiredInstance, MinInstance, MaxInstance}'\n```\n\n---\n\n## Scenario 3: Abnormal Instance Restarts\n\n### Diagnosis Flow\n\n```bash\nSERVICE=\"my-service\"\nREGION=\"cn-hangzhou\"\n\n# Step 1: Check restart events\naliyun eas describe-service-event --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '.Events[] | select(.Reason == \"Restarted\") | {Time, Message}'\n\n# Step 2: Check container restart count (requires --instance-name, get it from list-service-instances first)\nINSTANCE_NAME=$(aliyun eas list-service-instances --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | jq -r '.Instances[0].InstanceName')\naliyun eas list-service-containers --cluster-id $REGION --service-name $SERVICE --instance-name \"$INSTANCE_NAME\" --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '.Containers[] | {ContainerId, RestartCount, Status}'\n\n# Step 3: View pre-restart logs\naliyun eas describe-service-log --cluster-id $REGION --service-name $SERVICE \\\n  --keyword \"killed\" --limit 30 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n\n# Step 4: Check health check configuration\naliyun eas describe-service --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '{LivenessCheck, ReadinessCheck}'\n```\n\n### Restart Cause Analysis\n\n| Restart Cause | Symptom | Solution |\n|--------------|---------|----------|\n| OOMKilled | Memory usage exceeds limit | Increase memory specification |\n| Liveness failure | Health check keeps failing | Adjust health check parameters |\n| Application crash | Program exits abnormally | Investigate application logs |\n| Resource pressure | Insufficient node resources | Migrate to another node |\n\n---\n\n## Scenario 4: Service Inaccessible\n\n### Diagnosis Flow\n\n```bash\nSERVICE=\"my-service\"\nREGION=\"cn-hangzhou\"\n\n# Step 1: Check service status\naliyun eas describe-service --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '{Status, RunningInstance}'\n\n# Step 2: Get service endpoints\naliyun eas describe-service-endpoints --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n\n# Step 3: Check gateway status (if using dedicated gateway)\naliyun eas describe-service --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq -r '.ExtraData.GatewayName' | xargs -I {} \\\n  aliyun eas describe-gateway --cluster-id $REGION --gateway-name {} --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '{Name, Status: .LoadBalancerList[0].Status}'\n\n# Step 4: Check security group configuration (VPC direct connect scenario)\naliyun eas describe-service --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '.ExtraData.SecurityGroupId'\n```\n\n### Network Issue Troubleshooting\n\n| Access Method | Checkpoints |\n|--------------|-------------|\n| Public gateway | Gateway status, Token, security group |\n| VPC direct connect | Security group, VPC configuration, service endpoints |\n| Dedicated gateway | Gateway status, NLB status, security group |\n\n### Connectivity Test\n\n```bash\n# Get endpoint and Token\nENDPOINT=$(aliyun eas describe-service-endpoints --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | jq -r '.InternetEndpoint')\nTOKEN=$(aliyun eas describe-service-endpoints --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | jq -r '.Token')\n\n# Test connectivity\ncurl --connect-timeout 10 --max-time 30 -H \"Authorization: $TOKEN\" \"$ENDPOINT/health\"\n```\n\n---\n\n## Scenario 5: GPU Related Issues\n\n### Diagnosis Flow\n\n```bash\nSERVICE=\"my-service\"\nREGION=\"cn-hangzhou\"\n\n# Step 1: Check GPU specification\naliyun eas describe-service --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '{Cpu, Memory, GPU, GPUType}'\n\n# Step 2: Check instance GPU status\naliyun eas list-service-instances --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '.Instances[] | {InstanceId, Status, GPUUtilization, GPUMemoryUtilization}'\n\n# Step 3: View GPU-related errors\naliyun eas describe-service-log --cluster-id $REGION --service-name $SERVICE \\\n  --keyword \"cuda\" --limit 30 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### GPU Issue Analysis\n\n| Issue | Symptom | Solution |\n|-------|---------|----------|\n| Insufficient GPU memory | CUDA out of memory | Reduce batch size or upgrade GPU |\n| Low GPU utilization | Poor inference efficiency | Increase batch size or use multi-stream |\n| GPU driver error | Driver error | Check driver version compatibility |\n| Multi-GPU communication failure | NCCL error | Check network configuration, tp parameter |\n\n### Tensor Parallel Check\n\n```bash\n# Check tp parameter in command\naliyun eas describe-service --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq -r '.Image.Command' | grep -oP '(?<=--tp\\s)\\d+'\n\n# Verify GPU count matches\naliyun eas describe-service --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '{GPU, Command}'\n```\n\n---\n\n## Diagnosis Report Generation\n\n### One-Click Diagnosis Script\n\n```bash\n#!/bin/bash\nSERVICE=\"my-service\"\nREGION=\"cn-hangzhou\"\n\necho \"=== PAI-EAS Service Diagnosis Report ===\"\necho \"Service: $SERVICE\"\necho \"Region: $REGION\"\necho \"\"\n\necho \"## Service Status\"\naliyun eas describe-service --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '{Status, RunningInstance, TotalInstance, Message}'\necho \"\"\n\necho \"## Recent Events (Last 5)\"\naliyun eas describe-service-event --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '.Events[-5:] | .[] | {Time, Type, Reason, Message}'\necho \"\"\n\necho \"## Instance Status\"\naliyun eas list-service-instances --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '.Instances[] | {InstanceId, Status}'\necho \"\"\n\necho \"## Error Logs (Last 10)\"\naliyun eas describe-service-log --cluster-id $REGION --service-name $SERVICE \\\n  --keyword \"error\" --limit 10 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | jq '.'\necho \"\"\n\necho \"## Diagnosis Suggestions\"\naliyun eas describe-service-diagnosis --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '.DiagnosisItems[] | {Name, Status, Suggestion}'\n```\n\nFile v0.0.1:references/error-codes.md\n\n# Error Code Reference\n\n**Table of Contents**\n\n- [Container Startup Errors](#container-startup-errors)\n- [Service State Errors](#service-state-errors)\n- [Network Access Errors](#network-access-errors)\n- [Resource Group Errors](#resource-group-errors)\n- [Health Check Errors](#health-check-errors)\n\n## Container Startup Errors\n\n### ImagePullBackOff (Image Pull Failure)\n\n**Error message**:\n```\nFailed to pull image \"xxx\": rpc error: code = Unknown desc = Error response from daemon: pull access denied\n```\n\n**Possible causes**:\n1. Incorrect image address\n2. Insufficient image registry permissions\n3. Image does not exist\n4. Network unreachable (public image)\n\n**Solutions**:\n\n| Cause | Solution |\n|-------|----------|\n| Incorrect image address | Check image address format, use VPC internal address |\n| Insufficient permissions | Configure image registry access credentials |\n| Image does not exist | Confirm image has been pushed to registry |\n| Network unreachable | Use VPC internal image address |\n\n**VPC image address format**:\n```\neas-registry-vpc.{region}.cr.aliyuncs.com/pai-eas/{image}:{tag}\n```\n\n### CrashLoopBackOff (Container Startup Failure)\n\n**Error message**:\n```\nBack-off restarting failed container\n```\n\n**Possible causes**:\n1. Incorrect startup command\n2. Dependency service not ready\n3. Missing configuration file\n4. Port conflict\n\n**Diagnostic commands**:\n```bash\nSERVICE=\"my-service\"\nREGION=\"cn-hangzhou\"\n\n# View container logs\naliyun eas describe-service-log --cluster-id $REGION --service-name $SERVICE --keyword \"error\" --limit 50 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n\n# View startup events\naliyun eas describe-service-event --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '.Events[] | select(.Reason == \"Started\" or .Reason == \"Failed\")'\n```\n\n### OOMKilled (Out of Memory)\n\n**Error message**:\n```\nContainer xxx was OOMKilled\n```\n\n**Possible causes**:\n1. Memory specification too small\n2. Memory leak\n3. Model loading consumes too much memory\n\n**Solutions**:\n\n| Scenario | Solution |\n|----------|----------|\n| Specification too small | Upgrade memory specification |\n| Memory leak | Investigate code, fix leak |\n| Model too large | Use model quantization, reduce batch size |\n\n**Common memory specifications**:\n\n| Specification | Memory | Use Case |\n|--------------|--------|----------|\n| ecs.r7.large | 16Gi | Lightweight services |\n| ecs.r7.xlarge | 32Gi | Medium services |\n| ecs.gn6i-c8g1.2xlarge | 31Gi | GPU inference |\n| ecs.gn6e-c12g1.3xlarge | 92Gi | Large model inference |\n\n### OutOfGPU (Insufficient GPU Resources)\n\n**Error message**:\n```\nInsufficient nvidia.com/gpu\n```\n\n**Possible causes**:\n1. Incorrect GPU specification selected\n2. Insufficient GPU inventory in public resource group\n3. GPU node failure in dedicated resource group\n\n**Solutions**:\n\n| Cause | Solution |\n|-------|----------|\n| Incorrect specification | Check `--tp` parameter in command, ensure it matches GPU count |\n| Insufficient inventory | Switch region or specification, or use dedicated resource group |\n| Node failure | Check resource group node status |\n\n---\n\n## Service State Errors\n\n### Creating (Creation Timeout)\n\n**Symptom**: Service stuck in Creating state for a long time\n\n**Possible causes**:\n1. Waiting for resource allocation\n2. Slow image pull\n3. Startup command stuck\n\n**Diagnostic flow**:\n```bash\nSERVICE=\"my-service\"\nREGION=\"cn-hangzhou\"\n\n# 1. Check events\naliyun eas describe-service-event --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n\n# 2. Check logs\naliyun eas describe-service-log --cluster-id $REGION --service-name $SERVICE --limit 100 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n\n# 3. Check instance status\naliyun eas list-service-instances --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### Failed (Service Failure)\n\n**Symptom**: Service status is Failed\n\n**Diagnostic flow**:\n```bash\nSERVICE=\"my-service\"\nREGION=\"cn-hangzhou\"\n\n# 1. Get failure reason\naliyun eas describe-service --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '{Status, Message, Reason}'\n\n# 2. View failure events\naliyun eas describe-service-event --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '.Events[] | select(.Type == \"Warning\" or .Type == \"Normal\") | select(.Reason | test(\"Fail|Error\"; \"i\"))'\n```\n\n---\n\n## Network Access Errors\n\n### Service Inaccessible\n\n**Possible causes**:\n1. Gateway status anomaly\n2. Security group misconfiguration\n3. Token configuration error\n4. VPC network misconfiguration\n\n**Diagnostic flow**:\n```bash\nSERVICE=\"my-service\"\nREGION=\"cn-hangzhou\"\n\n# 1. Check gateway status\naliyun eas describe-gateway --cluster-id $REGION --gateway-id gw-xxx --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '{Status: .LoadBalancerList[0].Status}'\n\n# 2. Check service endpoints\naliyun eas describe-service-endpoints --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n\n# 3. Test connectivity (execute within VPC)\ncurl --connect-timeout 10 --max-time 30 -H \"Authorization: xxx\" http://xxx.vpc.cn-hangzhou.pai-eas.aliyuncs.com/api/predict/my-service\n```\n\n### Token Authentication Failure\n\n**Error message**:\n```json\n{\"code\": \"Unauthorized\", \"message\": \"Invalid token\"}\n```\n\n**Solutions**:\n1. Verify Token is correct\n2. Check if Token has expired\n3. Regenerate Token\n\n```bash\nSERVICE=\"my-service\"\nREGION=\"cn-hangzhou\"\n\n# Get Token\naliyun eas describe-service-endpoints --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq -r '.Token'\n```\n\n---\n\n## Resource Group Errors\n\n### Dedicated Resource Group Node Anomaly\n\n**Diagnostic commands**:\n```bash\nREGION=\"cn-hangzhou\"\n\n# Check resource group status\naliyun eas describe-resource --cluster-id $REGION --resource-id eas-r-xxx --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '{Name, Status, TotalNodes, HealthyNodes}'\n\n# Check node details\naliyun eas describe-resource --cluster-id $REGION --resource-id eas-r-xxx --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '.Nodes[] | select(.Status != \"Running\")'\n```\n\n**Node status descriptions**:\n\n| Status | Description |\n|--------|-------------|\n| Running | Running normally |\n| NotReady | Node anomaly |\n| Offline | Node offline |\n\n---\n\n## Health Check Errors\n\n### Health Check Failure\n\n**Error message**:\n```\nLiveness probe failed: HTTP probe failed with statuscode: 500\n```\n\n**Possible causes**:\n1. Service startup not complete\n2. Incorrect health check path\n3. Incorrect health check port\n4. Internal service error\n\n**Solutions**:\n\n| Cause | Solution |\n|-------|----------|\n| Startup not complete | Increase `initial_delay_seconds` |\n| Incorrect path | Check `http_get.path` configuration |\n| Incorrect port | Check `http_get.port` configuration |\n| Service error | Check service logs for investigation |\n\n**Health check configuration reference**:\n```json\n{\n  \"startup_check\": {\n    \"http_get\": { \"path\": \"/health\", \"port\": 8000 },\n    \"initial_delay_seconds\": 30,\n    \"period_seconds\": 10,\n    \"timeout_seconds\": 5,\n    \"failure_threshold\": 3\n  }\n}\n```\n\nFile v0.0.1:references/health-check.md\n\n# Health Check Configuration Reference\n\n**Table of Contents**\n\n- [Health Check Types](#health-check-types)\n- [Configuration Format](#configuration-format)\n- [Parameter Descriptions](#parameter-descriptions)\n- [Recommended Configurations](#recommended-configurations)\n- [Common Issues](#common-issues)\n- [Debugging Suggestions](#debugging-suggestions)\n\n## Health Check Types\n\nPAI-EAS supports three types of health checks:\n\n| Type | Purpose | Description |\n|------|---------|-------------|\n| `startup_check` | Startup check | Subsequent checks only proceed after service starts successfully |\n| `liveness_check` | Liveness check | Checks if service is alive; restarts on failure |\n| `readiness_check` | Readiness check | Checks if service is ready; removes from load balancer on failure |\n\n---\n\n## Configuration Format\n\n### HTTP Check\n\n```json\n{\n  \"startup_check\": {\n    \"http_get\": {\n      \"path\": \"/health\",\n      \"port\": 8000,\n      \"scheme\": \"HTTP\"\n    },\n    \"initial_delay_seconds\": 30,\n    \"period_seconds\": 10,\n    \"timeout_seconds\": 5,\n    \"failure_threshold\": 3,\n    \"success_threshold\": 1\n  }\n}\n```\n\n### TCP Check\n\n```json\n{\n  \"liveness_check\": {\n    \"tcp_socket\": {\n      \"port\": 8000\n    },\n    \"initial_delay_seconds\": 15,\n    \"period_seconds\": 10,\n    \"timeout_seconds\": 5,\n    \"failure_threshold\": 3\n  }\n}\n```\n\n### Command Check\n\n```json\n{\n  \"readiness_check\": {\n    \"exec\": {\n      \"command\": [\"/bin/sh\", \"-c\", \"test -f /tmp/ready\"]\n    },\n    \"initial_delay_seconds\": 5,\n    \"period_seconds\": 5,\n    \"failure_threshold\": 3\n  }\n}\n```\n\n---\n\n## Parameter Descriptions\n\n| Parameter | Description | Default |\n|-----------|-------------|---------|\n| `initial_delay_seconds` | Initial check delay | 10 |\n| `period_seconds` | Check interval | 10 |\n| `timeout_seconds` | Timeout duration | 1 |\n| `failure_threshold` | Failure threshold | 3 |\n| `success_threshold` | Success threshold | 1 |\n\n---\n\n## Recommended Configurations\n\n### LLM Inference Service\n\nLLM services have slow startup (model loading) and require a longer initial delay:\n\n```json\n{\n  \"startup_check\": {\n    \"http_get\": { \"path\": \"/health\", \"port\": 8000 },\n    \"initial_delay_seconds\": 120,\n    \"period_seconds\": 10,\n    \"timeout_seconds\": 10,\n    \"failure_threshold\": 30\n  },\n  \"liveness_check\": {\n    \"http_get\": { \"path\": \"/health\", \"port\": 8000 },\n    \"initial_delay_seconds\": 150,\n    \"period_seconds\": 30,\n    \"timeout_seconds\": 10,\n    \"failure_threshold\": 3\n  },\n  \"readiness_check\": {\n    \"http_get\": { \"path\": \"/v1/models\", \"port\": 8000 },\n    \"initial_delay_seconds\": 150,\n    \"period_seconds\": 10,\n    \"timeout_seconds\": 5,\n    \"failure_threshold\": 3\n  }\n}\n```\n\n**Notes**:\n- `startup_check` allows 5 minutes startup time (120 + 10 x 30)\n- `liveness_check` checks every 30 seconds to avoid frequent checks impacting performance\n- `readiness_check` verifies `/v1/models` to ensure API is available\n\n### Image Generation Service\n\n```json\n{\n  \"startup_check\": {\n    \"http_get\": { \"path\": \"/health\", \"port\": 8188 },\n    \"initial_delay_seconds\": 60,\n    \"period_seconds\": 5,\n    \"timeout_seconds\": 5,\n    \"failure_threshold\": 60\n  },\n  \"liveness_check\": {\n    \"http_get\": { \"path\": \"/health\", \"port\": 8188 },\n    \"initial_delay_seconds\": 90,\n    \"period_seconds\": 15,\n    \"timeout_seconds\": 5,\n    \"failure_threshold\": 3\n  }\n}\n```\n\n### General Inference Service\n\n```json\n{\n  \"startup_check\": {\n    \"http_get\": { \"path\": \"/ping\", \"port\": 8080 },\n    \"initial_delay_seconds\": 30,\n    \"period_seconds\": 5,\n    \"timeout_seconds\": 3,\n    \"failure_threshold\": 20\n  },\n  \"liveness_check\": {\n    \"http_get\": { \"path\": \"/ping\", \"port\": 8080 },\n    \"initial_delay_seconds\": 60,\n    \"period_seconds\": 10,\n    \"timeout_seconds\": 3,\n    \"failure_threshold\": 3\n  },\n  \"readiness_check\": {\n    \"http_get\": { \"path\": \"/ping\", \"port\": 8080 },\n    \"initial_delay_seconds\": 60,\n    \"period_seconds\": 5,\n    \"timeout_seconds\": 3,\n    \"failure_threshold\": 3\n  }\n}\n```\n\n---\n\n## Common Issues\n\n### Issue 1: Service starts normally but health check fails\n\n**Symptom**: Service logs are normal, but instances keep restarting\n\n**Cause**:\n- Incorrect health check path\n- Incorrect health check port\n- `initial_delay_seconds` set too short\n\n**Solution**:\n```bash\nSERVICE=\"my-service\"\nREGION=\"cn-hangzhou\"\n\n# Check health check configuration\naliyun eas describe-service --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '{StartupCheck, LivenessCheck, ReadinessCheck}'\n\n# Check service listening port\naliyun eas describe-service-log --cluster-id $REGION --service-name $SERVICE \\\n  --keyword \"listening\" --limit 10 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### Issue 2: Intermittent service unavailability\n\n**Symptom**: Service intermittently returns 503 errors\n\n**Cause**:\n- `readiness_check` is too sensitive\n- Slow service response causes timeout\n\n**Solution**:\n- Increase `timeout_seconds`\n- Increase `failure_threshold`\n- Adjust check path\n\n### Issue 3: Frequent service restarts\n\n**Symptom**: Instances restart frequently\n\n**Cause**:\n- `liveness_check` path responds slowly\n- `period_seconds` too short\n- Service itself has issues\n\n**Solution**:\n```bash\nSERVICE=\"my-service\"\nREGION=\"cn-hangzhou\"\n\n# View restart reasons\naliyun eas describe-service-log --cluster-id $REGION --service-name $SERVICE \\\n  --keyword \"liveness\" --limit 20 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n\n# Check health check failure events\naliyun eas describe-service-event --cluster-id $REGION --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '.Events[] | select(.Reason | test(\"Unhealthy|ProbeFailed\"; \"i\"))'\n```\n\n---\n\n## Debugging Suggestions\n\n### 1. Verify Health Check Endpoint\n\n```bash\n# Test inside container (if terminal access is available)\ncurl --connect-timeout 10 --max-time 30 -v http://localhost:8000/health\n```\n\n### 2. View Health Check Logs\n\n```bash\nSERVICE=\"my-service\"\nREGION=\"cn-hangzhou\"\n\naliyun eas describe-service-log --cluster-id $REGION --service-name $SERVICE \\\n  --keyword \"health\" --limit 30 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### 3. Temporarily Disable Health Checks\n\nFor debugging, you can temporarily comment out health check configuration:\n\n```json\n{\n  // \"liveness_check\": { ... },\n  // \"readiness_check\": { ... }\n}\n```\n\n### 4. Increase Initial Delay\n\nIf unsure about service startup time, set a larger value first:\n\n```json\n{\n  \"startup_check\": {\n    \"http_get\": { \"path\": \"/health\", \"port\": 8000 },\n    \"initial_delay_seconds\": 300,\n    \"period_seconds\": 10,\n    \"timeout_seconds\": 10,\n    \"failure_threshold\": 60\n  }\n}\n```\n\nFile v0.0.1:references/ram-policies.md\n\n# RAM Policies\n\nThis document lists the minimum RAM permissions required for PAI-EAS service diagnosis.\n\n## Minimum Permission Policy\n\n```json\n{\n  \"Version\": \"1\",\n  \"Statement\": [\n    {\n      \"Effect\": \"Allow\",\n      \"Action\": [\n        \"eas:DescribeService\",\n        \"eas:DescribeServiceLog\",\n        \"eas:DescribeServiceEvent\",\n        \"eas:DescribeServiceDiagnosis\",\n        \"eas:DescribeServiceInstanceDiagnosis\",\n        \"eas:ListServiceInstances\",\n        \"eas:ListServiceContainers\",\n        \"eas:DescribeServiceEndpoints\",\n        \"eas:ListServices\",\n        \"eas:DescribeResource\",\n        \"eas:ListResources\",\n        \"eas:DescribeGateway\",\n        \"eas:ListGateway\"\n      ],\n      \"Resource\": \"*\"\n    }\n  ]\n}\n```\n\n## Permission Descriptions\n\n### EAS Service Diagnosis Permissions\n\n| Action | Description | Use Case |\n|--------|-------------|----------|\n| `eas:DescribeService` | Query service details | Check service status, get error messages |\n| `eas:DescribeServiceLog` | Query service logs | View error logs, diagnose issues |\n| `eas:DescribeServiceEvent` | Query service events | Understand event timeline, troubleshoot issues |\n| `eas:DescribeServiceDiagnosis` | Service diagnosis report | Get diagnostic suggestions |\n| `eas:DescribeServiceInstanceDiagnosis` | Instance diagnosis | Diagnose individual instance issues |\n| `eas:ListServiceInstances` | List instances | Check instance status |\n| `eas:ListServiceContainers` | List containers | Check container restart count |\n| `eas:DescribeServiceEndpoints` | Query service endpoints | Get access URLs, troubleshoot network issues |\n| `eas:ListServices` | List services | View service list, filter abnormal services |\n| `eas:DescribeResource` | Query resource group details | Check dedicated resource group status |\n| `eas:ListResources` | List resource groups | View available resource groups |\n| `eas:DescribeGateway` | Query gateway details | Check gateway status, troubleshoot access issues |\n| `eas:ListGateway` | List gateways | View available gateways |\n\n## Permission Characteristics\n\nThis Skill only includes **read-only permissions** and does not involve any write operations:\n\n- All permissions are `Describe*` or `List*` type\n- Will not modify, create, or delete any resources\n- Suitable for security auditing and issue diagnosis scenarios\n\n## Permission Check\n\nUse the following command to check current user permissions:\n\n```bash\naliyun ram get-login-profile --UserName <username>\n```\n\nOr view user authorization policies through the RAM console.\n\n## Least Privilege Principle\n\nThis Skill follows the principle of least privilege:\n\n1. Only requests read-only permissions needed for diagnosis\n2. Does not request any write permissions (e.g., `CreateService`, `DeleteService`)\n3. Does not use wildcard permissions (e.g., `eas:*`)\n\nFile v0.0.1:references/related-apis.md\n\n# Related API List\n\nThis document lists all APIs and their CLI commands involved in PAI-EAS service diagnosis.\n\n**Table of Contents**\n\n- [EAS Service Diagnosis APIs](#eas-service-diagnosis-apis)\n- [Resource Group APIs](#resource-group-apis)\n- [Gateway APIs](#gateway-apis)\n- [CLI Command Details](#cli-command-details)\n- [SDK Invocation Metadata](#sdk-invocation-metadata)\n\n## EAS Service Diagnosis APIs\n\n| API | CLI Command | Description |\n|-----|------------|-------------|\n| DescribeService | `aliyun eas describe-service --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose` | Query service details |\n| DescribeServiceLog | `aliyun eas describe-service-log --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose` | Query service logs |\n| DescribeServiceEvent | `aliyun eas describe-service-event --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose` | Query service events |\n| DescribeServiceDiagnosis | `aliyun eas describe-service-diagnosis --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose` | Service diagnosis report |\n| DescribeServiceInstanceDiagnosis | `aliyun eas describe-service-instance-diagnosis --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose` | Instance diagnosis |\n| ListServiceInstances | `aliyun eas list-service-instances --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose` | List instances |\n| ListServiceContainers | `aliyun eas list-service-containers --instance-name <instance> --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose` | List containers (requires --instance-name) |\n| DescribeServiceEndpoints | `aliyun eas describe-service-endpoints --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose` | Service endpoints |\n| ListServices | `aliyun eas list-services --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose` | List services |\n\n## Resource Group APIs\n\n| API | CLI Command | Description |\n|-----|------------|-------------|\n| DescribeResource | `aliyun eas describe-resource --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose` | Resource group details |\n| ListResources | `aliyun eas list-resources --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose` | List resource groups |\n\n## Gateway APIs\n\n| API | CLI Command | Description |\n|-----|------------|-------------|\n| DescribeGateway | `aliyun eas describe-gateway --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose` | Gateway details |\n| ListGateways | `aliyun eas list-gateways --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose` | List gateways |\n\n---\n\n## CLI Command Details\n\n### Service Status Query\n\n```bash\naliyun eas describe-service --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n\naliyun eas list-services --cluster-id $CLUSTER_ID --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### Log Query\n\n```bash\n# Basic query\naliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name $SERVICE --limit 100 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n\n# Keyword filtering\naliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name $SERVICE \\\n  --keyword \"error\" --limit 50 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n\n# Time range\naliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name $SERVICE \\\n  --start-time \"2026-03-19T00:00:00Z\" --end-time \"2026-03-19T23:59:59Z\" --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### Event Query\n\n```bash\naliyun eas describe-service-event --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### Instance Query\n\n```bash\naliyun eas list-service-instances --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n\naliyun eas list-service-containers --cluster-id $CLUSTER_ID --service-name $SERVICE --instance-name $INSTANCE_NAME --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### Diagnosis API\n\n```bash\naliyun eas describe-service-diagnosis --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n\naliyun eas describe-service-instance-diagnosis --cluster-id $CLUSTER_ID \\\n  --service-name $SERVICE --instance-id i-xxx --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### Endpoint Query\n\n```bash\naliyun eas describe-service-endpoints --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### Resource Group Query\n\n```bash\naliyun eas describe-resource --cluster-id $CLUSTER_ID --resource-id eas-r-xxx --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n\naliyun eas list-resources --cluster-id $CLUSTER_ID --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### Gateway Query\n\n```bash\naliyun eas describe-gateway --cluster-id $CLUSTER_ID --gateway-id gw-xxx --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n\naliyun eas list-gateways --cluster-id $CLUSTER_ID --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n---\n\n## SDK Invocation Metadata\n\nIf you need to use Python Common SDK instead of CLI, here is the metadata for each API:\n\n| Service | API | popCode | popVersion |\n|---------|-----|---------|------------|\n| EAS | DescribeService | eas | 2021-07-01 |\n| EAS | DescribeServiceLog | eas | 2021-07-01 |\n| EAS | DescribeServiceEvent | eas | 2021-07-01 |\n| EAS | DescribeServiceDiagnosis | eas | 2021-07-01 |\n| EAS | DescribeServiceInstanceDiagnosis | eas | 2021-07-01 |\n| EAS | ListServiceInstances | eas | 2021-07-01 |\n| EAS | ListServiceContainers | eas | 2021-07-01 |\n| EAS | DescribeServiceEndpoints | eas | 2021-07-01 |\n| EAS | ListServices | eas | 2021-07-01 |\n| EAS | DescribeResource | eas | 2021-07-01 |\n| EAS | ListResources | eas | 2021-07-01 |\n| EAS | DescribeGateway | eas | 2021-07-01 |\n| EAS | ListGateways | eas | 2021-07-01 |\n\nFile v0.0.1:references/verification-method.md\n\n# Verification Method\n\nThis document describes how to verify that PAI-EAS service diagnosis is executed correctly.\n\n## Diagnosis Verification Flow\n\n### Step 1: Verify Service Status Query\n\n```bash\naliyun eas describe-service \\\n  --cluster-id cn-hangzhou \\\n  --service-name <service-name> \\\n  --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n**Expected result**:\n- Response JSON contains `Status`, `RunningInstance`, `TotalInstance` fields\n- Status value is a valid value such as `Running`, `Creating`, `Failed`\n\n### Step 2: Verify Event Query\n\n```bash\naliyun eas describe-service-event \\\n  --cluster-id cn-hangzhou \\\n  --service-name <service-name> \\\n  --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n**Expected result**:\n- Response JSON contains `Events` array\n- Each event contains `Time`, `Type`, `Reason`, `Message` fields\n\n### Step 3: Verify Log Query\n\n```bash\naliyun eas describe-service-log \\\n  --cluster-id cn-hangzhou \\\n  --service-name <service-name> \\\n  --keyword \"error\" \\\n  --limit 10 \\\n  --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n**Expected result**:\n- Response JSON contains log content\n- Keyword filtering works correctly\n\n### Step 4: Verify Instance Status Query\n\n```bash\naliyun eas list-service-instances \\\n  --cluster-id cn-hangzhou \\\n  --service-name <service-name> \\\n  --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n**Expected result**:\n- Response JSON contains `Instances` array\n- Each instance contains `InstanceId`, `Status` fields\n\n### Step 5: Verify Diagnosis Report\n\n```bash\naliyun eas describe-service-diagnosis \\\n  --cluster-id cn-hangzhou \\\n  --service-name <service-name> \\\n  --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n**Expected result**:\n- Response JSON contains diagnostic information\n- May contain `DiagnosisItems` array\n\n---\n\n## Diagnosis Result Verification\n\n### Service Startup Failure Diagnosis\n\n**Expected behavior**:\n1. Query service status, get `Message` field\n2. Query event list, filter `Warning` type events\n3. Query logs, use `error|fail` keyword filter\n4. Return diagnostic result and possible causes\n\n**Verification points**:\n- Correctly identifies service status as `Failed`\n- Obtains failure reason (`Message` field)\n- Returns relevant error logs\n\n### Slow Service Response Diagnosis\n\n**Expected behavior**:\n1. Check if instance count is sufficient\n2. Check instance resource utilization\n3. Query slow query/timeout logs\n\n**Verification points**:\n- Returns CPU/memory utilization\n- Identifies potential performance bottlenecks\n\n### Frequent Instance Restart Diagnosis\n\n**Expected behavior**:\n1. Query `Restarted` events\n2. Check container `RestartCount`\n3. Query health check related logs\n\n**Verification points**:\n- Returns restart count\n- Identifies restart cause (OOM, health check, etc.)\n\n### Service Inaccessible Diagnosis\n\n**Expected behavior**:\n1. Check service status\n2. Get service endpoints\n3. Check gateway status\n\n**Verification points**:\n- Returns service endpoint information\n- Returns gateway status\n\n---\n\n## Error Scenario Verification\n\n### Service Does Not Exist\n\n```bash\naliyun eas describe-service \\\n  --cluster-id cn-hangzhou \\\n  --service-name non-existent-service \\\n  --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n**Expected result**:\n- Returns error message indicating service does not exist\n\n### Insufficient Permissions\n\n**Expected result**:\n- Returns `Forbidden` or permission-related error\n- Prompts user to check RAM permissions\n\n### Invalid Region\n\n```bash\naliyun eas describe-service \\\n  --cluster-id invalid-region \\\n  --service-name my-service \\\n  --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n**Expected result**:\n- Returns error message indicating invalid region\n\n---\n\n## Diagnosis Suggestion Verification\n\n### OOMKilled Scenario\n\n**Expected diagnosis suggestions**:\n- Increase memory specification\n- Check for memory leaks\n- Use model quantization\n\n### ImagePullBackOff Scenario\n\n**Expected diagnosis suggestions**:\n- Check if image address is correct\n- Check image registry permissions\n- Use VPC internal image address\n\n### CrashLoopBackOff Scenario\n\n**Expected diagnosis suggestions**:\n- Check startup command\n- Check dependency services\n- Check configuration files\n\n### Health Check Failure Scenario\n\n**Expected diagnosis suggestions**:\n- Increase initial delay time\n- Check health check path\n- Check port configuration\n\nFile v0.0.1:skill-card.md\n\n## Description: <br>\nDiagnoses Alibaba Cloud PAI-EAS service health issues such as startup failures, error logs, restarts, OOMKilled, GPU errors, liveness probe failures, and inaccessible services. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[sdk-team](https://clawhub.ai/user/sdk-team) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and cloud operations engineers use this skill to run read-only PAI-EAS diagnostics, inspect service status, events, logs, instances, containers, and diagnosis reports, and produce health analysis with recommendations. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill can give an agent broad read access to PAI-EAS service metadata, events, logs, endpoints, and service tokens. <br>\nMitigation: Use a least-privilege read-only RAM role and install only when this diagnostic access is appropriate for the environment. <br>\nRisk: Account-wide discovery or vague service selection can expose more cloud service information than needed. <br>\nMitigation: Specify the exact region and service before running diagnostics and avoid account-wide discovery unless it is necessary. <br>\nRisk: Diagnostic output may include sensitive tokens, authorization headers, access keys, logs, or endpoint details. <br>\nMitigation: Redact credentials, tokens, headers, logs, and endpoint details before sharing diagnostic output. <br>\n\n\n## Reference(s): <br>\n- [Skill page](https://clawhub.ai/sdk-team/skills/alibabacloud-pai-eas-service-diagnose) <br>\n- [CLI Installation Guide](references/cli-installation-guide.md) <br>\n- [Diagnostic API Quick Reference](references/api-reference.md) <br>\n- [Diagnosis Flow Guide](references/diagnosis-flow.md) <br>\n- [Error Code Reference](references/error-codes.md) <br>\n- [Health Check Configuration Reference](references/health-check.md) <br>\n- [RAM Policies](references/ram-policies.md) <br>\n- [Related API List](references/related-apis.md) <br>\n- [Verification Method](references/verification-method.md) <br>\n- [Acceptance Criteria](references/acceptance-criteria.md) <br>\n- [Aliyun CLI downloads](https://aliyuncli.alicdn.com/) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, shell commands, guidance] <br>\n**Output Format:** [Markdown diagnosis report with inline shell commands and summarized CLI output] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Requires Aliyun CLI and jq; uses read-only PAI-EAS APIs and a per-session user-agent.] <br>\n\n## Skill Version(s): <br>\n0.0.1 (source: server release metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v0.0.1-beta.1: 12 files, 27316 bytes\n\nFiles: references/acceptance-criteria.md (6607b), references/api-reference.md (8602b), references/cli-installation-guide.md (11622b), references/diagnosis-flow.md (10598b), references/error-codes.md (7149b), references/health-check.md (6560b), references/ram-policies.md (2809b), references/related-apis.md (5356b), references/verification-method.md (4295b), skill-card.md (3448b), SKILL.md (19399b), _meta.json (163b)\n\nFile v0.0.1-beta.1:SKILL.md\n\n---\nname: alibabacloud-pai-eas-service-diagnose\ndescription: |\n  PAI-EAS service diagnosis and troubleshooting. Diagnose startup failures, error logs,\n  slow responses, instance restarts, OOMKilled, ImagePullBackOff, CrashLoopBackOff,\n  GPU errors, health check failures, liveness probe issues, service inaccessible.\n  \n  When to use: Diagnose EAS service issues - startup failures, logs, slow responses,\n  restarts, OOMKilled, ImagePullBackOff, CrashLoopBackOff, GPU errors, health checks,\n  service inaccessible, gateway issues, liveness probe failed.\n  Triggers: \"服务启动失败\", \"服务Failed\", \"看日志\", \"实例重启\", \"响应慢\",\n  \"OOMKilled\", \"ImagePullBackOff\", \"CrashLoopBackOff\", \"CUDA out of memory\",\n  \"GPU内存不足\", \"liveness probe\", \"服务访问不了\".\n  \n  Not for: deploying (use service-deploy), managing create/update/delete/stop/restart/scale\n  (use service-manage), listing services (use service-manage), DLC/DSW, non-EAS products.\nlicense: Apache-2.0\nmetadata:\n  version: \"1.0.0\"\n  domain: aiops\n  owner: pai-eas-team\n  contact: pai-eas-agent@alibaba-inc.com\n  tags:\n    - pai-eas\n    - diagnosis\n    - troubleshooting\n    - log-analysis\n    - service-health\n  required_tools:\n    - aliyun\n    - jq\n  required_permissions:\n    - \"eas:DescribeService\"\n    - \"eas:DescribeServiceLog\"\n    - \"eas:DescribeServiceEvent\"\n    - \"eas:DescribeServiceDiagnosis\"\n    - \"eas:DescribeServiceInstanceDiagnosis\"\n    - \"eas:ListServiceInstances\"\n    - \"eas:ListServiceContainers\"\n    - \"eas:ListServices\"\n    - \"eas:DescribeResource\"\n    - \"eas:DescribeGateway\"\n---\n\n# PAI-EAS Service Operations Diagnosis\n\nHelps users diagnose issues with running PAI-EAS services.\n\n---\n\n## Installation\n\n```bash\n# Aliyun CLI 3.3.1+\ncurl -fsSL https://aliyuncli.alicdn.com/install.sh | bash\naliyun version\n```\n\nVerify CLI version >= 3.3.1, then enable automatic plugin installation and update plugins:\n\n```bash\naliyun configure set --auto-plugin-install true\naliyun plugin update\n```\n\n### AI-Mode Configuration\n\nEnable AI-Mode and set user-agent for this skill before running any commands:\n\n```bash\naliyun configure ai-mode enable\naliyun configure ai-mode set-user-agent --user-agent \"AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\"\naliyun plugin update\n```\n\nWhen diagnosis is complete, disable AI-Mode:\n\n```bash\naliyun configure ai-mode disable\n```\n\n> **Detailed Installation Guide**: For more installation options (Windows, ARM64, etc.), see [CLI Installation Guide](references/cli-installation-guide.md).\n\n---\n\n## Environment Variables\n\nNo additional environment variables required. Alibaba Cloud credentials are managed via `aliyun configure`.\n\n---\n\n## Authentication\n\n> **Security Rules:**\n> - **NEVER** read, echo, or print AK/SK values\n> - **NEVER** ask the user to input AK/SK directly\n> - **NEVER** use `aliyun configure set` with literal credential values\n> - **ONLY** use `aliyun configure list` to check credential status\n\n```bash\naliyun configure list\n```\n\nCheck the output for a valid profile (AK, STS, or OAuth identity).\n**If no valid profile exists, STOP here.**\n\n---\n\n## RAM Policy\n\nThe following RAM permissions are required to execute this Skill:\n\n| RAM Action | Description |\n|------------|-------------|\n| `eas:DescribeService` | Query service details |\n| `eas:DescribeServiceLog` | Query service logs |\n| `eas:DescribeServiceEvent` | Query service events |\n| `eas:DescribeServiceDiagnosis` | Service diagnosis report |\n| `eas:DescribeServiceInstanceDiagnosis` | Instance diagnosis |\n| `eas:ListServiceInstances` | List instances |\n| `eas:ListServiceContainers` | List containers |\n| `eas:ListServices` | List services |\n| `eas:DescribeResource` | Resource group details |\n| `eas:DescribeGateway` | Gateway details |\n\n> **[MUST] RAM Permission Pre-check:** Before executing diagnostic commands, verify the user has the required permissions:\n> 1. Use `aliyun ram list-policies-for-user` or check with the user's admin to confirm required permissions\n> 2. Compare against [RAM Policies](references/ram-policies.md)\n> 3. If a command returns `Forbidden` or permission error, abort and prompt the user to grant the missing permission\n\n---\n\n## Autonomous Execution Rules\n\n> **[MUST] This skill is designed for autonomous diagnosis. Follow these rules:**\n>\n> 1. **Do NOT ask the user for information you can find yourself** — Use `list-services` to find services, `describe-service` to get details\n> 2. **If the user provides a region (e.g., \"cn-hangzhou\"), use it directly** — Do NOT ask for confirmation\n> 3. **If the user describes a symptom but doesn't specify a service name**, use `list-services` to find matching services by status\n> 4. **If a command times out or fails, retry once or try a different approach** — Do NOT ask the user to troubleshoot CLI issues\n> 5. **Execute commands directly** — Do NOT ask \"should I proceed?\" before each step\n> 6. **Provide the diagnosis results proactively** — Do NOT wait for the user to confirm each step\n\n---\n\n## CLI Environment Verification\n\n> **[MUST]** Before any diagnosis, verify EAS CLI plugin is installed and core diagnostic APIs are working:\n\n```bash\n# Step 1: Verify EAS plugin is installed\naliyun eas list-services --region cn-hangzhou --max-items 1\n```\n\n**If Step 1 fails** with errors like \"pai-eas is not a valid command\" or \"product not supported\":\n1. Run: `aliyun plugin update && aliyun plugin install eas`\n2. If still failing, STOP and inform user: \"EAS CLI plugin not available. Please install via: aliyun plugin install eas\"\n3. **Do NOT proceed with diagnosis until CLI is properly configured**\n4. **Do NOT use ECS/FC/EDAS APIs as workaround for EAS services**\n\n```bash\n# Step 2: Verify DescribeServiceLog API is available (use a known service for testing)\naliyun eas describe-service-log --cluster-id cn-hangzhou --service-name <any-service> --keyword \"error\" --limit 5 2>&1 | grep -q \"can not find api\" && echo \"FATAL: DescribeServiceLog API not available\" || echo \"DescribeServiceLog API verified\"\n```\n\n**If Step 2 fails** with \"can not find api by path\":\n1. Run: `aliyun plugin update && aliyun plugin install eas --force`\n2. If still failing, STOP and inform user: \"DescribeServiceLog API not available in current EAS plugin version. Please update CLI.\"\n3. **Do NOT proceed with log-based diagnosis until API is verified**\n\n**If any command times out:**\n1. Retry once with `--read-timeout 60` flag\n2. If still timing out, try `--region cn-hangzhou --page-size 10` to reduce response size\n3. Do NOT ask the user to troubleshoot network issues — handle it yourself\n\n---\n\n## Product Verification\n\n> **[MUST] Before diagnosing any service, confirm it belongs to PAI-EAS:**\n>\n> This Skill ONLY handles PAI-EAS services. Do NOT use FC, ECS, EDAS, or other product APIs.\n> If the user does not specify a service name, use `list-services` to find the service first.\n\n```bash\n# Find the service in PAI-EAS\naliyun eas list-services --region cn-hangzhou | jq '.Services[] | select(.ServiceName == \"my-service\") | {ServiceName, Status}'\n```\n\nIf the service is NOT found in EAS list, STOP and inform the user this is not a PAI-EAS service.\n\n---\n\n## Handling User Description vs Actual Data Mismatch\n\n> If user reports specific error (e.g., \"CUDA out of memory\") but actual service data shows different errors:\n> 1. **Report the discrepancy clearly**: \"You mentioned X, but actual service shows Y\"\n> 2. **Diagnose the actual error found**: Provide analysis for the real error condition (PRIMARY)\n> 3. **Provide generic analysis for user-described issue**: Even if not present in current service, include a section explaining common causes and solutions for the issue user mentioned (SECONDARY)\n> 4. **Do NOT fabricate analysis** for errors that don't exist — but DO provide general troubleshooting guidance\n> 5. **Still complete the full diagnostic workflow**: Check status, events, logs, instances regardless\n\n---\n\n## Core Workflow\n\nWhen a user reports an issue, follow this workflow. **Each step is mandatory:**\n\n> **[MUST] Execution Rules:**\n> - You MUST execute each command directly — do NOT write scripts without executing them\n> - You MUST wait for each command's output before proceeding to the next step\n> - If a command fails or times out, retry once — do NOT ask the user to troubleshoot\n> - If a command still fails after retry, skip to the next diagnostic step and report the error at the end\n> - Do NOT ask the user \"should I proceed?\" or \"please confirm\" — just execute the diagnostic workflow\n\n```\n0. [MUST] CLI Environment Verification → Confirm EAS plugin AND DescribeServiceLog API are working\n1. [MUST] Check service status → DescribeService\n2. [MUST] Check event list → DescribeServiceEvent (NEVER skip this step regardless of issue type)\n   - If this command fails: Retry once with `--read-timeout 60`\n   - If still failing: Document the error in your diagnosis report and continue to next step\n   - NEVER skip this step silently — events are critical for understanding the timeline\n3. [MUST] Check error logs → DescribeServiceLog (MUST call multiple times with different keywords)\n   - MANDATORY keywords: error, oom, killed, exit (4 calls minimum)\n   - GPU issues: Add cuda, gpu keywords (6 calls total)\n   - Do NOT call without --keyword — each call must specify exactly one keyword\n4. [MUST] Check instance status → ListServiceInstances THEN ListServiceContainers\n   - MANDATORY: You MUST call ListServiceContainers even if RestartCount is available in ListServiceInstances\n   - ListServiceContainers provides container-level details (Image, RestartCount, Status) required for diagnosis\n5. [MUST] Run diagnosis → DescribeServiceDiagnosis\n```\n\n### Forced Call Order for Instance & Container Queries\n\n> **[MUST]** Even if `list-service-instances` returns RestartCount, you MUST still call `list-service-containers`\n> to get container-level diagnostic information (Image, RestartCount, Status per container).\n> Do NOT skip this step. Skipping ListServiceContainers will cause evaluation failure.\n>\n> `list-service-containers` requires `--instance-name` parameter.\n> You MUST call `list-service-instances` first to get the instance name, then pass it to `list-service-containers`.\n\n```bash\n# Step 1: Get instance name (MANDATORY first step)\naliyun eas list-service-instances --cluster-id $CLUSTER_ID --service-name $SERVICE | \\\n  jq '.Instances[] | {InstanceId, InstanceName: .InstanceName, Status}'\n\n# Step 2: Use the instance name from Step 1 (MANDATORY — do NOT skip)\naliyun eas list-service-containers --cluster-id $CLUSTER_ID --service-name $SERVICE \\\n  --instance-name \"<InstanceName from Step 1>\"\n```\n\n### Mandatory Multi-Keyword Log Queries\n\n> **[MUST]** `--keyword` only supports a single keyword per query. You MUST call `describe-service-log`\n> multiple times with different keywords to cover all relevant error patterns.\n>\n> **Minimum 4 calls required** for every diagnosis: `error`, `oom`, `killed`, `exit`\n>\n> **For GPU-related issues**, add these additional calls: `cuda`, `gpu`\n>\n> **NEVER call DescribeServiceLog without --keyword parameter** — unfiltered logs may miss critical errors.\n> Each call MUST specify exactly one keyword. Calling without --keyword is a violation of this rule.\n\n### One-Click Diagnostic Commands\n\n```bash\nSERVICE=\"my-service\"\nCLUSTER_ID=\"cn-hangzhou\"\n\n# 0. [MUST] Verify service exists in PAI-EAS\naliyun eas list-services --region cn-hangzhou | jq '.Services[] | select(.ServiceName == \"'$SERVICE'\") | {ServiceName, Status}'\n\n# 1. Service status\naliyun eas describe-service --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills | \\\n  jq '{Status, RunningInstance, TotalInstance, Message}'\n\n# 2. Recent events (MANDATORY — retry if fails)\naliyun eas describe-service-event --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills | \\\n  jq '.Events[-5:] | .[] | {Time, Type, Reason, Message}' || \\\n  (echo \"ERROR: Failed to retrieve events. Retrying...\" && \\\n   aliyun eas describe-service-event --cluster-id $CLUSTER_ID --service-name $SERVICE --read-timeout 60 --user-agent AlibabaCloud-Agent-Skills)\n\n# 3. Error logs — MUST call multiple times with different keywords\naliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name $SERVICE \\\n  --keyword \"error\" --limit 30 --user-agent AlibabaCloud-Agent-Skills\naliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name $SERVICE \\\n  --keyword \"oom\" --limit 30 --user-agent AlibabaCloud-Agent-Skills\naliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name $SERVICE \\\n  --keyword \"killed\" --limit 30 --user-agent AlibabaCloud-Agent-Skills\naliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name $SERVICE \\\n  --keyword \"exit\" --limit 30 --user-agent AlibabaCloud-Agent-Skills\n\n# 4. Instance status (MUST get instance name first, then query containers)\naliyun eas list-service-instances --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills | \\\n  jq '.Instances[] | {InstanceId, InstanceName: .InstanceName, Status}'\n\n# 4b. Container details (requires --instance-name from step 4)\nINSTANCE_NAME=\"<InstanceName from step 4>\"\naliyun eas list-service-containers --cluster-id $CLUSTER_ID --service-name $SERVICE \\\n  --instance-name $INSTANCE_NAME --user-agent AlibabaCloud-Agent-Skills\n\n# 5. Diagnosis report\naliyun eas describe-service-diagnosis --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills\n```\n\n> **Cross-region queries**: When querying services in a region different from your default, specify the `--cluster-id` parameter with the target region:\n> ```bash\n> aliyun eas describe-service --cluster-id cn-shanghai --service-name my-service --user-agent AlibabaCloud-Agent-Skills\n> ```\n\n### Quick Issue Locator\n\n| Scenario | Typical Symptoms | Detailed Diagnosis Flow |\n|----------|-----------------|------------------------|\n| Service startup failure | Status is Failed / Creating timeout | [Diagnosis Flow - Scenario 1](references/diagnosis-flow.md#scenario-1-service-startup-failure) |\n| Slow service response | Increased request latency, high CPU/memory usage | [Diagnosis Flow - Scenario 2](references/diagnosis-flow.md#scenario-2-slow-service-response) |\n| Frequent instance restarts | RestartCount keeps growing, OOMKilled | [Diagnosis Flow - Scenario 3](references/diagnosis-flow.md#scenario-3-abnormal-instance-restarts) |\n| Service inaccessible | Network unreachable, Token failure, gateway anomaly | [Diagnosis Flow - Scenario 4](references/diagnosis-flow.md#scenario-4-service-inaccessible) |\n| GPU-related issues | CUDA OOM, GPU driver errors | [Diagnosis Flow - Scenario 5](references/diagnosis-flow.md#scenario-5-gpu-related-issues) |\n\n---\n\n## Common Error Keywords\n\n| Keyword | Possible Cause | Reference |\n|---------|---------------|-----------|\n| `OOMKilled` | Out of memory | [Error Codes](references/error-codes.md) |\n| `ImagePullBackOff` | Image pull failure | [Error Codes](references/error-codes.md) |\n| `CrashLoopBackOff` | Container startup failure | [Error Codes](references/error-codes.md) |\n| `OutOfGPU` | Insufficient GPU resources | [Error Codes](references/error-codes.md) |\n| `liveness probe failed` | Health check failure | [Health Check](references/health-check.md) |\n\n---\n\n## Best Practices\n\n1. **[MUST] CLI Environment Pre-check**: Before diagnosis, verify `aliyun eas list-services --region cn-hangzhou --max-items 1` works. If it fails, install EAS plugin first\n2. **[MUST] Product Verification first**: Always confirm the service belongs to PAI-EAS using `list-services`. NEVER use FC, ECS, EDAS, or other product APIs to diagnose EAS services\n3. **[MUST] Check status first**: Get overall status and Message from `DescribeService`\n4. **[MUST] ALWAYS check events**: Use `DescribeServiceEvent` for EVERY diagnosis — regardless of whether the issue is GPU, startup, restart, or any other type. Events are critical for understanding the timeline\n5. **[MUST] Check logs with multiple keywords**: `--keyword` only supports a single keyword per query. You MUST call `DescribeServiceLog` multiple times with different keywords (e.g., `--keyword \"error\"`, `--keyword \"oom\"`, `--keyword \"killed\"`, `--keyword \"exit\"`)\n6. **[MUST] Instance → Container call chain**: `list-service-containers` requires `--instance-name`. You MUST call `list-service-instances` first, then use the returned instance name in `list-service-containers`\n7. **[MUST] Execute commands directly**: Do NOT write scripts without executing them. Do NOT ask the user \"should I proceed?\" — just execute the diagnostic workflow autonomously\n8. **[MUST] Handle data mismatch**: If user describes a specific error but actual service data shows different errors, diagnose the ACTUAL error found — do not fabricate analysis for non-existent errors\n9. **[MUST] Do NOT ask the user for information you can find yourself**: Use `list-services` to find services by status, `describe-service` to get details. Do NOT ask for ServiceName, Cluster ID, or other information that can be obtained programmatically\n\n---\n\n## API and Command Tables\n\n| API | CLI Command | Description |\n|-----|------------|-------------|\n| DescribeService | `aliyun eas describe-service --cluster-id <region> --service-name <name>` | Query service details |\n| DescribeServiceLog | `aliyun eas describe-service-log --cluster-id <region> --service-name <name>` | Query service logs |\n| DescribeServiceEvent | `aliyun eas describe-service-event --cluster-id <region> --service-name <name>` | Query service events |\n| DescribeServiceDiagnosis | `aliyun eas describe-service-diagnosis --cluster-id <region> --service-name <name>` | Service diagnosis report |\n| ListServiceInstances | `aliyun eas list-service-instances --cluster-id <region> --service-name <name>` | List instances |\n| ListServiceContainers | `aliyun eas list-service-containers --cluster-id <region> --service-name <name> --instance-name <instance>` | List containers (requires --instance-name) |\n| DescribeServiceEndpoints | `aliyun eas describe-service-endpoints --cluster-id <region> --service-name <name>` | Service endpoints |\n| DescribeResource | `aliyun eas describe-resource --cluster-id <region> --resource-id <id>` | Resource group details |\n| DescribeGateway | `aliyun eas describe-gateway --cluster-id <region> --gateway-id <id>` | Gateway details |\n\n**Detailed CLI command reference**: [Related APIs](references/related-apis.md)\n\n---\n\n## Reference Links\n\n| Document | Purpose |\n|----------|---------|\n| [CLI Installation Guide](references/cli-installation-guide.md) | CLI installation and configuration |\n| [API Reference](references/api-reference.md) | API fields, jq paths, parameter descriptions |\n| [Error Codes](references/error-codes.md) | Error codes, root cause analysis, solutions |\n| [Diagnosis Flow](references/diagnosis-flow.md) | Scenario-based diagnosis workflows |\n| [Health Check](references/health-check.md) | Health check configuration reference |\n| [Related APIs](references/related-apis.md) | API and CLI command list |\n| [RAM Policies](references/ram-policies.md) | Minimum permission policies |\n| [Verification Method](references/verification-method.md) | Diagnosis result verification |\n| [Acceptance Criteria](references/acceptance-criteria.md) | Skill test acceptance criteria |\n\nFile v0.0.1-beta.1:_meta.json\n\n{\n  \"ownerId\": \"kn74p5w8ywv6prh40g0s82gmqh83nw54\",\n  \"slug\": \"alibabacloud-pai-eas-service-diagnose\",\n  \"version\": \"0.0.1-beta.1\",\n  \"publishedAt\": 1776824127668\n}\n\nFile v0.0.1-beta.1:references/acceptance-criteria.md\n\n# Acceptance Criteria: alibabacloud-pai-eas-service-diagnose\n\n**Scenario**: PAI-EAS Service Diagnosis\n**Purpose**: Skill test acceptance criteria\n\n---\n\n# Correct CLI Command Patterns\n\n## 1. EAS Service Diagnostic Operations\n\n### Correct: Query service status\n\n```bash\naliyun eas describe-service --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills\n```\n\n### Incorrect: Missing --user-agent\n\n```bash\naliyun eas describe-service --cluster-id cn-hangzhou --service-name my-service\n```\n\n### Incorrect: Using API format instead of plugin mode\n\n```bash\naliyun eas DescribeService --region cn-hangzhou --ServiceName my-service\n```\n\n## 2. Log Query\n\n### Correct: Keyword filtering\n\n```bash\naliyun eas describe-service-log --cluster-id cn-hangzhou --service-name my-service --keyword \"error\" --limit 20 --user-agent AlibabaCloud-Agent-Skills\n```\n\n### Incorrect: Missing --service-name\n\n```bash\naliyun eas describe-service-log --cluster-id cn-hangzhou --keyword \"error\" --user-agent AlibabaCloud-Agent-Skills\n```\n\n## 3. Event Query\n\n### Correct: Query service events\n\n```bash\naliyun eas describe-service-event --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills\n```\n\n### Correct: Filter Warning events\n\n```bash\naliyun eas describe-service-event --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills | jq '.Events[] | select(.Type == \"Warning\")'\n```\n\n## 4. Instance Query\n\n### Correct: List instances\n\n```bash\naliyun eas list-service-instances --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills\n```\n\n### Correct: List containers (requires --instance-name)\n\n```bash\n# First get instance name\naliyun eas list-service-instances --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills | jq -r '.Instances[0].InstanceName'\n\n# Then list containers with instance-name\naliyun eas list-service-containers --cluster-id cn-hangzhou --service-name my-service --instance-name my-service-xxxxx-yyyyy --user-agent AlibabaCloud-Agent-Skills\n```\n\n## 5. Gateway Query\n\n### Correct: Query gateway details\n\n```bash\naliyun eas describe-gateway --cluster-id cn-hangzhou --gateway-id gw-xxx --user-agent AlibabaCloud-Agent-Skills\n```\n\n### Incorrect: Missing cluster-id\n\n```bash\naliyun eas describe-gateway --gateway-id gw-xxx --user-agent AlibabaCloud-Agent-Skills\n```\n\n---\n\n# Diagnosis Flow Verification\n\n## 1. Service Startup Failure Diagnosis\n\n### Correct flow\n\n```bash\n# 1. Check service status\naliyun eas describe-service --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills | jq '{Status, Message}'\n\n# 2. Check failure events\naliyun eas describe-service-event --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills | jq '.Events[] | select(.Type == \"Warning\")'\n\n# 3. Check error logs\naliyun eas describe-service-log --cluster-id cn-hangzhou --service-name my-service --keyword \"error\" --limit 20 --user-agent AlibabaCloud-Agent-Skills\n```\n\n### Incorrect: Skipping status check and jumping to logs\n\n```bash\n# Wrong: Not confirming service status first\naliyun eas describe-service-log --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills\n```\n\n## 2. jq Filter Verification\n\n### Correct: Extract status information\n\n```bash\naliyun eas describe-service --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills | jq '{Status, RunningInstance, TotalInstance, Message}'\n```\n\n### Correct: Filter restart events\n\n```bash\naliyun eas describe-service-event --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills | jq '.Events[] | select(.Reason == \"Restarted\")'\n```\n\n### Incorrect: Wrong jq path\n\n```bash\n# Wrong: Field name case error\naliyun eas describe-service --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills | jq '{status, runningInstance}'\n```\n\n---\n\n# Security Rule Verification\n\n## Correct: Credential check\n\n```bash\naliyun configure list\n```\n\n## Incorrect: Reading AK/SK\n\n```bash\n# Forbidden: Reading or outputting AK/SK\necho $ALIBABA_CLOUD_ACCESS_KEY_ID\necho $ALIBABA_CLOUD_ACCESS_KEY_SECRET\n```\n\n## Incorrect: Asking user to input AK/SK\n\n```bash\n# Forbidden: Interactive credential input\nread -p \"Enter AccessKey ID: \" AK\nread -p \"Enter AccessKey Secret: \" SK\n```\n\n---\n\n# Parameter Confirmation Requirements\n\n## Correct: All user parameters must be confirmed\n\nThe following parameters must be confirmed before diagnosis:\n- Region\n- ServiceName\n- InstanceId (if querying a specific instance)\n\n## Incorrect: Using default values without confirmation\n\n```bash\n# Wrong: Using default region without asking the user\naliyun eas describe-service --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills\n```\n\n---\n\n# Error Keyword Identification\n\n## Correct: Identify common errors\n\n| Keyword | Correct Diagnosis Direction |\n|---------|-----------------------------|\n| `OOMKilled` | Out of memory, suggest increasing memory specification |\n| `ImagePullBackOff` | Image pull failure, check image address and permissions |\n| `CrashLoopBackOff` | Container startup failure, check startup command and configuration |\n| `OutOfGPU` | Insufficient GPU resources, check tp parameter or change specification |\n| `liveness probe failed` | Health check failure, adjust health check parameters |\n\n## Incorrect: Wrong diagnosis direction\n\n```markdown\n# Wrong: Diagnosing OOMKilled as a network issue\nOOMKilled → Check network configuration\n```\n\n---\n\n# Diagnosis Suggestion Verification\n\n## Correct: Provide actionable suggestions\n\n```markdown\nOOMKilled issue suggestions:\n1. Increase memory specification (e.g., upgrade from 16Gi to 32Gi)\n2. Check for memory leaks\n3. For large models, consider model quantization\n```\n\n## Incorrect: Provide vague suggestions\n\n```markdown\n# Wrong: Suggestions not specific\nOOMKilled issue suggestions:\n- Optimize memory usage\n- Check configuration\n```\n\n---\n\n# Boundary Verification\n\n## Correct: Identify Skill boundaries\n\nDiagnostic scenarios:\n- Service status check ✅\n- Log viewing and analysis ✅\n- Event analysis ✅\n- Instance status diagnosis ✅\n\nNon-diagnostic scenarios (should use other Skills):\n- Create/update/delete services → `alibabacloud-pai-eas-service-manage`\n- Start/stop/restart services → `alibabacloud-pai-eas-service-manage`\n- Auto-scaling configuration → `alibabacloud-pai-eas-service-manage`\n- Deploy new services → `alibabacloud-pai-eas-service-deploy`\n\nFile v0.0.1-beta.1:references/api-reference.md\n\n# Diagnostic API Quick Reference\n\n**Table of Contents**\n\n- [Service Status API](#service-status-api)\n- [Log API](#log-api)\n- [Event API](#event-api)\n- [Instance API](#instance-api)\n- [Diagnosis API](#diagnosis-api)\n- [Resource Group API](#resource-group-api)\n- [Gateway API](#gateway-api)\n- [Quick Diagnostic Command Summary](#quick-diagnostic-command-summary)\n\n## Service Status API\n\n### describe-service (Service Details)\n\n```bash\naliyun eas describe-service --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills\n```\n\n**Key fields**:\n\n| Field | Description | jq Path |\n|-------|-------------|---------|\n| Service name | Unique service identifier | `.ServiceName` |\n| Status | Current status | `.Status` |\n| Running instances | Normally running instances | `.RunningInstance` |\n| Total instances | Total instance count | `.TotalInstance` |\n| CPU | CPU cores | `.Cpu` |\n| Memory | Memory size | `.Memory` |\n| GPU | GPU count | `.GPU` |\n| Image | Container image | `.Image` |\n| Error message | Failure reason | `.Message` |\n\n**Status descriptions**:\n\n| Status | Description |\n|--------|-------------|\n| Creating | Being created |\n| Starting | Starting up |\n| Running | Running |\n| Updating | Being updated |\n| Stopping | Stopping |\n| Stopped | Stopped |\n| Failed | Failed |\n\n---\n\n### list-services (Service List)\n\n```bash\naliyun eas list-services --cluster-id $CLUSTER_ID --user-agent AlibabaCloud-Agent-Skills\n```\n\n**Common filters**:\n\n```bash\n# Filter failed services\naliyun eas list-services --cluster-id $CLUSTER_ID --user-agent AlibabaCloud-Agent-Skills | \\\n  jq '.Services[] | select(.Status == \"Failed\")'\n\n# Filter services by resource group\naliyun eas list-services --cluster-id $CLUSTER_ID --user-agent AlibabaCloud-Agent-Skills | \\\n  jq '.Services[] | select(.ResourceId == \"eas-r-xxx\")'\n```\n\n---\n\n### describe-service-endpoints (Service Endpoints)\n\n```bash\naliyun eas describe-service-endpoints --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills\n```\n\n**Response fields**:\n\n| Field | Description |\n|-------|-------------|\n| `.InternetEndpoint` | Public endpoint |\n| `.IntranetEndpoint` | Internal endpoint |\n| `.Token` | Access Token |\n\n---\n\n## Log API\n\n### describe-service-log (Service Logs)\n\n```bash\n# Basic query\naliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name $SERVICE --limit 100 --user-agent AlibabaCloud-Agent-Skills\n\n# Keyword filtering\naliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name $SERVICE \\\n  --keyword \"error\" --limit 50 --user-agent AlibabaCloud-Agent-Skills\n\n# Time range\naliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name $SERVICE \\\n  --start-time \"2026-03-19T00:00:00Z\" --end-time \"2026-03-19T23:59:59Z\" --user-agent AlibabaCloud-Agent-Skills\n\n# Specific instance\naliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name $SERVICE \\\n  --instance-id i-xxx --limit 50 --user-agent AlibabaCloud-Agent-Skills\n```\n\n**Parameter descriptions**:\n\n| Parameter | Description | Default |\n|-----------|-------------|---------|\n| `--limit` | Number of entries to return | 100 |\n| `--keyword` | Single keyword filter (does not support pipe-separated multiple keywords; run multiple queries for different keywords) | - |\n| `--start-time` | Start time | - |\n| `--end-time` | End time | - |\n| `--instance-id` | Specific instance | - |\n\n---\n\n## Event API\n\n### describe-service-event (Service Events)\n\n```bash\naliyun eas describe-service-event --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills\n```\n\n**Key fields**:\n\n| Field | Description |\n|-------|-------------|\n| `.Time` | Event time |\n| `.Type` | Event type |\n| `.Reason` | Event reason |\n| `.Message` | Event details |\n\n**Event types**:\n\n| Type | Description |\n|------|-------------|\n| Normal | Normal event |\n| Warning | Warning event |\n\n**Common Reasons**:\n\n| Reason | Description |\n|--------|-------------|\n| Started | Started successfully |\n| Failed | Failed |\n| Scaled | Scaled up/down |\n| Restarted | Restarted |\n| Updated | Updated |\n| Unhealthy | Unhealthy |\n\n---\n\n## Instance API\n\n### list-service-instances (Instance List)\n\n```bash\naliyun eas list-service-instances --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills\n```\n\n**Response fields**:\n\n| Field | Description |\n|-------|-------------|\n| `.InstanceId` | Instance ID |\n| `.Status` | Instance status |\n| `.IpAddress` | Instance IP |\n| `.CreateTime` | Creation time |\n| `.CpuUtilization` | CPU utilization |\n| `.MemoryUtilization` | Memory utilization |\n\n**Instance statuses**:\n\n| Status | Description |\n|--------|-------------|\n| Pending | Pending |\n| Creating | Being created |\n| Running | Running |\n| Failed | Failed |\n| Stopping | Stopping |\n| Stopped | Stopped |\n\n---\n\n### list-service-containers (Container List)\n\n> **Note**: `--instance-name` is a required parameter. Get it from `list-service-instances` first.\n\n```bash\n# Step 1: Get instance name\nINSTANCE_NAME=$(aliyun eas list-service-instances --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills | jq -r '.Instances[0].InstanceName')\n\n# Step 2: List containers\naliyun eas list-service-containers --cluster-id $CLUSTER_ID --service-name $SERVICE --instance-name \"$INSTANCE_NAME\" --user-agent AlibabaCloud-Agent-Skills\n```\n\n**Key fields**:\n\n| Field | Description |\n|-------|-------------|\n| `.ContainerId` | Container ID |\n| `.InstanceId` | Parent instance |\n| `.Status` | Container status |\n| `.RestartCount` | Restart count |\n\n---\n\n## Diagnosis API\n\n### describe-service-diagnosis (Service Diagnosis)\n\n```bash\naliyun eas describe-service-diagnosis --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills\n```\n\n**Diagnosis items**:\n\n| Diagnosis Item | Description |\n|---------------|-------------|\n| Resource usage | CPU/memory utilization |\n| Network status | Network connectivity |\n| Health status | Health check status |\n| Storage status | Storage mount status |\n\n---\n\n### describe-service-instance-diagnosis (Instance Diagnosis)\n\n```bash\naliyun eas describe-service-instance-diagnosis --cluster-id $CLUSTER_ID \\\n  --service-name $SERVICE --instance-id i-xxx --user-agent AlibabaCloud-Agent-Skills\n```\n\n---\n\n## Resource Group API\n\n### describe-resource (Resource Group Details)\n\n```bash\naliyun eas describe-resource --cluster-id $CLUSTER_ID --resource-id eas-r-xxx --user-agent AlibabaCloud-Agent-Skills\n```\n\n**Key fields**:\n\n| Field | Description |\n|-------|-------------|\n| `.Name` | Resource group name |\n| `.Status` | Resource group status |\n| `.TotalNodes` | Total node count |\n| `.HealthyNodes` | Healthy node count |\n| `.Nodes[]` | Node list |\n\n---\n\n### list-resources (Resource Group List)\n\n```bash\naliyun eas list-resources --cluster-id $CLUSTER_ID --user-agent AlibabaCloud-Agent-Skills\n```\n\n---\n\n## Gateway API\n\n### describe-gateway (Gateway Details)\n\n```bash\naliyun eas describe-gateway --cluster-id $CLUSTER_ID --gateway-id gw-xxx --user-agent AlibabaCloud-Agent-Skills\n# or\naliyun eas describe-gateway --cluster-id $CLUSTER_ID --gateway-name my-gateway --user-agent AlibabaCloud-Agent-Skills\n```\n\n**Key fields**:\n\n| Field | Description |\n|-------|-------------|\n| `.Name` | Gateway name |\n| `.LoadBalancerList[0].Status` | Load balancer status |\n| `.LoadBalancerList[0].Address` | Load balancer address |\n\n---\n\n### list-gateways (Gateway List)\n\n```bash\naliyun eas list-gateways --cluster-id $CLUSTER_ID --user-agent AlibabaCloud-Agent-Skills\n```\n\n---\n\n## Quick Diagnostic Command Summary\n\n```bash\nCLUSTER_ID=\"cn-hangzhou\"\nSERVICE=\"my-service\"\n\n# Quick view service status\nalias ds='aliyun eas describe-service --cluster-id $CLUSTER_ID --service-name'\n\n# Quick view logs\nalias dl='aliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name'\n\n# Quick view events\nalias de='aliyun eas describe-service-event --cluster-id $CLUSTER_ID --service-name'\n\n# Quick view instances\nalias di='aliyun eas list-service-instances --cluster-id $CLUSTER_ID --service-name'\n\n# Quick diagnose\nalias dd='aliyun eas describe-service-diagnosis --cluster-id $CLUSTER_ID --service-name'\n\n# Usage examples\nds $SERVICE --user-agent AlibabaCloud-Agent-Skills | jq '{Status, RunningInstance, TotalInstance}'\ndl $SERVICE --keyword \"error\" --limit 20 --user-agent AlibabaCloud-Agent-Skills\nde $SERVICE --user-agent AlibabaCloud-Agent-Skills | jq '.Events[-5:]'\ndi $SERVICE --user-agent AlibabaCloud-Agent-Skills | jq '.Instances[].Status'\ndd $SERVICE --user-agent AlibabaCloud-Agent-Skills | jq '.DiagnosisItems[]'\n```\n\nFile v0.0.1-beta.1:references/cli-installation-guide.md\n\n# Aliyun CLI Installation & Configuration Guide\n\nComplete guide for installing and configuring Aliyun CLI.\n\n> **Aliyun CLI 3.3.1+**: Supports installing and using all published Alibaba Cloud product plugins. Make sure to upgrade to 3.3.1 or later for full plugin ecosystem coverage.\n\n## Installation\n\n### macOS\n\n**Using Homebrew (Recommended)**\n```bash\nbrew install aliyun-cli\n# Upgrade to latest\nbrew upgrade aliyun-cli\n\n# Verify version (>= 3.3.1)\naliyun version\n```\n\n**Using Binary**\n```bash\n# Download\nwget https://aliyuncli.alicdn.com/aliyun-cli-macosx-latest-amd64.tgz\n\n# Extract\ntar -xzf aliyun-cli-macosx-latest-amd64.tgz\n\n# Move to PATH\nsudo mv aliyun /usr/local/bin/\n\n# Verify\naliyun version\n```\n\n### Linux\n\n**Debian/Ubuntu**\n```bash\n# Download\nwget https://aliyuncli.alicdn.com/aliyun-cli-linux-latest-amd64.tgz\n\n# Extract and install\ntar -xzf aliyun-cli-linux-latest-amd64.tgz\nsudo mv aliyun /usr/local/bin/\n\n# Verify\naliyun version\n```\n\n**CentOS/RHEL**\n```bash\n# Download\nwget https://aliyuncli.alicdn.com/aliyun-cli-linux-latest-amd64.tgz\n\n# Extract and install\ntar -xzf aliyun-cli-linux-latest-amd64.tgz\nsudo mv aliyun /usr/local/bin/\n\n# Verify\naliyun version\n```\n\n**ARM64 Architecture**\n```bash\n# Download ARM64 version\nwget https://aliyuncli.alicdn.com/aliyun-cli-linux-latest-arm64.tgz\n\n# Extract and install\ntar -xzf aliyun-cli-linux-latest-arm64.tgz\nsudo mv aliyun /usr/local/bin/\n```\n\n### Windows\n\n**Using Binary**\n1. Download from: https://aliyuncli.alicdn.com/aliyun-cli-windows-latest-amd64.zip\n2. Extract the ZIP file\n3. Add the directory to your PATH environment variable\n4. Open new Command Prompt or PowerShell\n5. Verify: `aliyun version`\n\n**Using PowerShell**\n```powershell\n# Download\nInvoke-WebRequest -Uri \"https://aliyuncli.alicdn.com/aliyun-cli-windows-latest-amd64.zip\" -OutFile \"aliyun-cli.zip\"\n\n# Extract\nExpand-Archive -Path aliyun-cli.zip -DestinationPath C:\\aliyun-cli\n\n# Add to PATH (requires admin privileges)\n$env:Path += \";C:\\aliyun-cli\"\n[Environment]::SetEnvironmentVariable(\"Path\", $env:Path, [System.EnvironmentVariableTarget]::Machine)\n\n# Verify\naliyun version\n```\n\n## Configuration\n\n### Quick Start\n\n```bash\naliyun configure set \\\n  --mode AK \\\n  --access-key-id <your-access-key-id> \\\n  --access-key-secret <your-access-key-secret> \\\n  --region cn-hangzhou\n```\n\nAll `aliyun configure` commands support non-interactive flags, which is the recommended approach —\nit works in scripts, CI/CD pipelines, and agent-driven automation without hanging on stdin prompts.\n\n**Where to Get Access Keys**\n\n1. Log in to Aliyun Console: https://ram.console.aliyun.com/\n2. Navigate to: AccessKey Management\n3. Create a new AccessKey pair\n4. Save the secret immediately — it's only shown once\n\n### Configuration Modes\n\nAliyun CLI supports 6 authentication modes. All examples below use non-interactive flags.\n\n#### 1. AK Mode (Access Key)\n\nMost common mode for personal accounts and scripts.\n\n```bash\naliyun configure set \\\n  --mode AK \\\n  --access-key-id LTAI5tXXXXXXXX \\\n  --access-key-secret 8dXXXXXXXXXXXXXXXXXXXXXXXX \\\n  --region cn-hangzhou\n```\n\nConfiguration is stored in `~/.aliyun/config.json`:\n\n```json\n{\n  \"current\": \"default\",\n  \"profiles\": [\n    {\n      \"name\": \"default\",\n      \"mode\": \"AK\",\n      \"access_key_id\": \"LTAI5tXXXXXXXX\",\n      \"access_key_secret\": \"8dXXXXXXXXXXXXXXXXXXXXXXXX\",\n      \"region_id\": \"cn-hangzhou\",\n      \"output_format\": \"json\",\n      \"language\": \"en\"\n    }\n  ]\n}\n```\n\n#### 2. StsToken Mode (Temporary Credentials)\n\nFor short-lived access (tokens expire in 1-12 hours).\n\n```bash\naliyun configure set \\\n  --mode StsToken \\\n  --access-key-id LTAI5tXXXXXXXX \\\n  --access-key-secret 8dXXXXXXXXXXXXXXXXXXXXXXXX \\\n  --sts-token v1.0:XXXXXXXXXXXXXXXX \\\n  --region cn-hangzhou\n```\n\nUse cases: CI/CD pipelines, temporary access for external contractors, cross-account access.\n\n#### 3. RamRoleArn Mode (Assume RAM Role)\n\nAssume a RAM role for elevated or cross-account access.\n\n```bash\naliyun configure set \\\n  --mode RamRoleArn \\\n  --access-key-id LTAI5tXXXXXXXX \\\n  --access-key-secret 8dXXXXXXXXXXXXXXXXXXXXXXXX \\\n  --ram-role-arn acs:ram::123456789012:role/AdminRole \\\n  --role-session-name my-session \\\n  --region cn-hangzhou\n```\n\nUse cases: cross-account resource access, temporary elevated privileges, role-based access control.\n\n#### 4. EcsRamRole Mode (ECS Instance RAM Role)\n\nUse the RAM role attached to an ECS instance — no credentials needed.\n\n```bash\naliyun configure set \\\n  --mode EcsRamRole \\\n  --ram-role-name MyEcsRole \\\n  --region cn-hangzhou\n```\n\nRequirements: must be running on an ECS instance with a RAM role attached.\n\nUse cases: scripts and automation running on ECS instances.\n\n#### 5. RsaKeyPair Mode (RSA Key Pair)\n\nUse RSA key pair for authentication (generate key pair in Aliyun Console first).\n\n```bash\naliyun configure set \\\n  --mode RsaKeyPair \\\n  --private-key /path/to/private-key.pem \\\n  --key-pair-name my-key-pair \\\n  --region cn-hangzhou\n```\n\n#### 6. RamRoleArnWithEcs Mode (ECS + RAM Role)\n\nCombine ECS instance role with RAM role assumption for cross-account access from ECS.\n\n```bash\naliyun configure set \\\n  --mode RamRoleArnWithEcs \\\n  --ram-role-name MyEcsRole \\\n  --ram-role-arn acs:ram::123456789012:role/TargetRole \\\n  --role-session-name my-session \\\n  --region cn-hangzhou\n```\n\n### Environment Variables\n\n**Highest priority** - overrides config file\n\n**Access Key Mode**\n```bash\nexport ALIBABA_CLOUD_ACCESS_KEY_ID=your_access_key_id\nexport ALIBABA_CLOUD_ACCESS_KEY_SECRET=your_access_key_secret\nexport ALIBABA_CLOUD_REGION_ID=cn-hangzhou\n```\n\n**STS Token Mode**\n```bash\nexport ALIBABA_CLOUD_ACCESS_KEY_ID=your_access_key_id\nexport ALIBABA_CLOUD_ACCESS_KEY_SECRET=your_access_key_secret\nexport ALIBABA_CLOUD_SECURITY_TOKEN=your_sts_token\nexport ALIBABA_CLOUD_REGION_ID=cn-hangzhou\n```\n\n**ECS RAM Role Mode**\n```bash\nexport ALIBABA_CLOUD_ECS_METADATA=role_name\n```\n\n**Use Case**:\n- CI/CD pipelines\n- Docker containers\n- Temporary credential override\n\n### Managing Multiple Profiles\n\n**Create Named Profiles**\n\n```bash\naliyun configure set --profile projectA \\\n  --mode AK \\\n  --access-key-id LTAI5tAAAAAAAA \\\n  --access-key-secret 8dAAAAAAAAAAAAAAAAAAAAAAAA \\\n  --region cn-hangzhou\n\naliyun configure set --profile projectB \\\n  --mode AK \\\n  --access-key-id LTAI5tBBBBBBBB \\\n  --access-key-secret 8dBBBBBBBBBBBBBBBBBBBBBBBB \\\n  --region cn-shanghai\n```\n\n**Use Specific Profile**\n\n```bash\naliyun ecs describe-instances --profile projectA\n\nexport ALIBABA_CLOUD_PROFILE=projectA\naliyun ecs describe-instances   # Uses projectA\n```\n\n**List and Switch Profiles**\n\n```bash\naliyun configure list                      # List all profiles\naliyun configure set --current projectA    # Switch default profile\n```\n\n### Credential Priority\n\nCredentials are loaded in this order (first found wins):\n\n1. **Command-line flag**: `--profile <name>`\n2. **Environment variable**: `ALIBABA_CLOUD_PROFILE`\n3. **Environment credentials**: `ALIBABA_CLOUD_ACCESS_KEY_ID`, etc.\n4. **Configuration file**: `~/.aliyun/config.json` (current profile)\n5. **ECS Instance RAM Role**: If running on ECS with attached role\n\n## Verification\n\n### Test Authentication\n\n```bash\n# Basic test - list regions\naliyun ecs describe-regions\n\n# Expected output: JSON array of regions\n```\n\n**If successful**, you'll see:\n```json\n{\n  \"Regions\": {\n    \"Region\": [\n      {\n        \"RegionId\": \"cn-hangzhou\",\n        \"RegionEndpoint\": \"ecs.cn-hangzhou.aliyuncs.com\",\n        \"LocalName\": \"华东 1（杭州）\"\n      },\n      ...\n    ]\n  },\n  \"RequestId\": \"...\"\n}\n```\n\n**If failed**, you'll see error messages:\n- `InvalidAccessKeyId.NotFound` - Wrong Access Key ID\n- `SignatureDoesNotMatch` - Wrong Access Key Secret\n- `InvalidSecurityToken.Expired` - STS token expired (for StsToken mode)\n- `Forbidden.RAM` - Insufficient permissions\n\n### Debug Configuration\n\n```bash\n# Show current configuration\naliyun configure get\n\n# Test with debug logging\naliyun ecs describe-regions --log-level=debug\n\n# Check credential provider\naliyun configure get mode\n```\n\n## Security Best Practices\n\n### 1. Use RAM Users (Not Root Account)\n\n❌ **Don't**: Use Aliyun root account credentials\n✅ **Do**: Create RAM users with specific permissions\n\n```bash\n# Create RAM user in console\n# Attach only necessary policies\n# Use RAM user's access keys\n```\n\n### 2. Principle of Least Privilege\n\nGrant only the minimum permissions needed:\n\n```bash\n# Example: Read-only ECS access\n# Attach policy: AliyunECSReadOnlyAccess\n```\n\n### 3. Rotate Access Keys Regularly\n\n```bash\n# Create new access key in RAM Console, then update configuration\naliyun configure set --access-key-id NEW_KEY --access-key-secret NEW_SECRET\n# Delete old access key from console\n```\n\n### 4. Use STS Tokens for Temporary Access\n\n```bash\naliyun configure set --mode StsToken \\\n  --access-key-id XXXX --access-key-secret XXXX \\\n  --sts-token XXXX --region cn-hangzhou\n```\n\n### 5. Use ECS RAM Roles When Possible\n\n```bash\naliyun configure set --mode EcsRamRole --ram-role-name MyRole --region cn-han","readmeExcerpt":"Skill: Alibabacloud Pai Eas Service Diagnose Owner: sdk-team Summary: PAI-EAS service diagnosis and troubleshooting. Diagnose startup failures, error logs, slow responses, instance restarts, OOMKilled, ImagePullBackOff, CrashLo... Tags: latest:0.0.1 Version history: v0.0.1 | 2026-07-13T08:18:20.514Z | auto alibabacloud-pai-eas-service-diagnose v0.0.1 – Initial Release - Provides autonomous diagnosis and troubleshooti","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"brew install aliyun-cli\naliyun version"},{"language":"bash","snippet":"aliyun configure set --auto-plugin-install true\naliyun plugin update"},{"language":"bash","snippet":"# Generate once per session (32-char lowercase hex), reuse for every call\nSESSION_ID=$(openssl rand -hex 16)\nUA=\"AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose/${SESSION_ID}\"\naliyun plugin update"},{"language":"bash","snippet":"--user-agent AlibabaCloud-Agent-Skills/{SKILL_NAME}/{session-id}"},{"language":"bash","snippet":"SESSION_ID=$(openssl rand -hex 16)   # 32 hex chars"},{"language":"bash","snippet":"# Generate once per session\nSESSION_ID=$(openssl rand -hex 16)\nUA=\"AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose/${SESSION_ID}\"\n\n# Pass it on every command, e.g.:\naliyun eas describe-service --cluster-id <region> --service-name <service> \\\n  --user-agent \"AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose/${SESSION_ID}\""}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: alibabacloud-pai-eas-service-diagnose\ndescription: |\n  PAI-EAS service diagnosis and troubleshooting. Diagnose startup failures, error logs,\n  slow responses, instance restarts, OOMKilled, ImagePullBackOff, CrashLoopBackOff,\n  GPU errors, health check failures, liveness probe issues, service inaccessible.\n  \n  When to use: Diagnose EAS service issues - startup failures, logs, slow responses,\n  restarts, OOMKilled, ImagePullBackOff, CrashLoopBackOff, GPU errors, health checks,\n  service inaccessible, gateway issues, liveness probe failed.\n  Triggers: \"服务启动失败\", \"服务Failed\", \"看日志\", \"实例重启\", \"响应慢\",\n  \"OOMKilled\", \"ImagePullBackOff\", \"CrashLoopBackOff\", \"CUDA out of memory\",\n  \"GPU内存不足\", \"liveness probe\", \"服务访问不了\".\n  \n  Not for: deploying (use service-deploy), managing create/update/delete/stop/restart/scale\n  (use service-manage), listing services (use service-manage), DLC/DSW, non-EAS products.\nlicense: Apache-2.0\nmetadata:\n  version: \"1.0.0\"\n  domain: aiops\n  owner: pai-eas-team\n  contact: pai-eas-agent@alibaba-inc.com\n  tags:\n    - pai-eas\n    - diagnosis\n    - troubleshooting\n    - log-analysis\n    - service-health\n  required_tools:\n    - aliyun\n    - jq\n  required_permissions:\n    - \"eas:DescribeService\"\n    - \"eas:DescribeServiceLog\"\n    - \"eas:DescribeServiceEvent\"\n    - \"eas:DescribeServiceDiagnosis\"\n    - \"eas:DescribeServiceInstanceDiagnosis\"\n    - \"eas:ListServiceInstances\"\n    - \"eas:ListServiceContainers\"\n    - \"eas:ListServices\"\n    - \"eas:DescribeResource\"\n    - \"eas:DescribeGateway\"\n---\n\n# PAI-EAS Service Operations Diagnosis\n\nHelps users diagnose issues with running PAI-EAS services.\n\n> **🔴🔴 [MUST — READ BEFORE ANYTHING ELSE] Two failure modes that cause task failure:**\n>\n> 1. **Listing is NOT diagnosing.** After `ListServices`/`list-services` returns\n>    the target service, you MUST immediately continue and run the FULL diagnostic\n>    workflow (`DescribeService` → `DescribeServiceEvent` → `DescribeServiceLog`\n>    multi-keyword → `ListServiceInstances` → `ListServiceContainers` →\n>    `DescribeServiceDiagnosis`) on a chosen service. NEVER stop after listing.\n> 2. **Always finish with an analysis section.** Every diagnosis MUST end with a\n>    written health-analysis + recommendation section. If the user described a\n>    symptom (e.g. \"CUDA out of memory\") that the live data does NOT show, you\n>    MUST still add a secondary section explaining common causes and fixes for\n>    that described symptom. Never end the turn with only raw command output.\n\n---\n\n## Installation\n\nInstall the Aliyun CLI (>= 3.3.3, required baseline for plugin ecosystem\ncompatibility) using a trusted package manager. On macOS, use Homebrew:\n\n```bash\nbrew install aliyun-cli\naliyun version\n```\n\nOn other platforms, download the versioned release tarball directly from\n`https://aliyuncli.alicdn.com/`, verify its checksum, and extract it manually\nbefore use (see the [CLI Installation Guide](references/cli-installation-guide.md)).\nDo not pipe remote scripts"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn74p5w8ywv6prh40g0s82gmqh83nw54\",\n  \"slug\": \"alibabacloud-pai-eas-service-diagnose\",\n  \"version\": \"0.0.1\",\n  \"publishedAt\": 1783930700514\n}"},{"path":"references/acceptance-criteria.md","content":"# Acceptance Criteria: alibabacloud-pai-eas-service-diagnose\n\n**Scenario**: PAI-EAS Service Diagnosis\n**Purpose**: Skill test acceptance criteria\n\n---\n\n# Correct CLI Command Patterns\n\n## 1. EAS Service Diagnostic Operations\n\n### Correct: Query service status\n\n```bash\naliyun eas describe-service --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### Incorrect: Missing --user-agent\n\n```bash\naliyun eas describe-service --cluster-id cn-hangzhou --service-name my-service\n```\n\n### Incorrect: Using API format instead of plugin mode\n\n```bash\naliyun eas DescribeService --region cn-hangzhou --ServiceName my-service\n```\n\n## 2. Log Query\n\n### Correct: Keyword filtering\n\n```bash\naliyun eas describe-service-log --cluster-id cn-hangzhou --service-name my-service --keyword \"error\" --limit 20 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### Incorrect: Missing --service-name\n\n```bash\naliyun eas describe-service-log --cluster-id cn-hangzhou --keyword \"error\" --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n## 3. Event Query\n\n### Correct: Query service events\n\n```bash\naliyun eas describe-service-event --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### Correct: Filter Warning events\n\n```bash\naliyun eas describe-service-event --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | jq '.Events[] | select(.Type == \"Warning\")'\n```\n\n## 4. Instance Query\n\n### Correct: List instances\n\n```bash\naliyun eas list-service-instances --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### Correct: List containers (requires --instance-name)\n\n```bash\n# First get instance name\naliyun eas list-service-instances --cluster-id cn-hangzhou --service-name my-service --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | jq -r '.Instances[0].InstanceName'\n\n# Then list containers with instance-name\naliyun eas list-service-containers --cluster-id cn-hangzhou --service-name my-service --instance-name my-service-xxxxx-yyyyy --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n## 5. Gateway Query\n\n### Correct: Query gateway details\n\n```bash\naliyun eas describe-gateway --cluster-id cn-hangzhou --gateway-id gw-xxx --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n### Incorrect: Missing cluster-id\n\n```bash\naliyun eas describe-gateway --gateway-id gw-xxx --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n---\n\n# Diagnosis Flow Verification\n\n## 1. Service Startup Failure Diagnosis\n\n### Correct flow\n\n```bash\n# 1. Check service status\naliyun eas describe-service --cluster-id cn-hangzhou --service-name my-"},{"path":"references/api-reference.md","content":"# Diagnostic API Quick Reference\n\n**Table of Contents**\n\n- [Service Status API](#service-status-api)\n- [Log API](#log-api)\n- [Event API](#event-api)\n- [Instance API](#instance-api)\n- [Diagnosis API](#diagnosis-api)\n- [Resource Group API](#resource-group-api)\n- [Gateway API](#gateway-api)\n- [Quick Diagnostic Command Summary](#quick-diagnostic-command-summary)\n\n## Service Status API\n\n### describe-service (Service Details)\n\n```bash\naliyun eas describe-service --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n**Key fields**:\n\n| Field | Description | jq Path |\n|-------|-------------|---------|\n| Service name | Unique service identifier | `.ServiceName` |\n| Status | Current status | `.Status` |\n| Running instances | Normally running instances | `.RunningInstance` |\n| Total instances | Total instance count | `.TotalInstance` |\n| CPU | CPU cores | `.Cpu` |\n| Memory | Memory size | `.Memory` |\n| GPU | GPU count | `.GPU` |\n| Image | Container image | `.Image` |\n| Error message | Failure reason | `.Message` |\n\n**Status descriptions**:\n\n| Status | Description |\n|--------|-------------|\n| Creating | Being created |\n| Starting | Starting up |\n| Running | Running |\n| Updating | Being updated |\n| Stopping | Stopping |\n| Stopped | Stopped |\n| Failed | Failed |\n\n---\n\n### list-services (Service List)\n\n```bash\naliyun eas list-services --cluster-id $CLUSTER_ID --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n**Common filters**:\n\n```bash\n# Filter failed services\naliyun eas list-services --cluster-id $CLUSTER_ID --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '.Services[] | select(.Status == \"Failed\")'\n\n# Filter services by resource group\naliyun eas list-services --cluster-id $CLUSTER_ID --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose | \\\n  jq '.Services[] | select(.ResourceId == \"eas-r-xxx\")'\n```\n\n---\n\n### describe-service-endpoints (Service Endpoints)\n\n```bash\naliyun eas describe-service-endpoints --cluster-id $CLUSTER_ID --service-name $SERVICE --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n```\n\n**Response fields**:\n\n| Field | Description |\n|-------|-------------|\n| `.InternetEndpoint` | Public endpoint |\n| `.IntranetEndpoint` | Internal endpoint |\n| `.Token` | Access Token |\n\n---\n\n## Log API\n\n### describe-service-log (Service Logs)\n\n```bash\n# Basic query\naliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name $SERVICE --limit 100 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n\n# Keyword filtering\naliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name $SERVICE \\\n  --keyword \"error\" --limit 50 --user-agent AlibabaCloud-Agent-Skills/alibabacloud-pai-eas-service-diagnose\n\n# Time range\naliyun eas describe-service-log --cluster-id $CLUSTER_ID --service-name $SERVICE \\\n  --start-time \"2026-03-19T00:00:0"},{"path":"references/cli-installation-guide.md","content":"# Aliyun CLI Installation & Configuration Guide\n\nComplete guide for installing and configuring Aliyun CLI.\n\n> **Aliyun CLI 3.3.3+**: Supports installing and using all published Alibaba Cloud product plugins. Make sure to upgrade to 3.3.3 or later for full plugin ecosystem coverage.\n\n## Installation\n\n> **Note on `sudo mv ... /usr/local/bin/`**: The binary-install steps below use\n> `sudo` only to place the `aliyun` executable into the system-wide `/usr/local/bin`\n> directory, which requires administrator privileges. This is the standard,\n> well-understood way to make a CLI available on `PATH` for all users. If you\n> prefer NOT to use elevated privileges, install into a user-local directory\n> instead and add it to your `PATH` — no `sudo` required:\n>\n> ```bash\n> mkdir -p \"$HOME/.local/bin\"\n> mv aliyun \"$HOME/.local/bin/\"\n> export PATH=\"$HOME/.local/bin:$PATH\"   # add to ~/.bashrc or ~/.zshrc to persist\n> ```\n\n### macOS\n\n**Using Homebrew (Recommended)**\n```bash\nbrew install aliyun-cli\n# Upgrade to latest\nbrew upgrade aliyun-cli\n\n# Verify version (>= 3.3.3)\naliyun version\n```\n\n**Using Binary**\n```bash\n# Download\nwget https://aliyuncli.alicdn.com/aliyun-cli-macosx-latest-amd64.tgz\n\n# Extract\ntar -xzf aliyun-cli-macosx-latest-amd64.tgz\n\n# Move to PATH\nsudo mv aliyun /usr/local/bin/\n\n# Verify\naliyun version\n```\n\n### Linux\n\n**Debian/Ubuntu**\n```bash\n# Download\nwget https://aliyuncli.alicdn.com/aliyun-cli-linux-latest-amd64.tgz\n\n# Extract and install\ntar -xzf aliyun-cli-linux-latest-amd64.tgz\nsudo mv aliyun /usr/local/bin/\n\n# Verify\naliyun version\n```\n\n**CentOS/RHEL**\n```bash\n# Download\nwget https://aliyuncli.alicdn.com/aliyun-cli-linux-latest-amd64.tgz\n\n# Extract and install\ntar -xzf aliyun-cli-linux-latest-amd64.tgz\nsudo mv aliyun /usr/local/bin/\n\n# Verify\naliyun version\n```\n\n**ARM64 Architecture**\n```bash\n# Download ARM64 version\nwget https://aliyuncli.alicdn.com/aliyun-cli-linux-latest-arm64.tgz\n\n# Extract and install\ntar -xzf aliyun-cli-linux-latest-arm64.tgz\nsudo mv aliyun /usr/local/bin/\n```\n\n### Windows\n\n**Using Binary**\n1. Download from: https://aliyuncli.alicdn.com/aliyun-cli-windows-latest-amd64.zip\n2. Extract the ZIP file\n3. Add the directory to your PATH environment variable\n4. Open new Command Prompt or PowerShell\n5. Verify: `aliyun version`\n\n**Using PowerShell**\n```powershell\n# Download\nInvoke-WebRequest -Uri \"https://aliyuncli.alicdn.com/aliyun-cli-windows-latest-amd64.zip\" -OutFile \"aliyun-cli.zip\"\n\n# Extract\nExpand-Archive -Path aliyun-cli.zip -DestinationPath C:\\aliyun-cli\n\n# Add to PATH (requires admin privileges)\n$env:Path += \";C:\\aliyun-cli\"\n[Environment]::SetEnvironmentVariable(\"Path\", $env:Path, [System.EnvironmentVariableTarget]::Machine)\n\n# Verify\naliyun version\n```\n\n## Configuration\n\n### Quick Start\n\n```bash\naliyun configure set \\\n  --mode AK \\\n  --access-key-id <your-access-key-id> \\\n  --access-key-secret <your-access-key-secret> \\\n  --region cn-hangzhou\n```\n\nAll `aliyun configure` commands support non-interactive flags, which is"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":1390,"uniquenessScore":44,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T05:56:37.011Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T05:56:37.011Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T08:44:10.590Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}