{"id":"2983399c-4f1f-486e-9ae6-d023604712a9","entityType":"agent","slug":"clawhub-sdk-team-alibabacloud-sysom-diagnosis","name":"alibabacloud-sysom-diagnosis","canonicalUrl":"https://www.xpersona.co/agent/clawhub-sdk-team-alibabacloud-sysom-diagnosis","canonicalPath":"/agent/clawhub-sdk-team-alibabacloud-sysom-diagnosis","generatedAt":"2026-10-10T17:36:55.469Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T14:45:50.559Z","emptyReason":null},"description":"Use when troubleshooting Linux server performance or stability issues — CPU saturation, high load, scheduling delay, memory pressure, OOM events, high RSS, page cache / shared memory growth, memory cgroup residue, Java heap issues, disk IO saturation or latency, packet loss, network jitter, or a server that is slow, stuck, or unstable. Performs diagnosis and surfaces recommendations; does not apply fixes automatically.","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.4K downloads reported by the source. Last updated 10/10/2026.","installCommand":"clawhub skill install s173swjet2yrebzqrp6hjkvmy583mxef:alibabacloud-sysom-diagnosis","sourceUrl":"https://clawhub.ai/sdk-team/alibabacloud-sysom-diagnosis","homepage":"https://clawhub.ai/sdk-team/skills/alibabacloud-sysom-diagnosis","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/sdk-team/alibabacloud-sysom-diagnosis","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/sdk-team/skills/alibabacloud-sysom-diagnosis","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":63,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"alibabacloud-sysom-diagnosis technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-10T14:45:50.559Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T14:45:50.559Z","emptyReason":null},"stars":null,"forks":null,"downloads":1378,"packageName":null,"latestVersion":"0.0.7","tractionLabel":"1.4K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T14:45:50.559Z","emptyReason":null},"lastUpdatedAt":"2026-10-10T14:45:50.559Z","lastCrawledAt":"2026-10-10T14:45:50.559Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-11T14:45:50.559Z","lastVerifiedAt":null,"highlights":[{"version":"0.0.7","createdAt":"2026-09-08T03:49:51.292Z","changelog":"alibabacloud-sysom-diagnosis v0.0.7 changelog - Added manifest.json for clearer skill metadata. - Expanded CLI setup instructions: now includes explicit Linux/macOS guidance, user-local install, and environment variable steps. - Updated compatibility section to note control host OS support (now: Linux or macOS, not Windows). - Improved documentation on credential impact: details on command visibility and explanation of differences in available commands based on credentials. - Removed legacy skill-card.md file. - Documentation and reference guides updated for clarity and completeness on supported environments and usage instructions.","fileCount":23,"zipByteSize":46479},{"version":"0.0.6","createdAt":"2026-08-22T01:41:24.292Z","changelog":"- Removed the skill-card.md file, consolidating documentation into primary sources. - No feature or functional changes made in this version.","fileCount":22,"zipByteSize":43285},{"version":"0.0.5","createdAt":"2026-08-20T03:47:42.920Z","changelog":"**Expanded diagnosis workflow with Java deep-dive and mandatory reference loading.** - Added comprehensive Java troubleshooting flow: new guides for Java memory, GC, CPU, glossary, profiling, and routing; diagnosis now requires loading these references to interpret findings and next steps. - Requires loading and applying domain-specific reference content before all major answer decisions, even if agent findings appear complete. - Introduced multi-hop guided-diagnosis loop rules for Java: command execution only as specified, session hop ceilings, mandatory user confirmation before profiling steps, and session ownership discipline. - Java domain now split into GC, memory, and CPU subroutes; new references enable accurate routing and entity interpretation for each. - Updated domain routing and workflow to prioritize safety, explicit user consent, and bounded fallback to raw Linux checks only as allowed. - Removed deprecated files and refactored documentation to reflect new diagnosis flow and strict answer-shaping rules.","fileCount":22,"zipByteSize":43492},{"version":"0.0.4","createdAt":"2026-06-16T02:04:51.757Z","changelog":"**Major update: Documentation and workflow overhaul, simplified structure, and safer usage.** - Replaced detailed, process-heavy documentation with a concise, user-oriented guide. - Centralized all remote diagnostic references and step guides into new, consolidated markdown files. - Removed extensive subcommand, parameter, and raw Linux command details; now relies on backend envelope output and focused command routing. - Clear separation of core workflow, domain/classification routing, and credential handling for safer operation. - Added multiple new reference guides: classification, deep actions, parameter and environment guides, memory and non-memory triage, and report interpretation. - Greatly reduced in-repo documentation files; legacy diagnosis guides and scripts removed.","fileCount":11,"zipByteSize":16692},{"version":"0.0.3","createdAt":"2026-05-15T03:47:00.378Z","changelog":"- Major documentation update: all skill documentation and usage guides are now available in both English and Chinese. - SKILL.md rewritten in English for broader accessibility. - References and diagnosis guide documents translated and expanded for clarity. - No changes to core logic or CLI scripts; only documentation files updated. - Improved quick start and subcommand explanations for easier onboarding.","fileCount":102,"zipByteSize":147084},{"version":"0.0.2","createdAt":"2026-04-16T09:49:48.778Z","changelog":"- Major CLI refactor: switched diagnosis entrypoint from shared/scripts/osops to scripts/osops.sh. - Added new scripts for CLI-based diagnosis: init.sh, osops.sh, and full CLI source under scripts/sysom_cli/. - Introduced new documentation: authentication, CLI development guide, metadata API, and OpenAPI permission guides. - Updated agent workflow: now requires Aliyun CLI (≥3.3.3), AI-Mode enable/disable steps, and explicit plugin setup before execution. - Enforced that remote/advanced diagnosis must be routed through SysOM InvokeDiagnosis API with osops.sh, disallowing fallback to ECS generic diagnosis. - Documentation and command examples revised to reflect CLI changes and new script locations.","fileCount":101,"zipByteSize":159019},{"version":"0.0.1","createdAt":"2026-04-15T06:33:58.244Z","changelog":"Initial release with deep diagnostic capabilities for Linux system issues - Provides in-depth analysis and diagnosis for memory, network, IO, and system load issues on Linux systems. - Outputs structured diagnostic results in a standard JSON envelope format. - Supports both quick local diagnostics and advanced remote diagnostics triggered by specific subcommands. - Implements intent-based routing for memory, IO, network, and load diagnostics based on observed symptoms. - Guides users through authentication and precheck steps as needed.","fileCount":19,"zipByteSize":25315}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s173swjet2yrebzqrp6hjkvmy583mxef:alibabacloud-sysom-diagnosis","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s173swjet2yrebzqrp6hjkvmy583mxef:alibabacloud-sysom-diagnosis` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/sdk-team/alibabacloud-sysom-diagnosis before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-sdk-team-alibabacloud-sysom-diagnosis/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-sdk-team-alibabacloud-sysom-diagnosis/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-sdk-team-alibabacloud-sysom-diagnosis/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-sdk-team-alibabacloud-sysom-diagnosis/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-sdk-team-alibabacloud-sysom-diagnosis/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-sdk-team-alibabacloud-sysom-diagnosis/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T17:36:55.464Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-sdk-team-alibabacloud-sysom-diagnosis/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-sdk-team-alibabacloud-sysom-diagnosis/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-sdk-team-alibabacloud-sysom-diagnosis/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-sdk-team-alibabacloud-sysom-diagnosis/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T14:45:50.559Z","emptyReason":null},"readme":"Skill: alibabacloud-sysom-diagnosis\n\nOwner: sdk-team\n\nSummary: Use when troubleshooting Linux server performance or stability issues — CPU saturation, high load, scheduling delay, memory pressure, OOM events, high RSS, page cache / shared memory growth, memory cgroup residue, Java heap issues, disk IO saturation or latency, packet loss, network jitter, or a server that is slow, stuck, or unstable. Performs diagnosis and surfaces recommendations; does not apply fixes automatically.\n\nTags: latest:0.0.7\n\nVersion history:\n\nv0.0.7 | 2026-09-08T03:49:51.292Z | auto\n\nalibabacloud-sysom-diagnosis v0.0.7 changelog\n\n- Added manifest.json for clearer skill metadata.\n- Expanded CLI setup instructions: now includes explicit Linux/macOS guidance, user-local install, and environment variable steps.\n- Updated compatibility section to note control host OS support (now: Linux or macOS, not Windows).\n- Improved documentation on credential impact: details on command visibility and explanation of differences in available commands based on credentials.\n- Removed legacy skill-card.md file.\n- Documentation and reference guides updated for clarity and completeness on supported environments and usage instructions.\n\nv0.0.6 | 2026-08-22T01:41:24.292Z | auto\n\n- Removed the skill-card.md file, consolidating documentation into primary sources.\n- No feature or functional changes made in this version.\n\nv0.0.5 | 2026-08-20T03:47:42.920Z | auto\n\n**Expanded diagnosis workflow with Java deep-dive and mandatory reference loading.**\n\n- Added comprehensive Java troubleshooting flow: new guides for Java memory, GC, CPU, glossary, profiling, and routing; diagnosis now requires loading these references to interpret findings and next steps.\n- Requires loading and applying domain-specific reference content before all major answer decisions, even if agent findings appear complete.\n- Introduced multi-hop guided-diagnosis loop rules for Java: command execution only as specified, session hop ceilings, mandatory user confirmation before profiling steps, and session ownership discipline.\n- Java domain now split into GC, memory, and CPU subroutes; new references enable accurate routing and entity interpretation for each.\n- Updated domain routing and workflow to prioritize safety, explicit user consent, and bounded fallback to raw Linux checks only as allowed.\n- Removed deprecated files and refactored documentation to reflect new diagnosis flow and strict answer-shaping rules.\n\nv0.0.4 | 2026-06-16T02:04:51.757Z | auto\n\n**Major update: Documentation and workflow overhaul, simplified structure, and safer usage.**\n\n- Replaced detailed, process-heavy documentation with a concise, user-oriented guide.\n- Centralized all remote diagnostic references and step guides into new, consolidated markdown files.\n- Removed extensive subcommand, parameter, and raw Linux command details; now relies on backend envelope output and focused command routing.\n- Clear separation of core workflow, domain/classification routing, and credential handling for safer operation.\n- Added multiple new reference guides: classification, deep actions, parameter and environment guides, memory and non-memory triage, and report interpretation.\n- Greatly reduced in-repo documentation files; legacy diagnosis guides and scripts removed.\n\nv0.0.3 | 2026-05-15T03:47:00.378Z | auto\n\n- Major documentation update: all skill documentation and usage guides are now available in both English and Chinese.\n- SKILL.md rewritten in English for broader accessibility.\n- References and diagnosis guide documents translated and expanded for clarity.\n- No changes to core logic or CLI scripts; only documentation files updated.\n- Improved quick start and subcommand explanations for easier onboarding.\n\nv0.0.2 | 2026-04-16T09:49:48.778Z | auto\n\n- Major CLI refactor: switched diagnosis entrypoint from shared/scripts/osops to scripts/osops.sh.\n- Added new scripts for CLI-based diagnosis: init.sh, osops.sh, and full CLI source under scripts/sysom_cli/.\n- Introduced new documentation: authentication, CLI development guide, metadata API, and OpenAPI permission guides.\n- Updated agent workflow: now requires Aliyun CLI (≥3.3.3), AI-Mode enable/disable steps, and explicit plugin setup before execution.\n- Enforced that remote/advanced diagnosis must be routed through SysOM InvokeDiagnosis API with osops.sh, disallowing fallback to ECS generic diagnosis.\n- Documentation and command examples revised to reflect CLI changes and new script locations.\n\nv0.0.1 | 2026-04-15T06:33:58.244Z | auto\n\nInitial release with deep diagnostic capabilities for Linux system issues\n\n- Provides in-depth analysis and diagnosis for memory, network, IO, and system load issues on Linux systems.\n- Outputs structured diagnostic results in a standard JSON envelope format.\n- Supports both quick local diagnostics and advanced remote diagnostics triggered by specific subcommands.\n- Implements intent-based routing for memory, IO, network, and load diagnostics based on observed symptoms.\n- Guides users through authentication and precheck steps as needed.\n\nArchive index:\n\nArchive v0.0.7: 23 files, 46479 bytes\n\nFiles: references/classify-output-guide.md (2196b), references/deep-actions.md (4337b), references/java-triage.md (228b), references/java/cpu/cpu-guide.md (3286b), references/java/gc/gc-guide.md (4342b), references/java/memory/case-library.md (5842b), references/java/memory/decision-tree.md (3340b), references/java/memory/glossary.md (4645b), references/java/memory/javamem-envelope-guide.md (9044b), references/java/memory/memory-guide.md (3625b), references/java/memory/profiling-interpretation.md (5829b), references/java/memory/profiling-playbook.md (3432b), references/java/README.md (3515b), references/manifest.json (19b), references/memory-triage.md (5472b), references/non-memory-triage.md (2153b), references/parameter-guide.md (4600b), references/ram-policies.md (1221b), references/report-interpretation.md (2788b), references/supported-environments.md (1835b), skill-card.md (2893b), SKILL.md (21283b), _meta.json (147b)\n\nFile v0.0.7:SKILL.md\n\n---\nname: alibabacloud-sysom-diagnosis\ndescription: >\n  Use when troubleshooting Linux server performance or stability issues —\n  CPU saturation, high load, scheduling delay, memory pressure, OOM events,\n  high RSS, page cache / shared memory growth, memory cgroup residue, Java\n  heap issues, disk IO saturation or latency, packet loss, network jitter,\n  or a server that is slow, stuck, or unstable. Performs diagnosis and\n  surfaces recommendations; does not apply fixes automatically.\nlicense: Apache-2.0\ncompatibility: >\n  Requires sysom-osops CLI. The control host running the CLI can be Linux or\n  macOS (x86_64 or aarch64); Windows is not supported. Remote diagnosis targets\n  a Linux ECS instance and requires Alibaba Cloud credentials through AK/SK or\n  an ECS RAM Role, an online Cloud Assistant on the target ECS, and a supported\n  China Mainland or Hong Kong region.\nmetadata:\n  domain: aiops\n  product: sysom\n  supported_domains:\n    - cpu\n    - io\n    - memory\n    - network\n    - java\n  owner: sysom-team\n  contact: sysom-team@alibaba-inc.com\nallowed-tools: Bash Read\n---\n\n# alibabacloud-sysom-diagnosis\n\nUse SysOM CLI and backend envelopes as the diagnosis source of truth. This Skill replaces the older SysOM diagnosis Skill and is the single entry point for SysOM ECS performance and stability diagnosis.\n\n## Immediate Route\n\nWhen the user reports a symptom and has not provided fresh SysOM envelope output,\nrun the matching SysOM command from **Domain Routing** below before ad hoc Linux\ninspection or manual probing. Then follow the returned `agent.summary`,\n`agent.findings[].detail/category`, and `agent.next_steps[]`. Raw Linux commands\nare bounded fallbacks only when a SysOM command is unavailable, outputs\ncontradict each other, or a required entity remains missing after the focused\nSysOM command.\n\n## Credential Security\n\nNever print, echo, or ask for AccessKey ID or AccessKey Secret values. Remote\ncommands perform their own authentication checks. If a command returns an\nauthentication or permission error, explain the error and point the user to\n`references/ram-policies.md`; credential setup must happen outside the\nconversation.\n\n## CLI Setup\n\nCheck whether the CLI is available:\n\n```bash\ncommand -v sysom-osops\n```\n\nIf it is missing, install it. The installer runs on **Linux or macOS** — it does\nnot run on Windows. The target ECS being diagnosed must be Linux (see\n`references/supported-environments.md`), but the control host can be either OS.\n\nSystem-wide install (needs write access to `/usr/local/bin`, typically via\n`sudo`):\n\n```bash\ncurl -fsSL --connect-timeout 1000 https://sysom-prd-cn-hangzhou.oss-cn-hangzhou.aliyuncs.com/sysom_prd/skill_cli/install.sh \\\n  | sudo bash\n```\n\nUser-local install — no sudo, no root-owned paths. Works on both Linux and\nmacOS, and is the recommended path when you do not have administrator\nprivileges:\n\n```bash\nmkdir -p ~/.local/bin\ncurl -fsSL --connect-timeout 1000 https://sysom-prd-cn-hangzhou.oss-cn-hangzhou.aliyuncs.com/sysom_prd/skill_cli/install.sh \\\n  | bash -s -- -d \"$HOME/.local/bin\"\n```\n\nThen make sure the install directory is on your PATH (e.g. `~/.bashrc` /\n`~/.zshrc`):\n\n```bash\nexport PATH=\"$HOME/.local/bin:$PATH\"\n```\n\nThen verify only the binary:\n\n```bash\ncommand -v sysom-osops\n```\n\nOn macOS, the installer performs ad-hoc codesign and strips quarantine\nattributes automatically. If you are on Apple Silicon running an x86_64 shell\nunder Rosetta, the installer detects the mismatch; pass `-f` to override only\nwhen you know the binary will run under translation.\n\n## Command Visibility Depends On Credentials\n\nOnly local commands such as `memory classify` are always present. Every remote\ndeep command is discovered at runtime from the SysOM skills catalog, which needs\ncredentials. On a machine without credentials configured, expect:\n\n- `sysom-osops memory --help` to list only `classify`, with the deep memory\n  commands absent.\n- The top-level `sysom-osops --help` to omit the `io`, `net`, and `load` groups\n  entirely.\n\nThis is a visibility limitation, not a capability limitation. Treat\n`references/deep-actions.md` as the authoritative command inventory for this\nSkill, and never infer from `--help` output that a domain or command is\nunsupported. Use `sysom-osops precheck` to report auth status.\n\n## Core Workflow\n\n1. Classify the user's symptom into one SysOM domain: memory, IO, load/CPU,\n   network, or Java (GC/memory/CPU).\n2. Run the smallest SysOM command that matches that domain. Prefer a local\n   memory classify for unclear memory symptoms; for other domains, use the\n   matching documented remote action.\n3. Read only the default envelope fields: `ok`, `error`, `command`, and\n   `agent`.\n4. **Load domain references before building the answer.** This step is mandatory\n   and must not be skipped even when `agent.findings` and `agent.next_steps`\n   appear complete. Which references to load depends on the domain:\n   - Java (any type: gc/memory/cpu) → read `references/java/README.md` first\n     for symptom routing and parameter validation; then by type:\n     - gc: `references/java/gc/gc-guide.md`\n     - memory: `references/java/memory/memory-guide.md` (then glossary, envelope\n       guide, profiling playbook, decision tree under `references/java/memory/`)\n     - cpu: `references/java/cpu/cpu-guide.md`\n   - Other domains → load the matching reference from the References table below.\n   References add interpretation rules, entity definitions, and answer-shaping\n   guidance that the envelope alone does not convey. Do not infer Java terms,\n   native memory categories, or profiling semantics from raw envelope text.\n5. Relay the hop as visible progress: present `agent.summary` (plus key\n   findings) to the user, interpreted through the reference material loaded in\n   step 4. Keep evidence qualifiers that change interpretation, including\n   currentness, unavailable direct signals, fallback evidence, and remediation\n   preconditions.\n6. Branch on `agent.status`:\n   - `concluded` (or missing/unrecognized) → build the final answer from\n     `agent.summary`, `agent.findings[].detail/category`, and\n     `agent.next_steps[]`, then stop the loop.\n   - `in_progress` → the backend is requesting another collection hop: take\n     the `kind=command` entry from `agent.next_steps[]`, apply the\n     confirmation rules below, run it, and feed the new envelope back into\n     step 4.\n\n**Guided-diagnosis loop hard rules** (Java multi-hop sessions):\n\n- Pace ownership: never skip `agent.next_steps[]` to decide collection on\n  your own, and never run diagnostic commands outside the envelope.\n- Run commands **exactly as shown** — the gateway has already injected\n  `--session-id`; never rewrite, add, or remove flags. If a command fails,\n  relay the error envelope as-is instead of retrying with tweaked parameters.\n- Hop limit: stop the loop after 4 hops in the same session even if the\n  backend still says `in_progress`; present the conclusions so far and state\n  the evidence limits (the backend force-concludes at the same ceiling —\n  double safety).\n- User refusal: if the user declines a proposed command, stop the loop,\n  summarize from the evidence already collected, and state explicitly which\n  conclusions remain unconfirmed because that hop was not run.\n\n**Before executing a profiling or long-running follow-up command** (e.g.,\n`java analyze --type memory --duration N`, any command that injects an agent\ninto the target process, or any command expected to run for multiple minutes):\n- Tell the user what the command does, how long it takes, and what performance\n  impact it may have on the target process (e.g., CPU overhead from sampling,\n  extra memory from the injected agent, potential safepoint pauses).\n- Ask the user whether to proceed. Do not run the command until the user\n  confirms, or until the user has previously given a standing instruction to\n  auto-run follow-ups.\n- Once the command starts, tell the user the expected wait time and keep them\n  informed if the operation is still in progress.\n\nRead-only query commands (e.g. `java analyze --type cpu`) do not apply here —\nsee each domain guide's Execution Model for specifics.\n\nWhen classify returns a command in `agent.next_steps[]` and no root-cause\nfinding already contains enough evidence to answer, run the first command next.\nDo not replace an Agent-visible SysOM next step with manual shell probing. Raw\nLinux checks are bounded fallbacks after the SysOM next step succeeds, fails, or\ntimes out.\n\nUse the documented commands exactly as shown by default. Do not add raw,\ndebug, or backend evidence expansion flags unless the user explicitly asks for\nthat view.\n\nFinal answers should name evidence, root cause, owner/scope, and operational\naction targets. Do not add shell snippets for verification or remediation unless\nthe user explicitly asks for commands. Prefer phrases such as \"review dependency\nand disable or upgrade the leaking component in a change window\" over raw module,\ncgroup, sysctl, cache-drop, or process-kill commands.\nDo not include command-looking inline snippets such as module inspection/removal,\nmemory summary commands, cgroup file writes, cache-drop controls, sysctl changes,\nor process-kill commands as default final-answer steps.\n\nThe `agent` view must be self-contained for diagnosis. Structured evidence is a\nbackend/UI view and must not be treated as the default Agent source for required\nentities.\n\n## Domain Routing\n\n| User symptom | First route |\n|--------------|-------------|\n| Unclear memory issue, OOM, high RSS, file cache, shmem/tmpfs, memory cgroup, socket memory, kernel memory | `sysom-osops memory classify` |\n| Java issue (symptom unclear) | Follow the Symptom Triage rules in `references/java/README.md` — ask the user about the symptom, then route to the matching type |\n| Java GC pause / frequent GC / low GC throughput | `sysom-osops java analyze --type gc` |\n| Java heap / OOM / heap leak / native leak | `sysom-osops java analyze --type memory` — without a pid/pod it returns a candidate list; STOP and wait for the user to choose before retrying |\n| Java CPU hotspot / high thread CPU / flame graph | `sysom-osops java analyze --type cpu --pid <PID>` |\n| Slow disk, high iowait, disk latency, blocked IO | `sysom-osops io iofsstat`, then `io iodiagnose` if the overview points to slow IO |\n| High load, runqueue backlog, task stuck waiting for CPU | `sysom-osops load loadtask` or `load delay` based on the visible symptom |\n| Packet loss, retransmits, network timeout, jitter | `sysom-osops net packetdrop` for loss/drop symptoms; `net netjitter` for latency fluctuation |\n\nFor command parameters, read `references/deep-actions.md` and\n`references/parameter-guide.md`. For OS and region support, read\n`references/supported-environments.md`. These references are Skill material; do\nnot use remote target file tools to open `.claude/skills` paths on the diagnosed\nhost.\nFor Java symptom-to-type routing, consult `references/java/README.md`.\n\n## Memory Routing\n\nMemory follows the same Core Workflow and Follow-up Rules as every domain: start\nfrom `sysom-osops memory classify`, then pick the next action from visible output\nor `agent.next_steps[]`. For choosing among memory deep actions or checking which\nentity is still missing, load `references/memory-triage.md` (parallel to\n`references/non-memory-triage.md` for other domains).\n\nNote: Java-related memory symptoms (OOM in Java process, heap leak, native leak)\nroute to Java domain via `sysom-osops java analyze --type memory`, not through\nmemory classify. See Domain Routing table above.\n\nChoose the next memory action from visible SysOM output. Do not infer a memory mechanism from symptom wording alone.\n\n## Envelope Contract\n\nDefault command output is the Agent contract:\n\n```json\n{\n  \"ok\": true,\n  \"command\": \"sysom-osops memory classify\",\n  \"agent\": {\n    \"status\": \"concluded\",\n    \"session_id\": \"a1b2c3d4e5f6\",\n    \"summary\": \"Concise diagnosis summary.\",\n    \"findings\": [\n      {\n        \"severity\": \"high\",\n        \"title\": \"Short finding title\",\n        \"detail\": \"Root cause, key entities, and evidence summary.\",\n        \"category\": \"root_cause\"\n      }\n    ],\n    \"next_steps\": [\n      {\n        \"kind\": \"command\",\n        \"label\": \"Run focused deep diagnosis\",\n        \"command\": \"sysom-osops memory oom\",\n        \"reason\": \"The missing entity this command can fill.\"\n      }\n    ]\n  }\n}\n```\n\n`agent.findings[]` may contain only `severity`, `title`, `detail`, and\n`category`. Required entities such as PID, cgroup, service, file path, OOM\nvictim, limit/current, residue, holder, or cleanup target must be written in\n`agent.summary` or `agent.findings[].detail`.\n\nField semantics for guided diagnosis sessions:\n\n- `agent.status`: `in_progress` means the backend diagnosis agent requests\n  another collection hop; `concluded` means diagnosis has converged. Treat a\n  missing or unrecognized value as `concluded`. Legacy collector-level states\n  (`success`, `warning`, ...) may still appear on non-Java actions; interpret\n  them as before.\n- `agent.session_id`: backend-generated multi-hop session identifier. Never\n  generate or modify it; the gateway already injects it into `command`\n  strings, so run them verbatim.\n- `agent.next_steps[].kind`: `command` = backend-requested collection command\n  (subject to the confirmation rules above); `info` = user-side suggestion —\n  present it but never auto-run; `warning` = evidence or data-quality caveat.\n\n## Follow-up Rules\n\n- Prefer `category=root_cause`, then highest severity, then the finding that\n  best matches the user's reported symptom.\n- Treat `root_cause` as stop-ready when visible `detail` contains the entities\n  needed to explain the symptom and a safe next action.\n- Treat `agent.next_steps[]` as a priority plan, not a checklist.\n- Run another SysOM command only when it can fill a named missing entity or\n  change remediation.\n- For long-running Java collection commands — `java analyze --type memory\n  --duration N` (profiling; legacy `memory javamem --duration N`) and `java\n  analyze --type gc` in `collect` mode (5–10 min JFR/GC collection) — the wait is\n  minutes-scale: tell the user, size the tool timeout accordingly, and never\n  re-fire the same command on client timeout. See\n  `references/java/memory/profiling-playbook.md` (memory) and\n  `references/java/gc/gc-guide.md` (gc).\n- Preserve visible qualifiers that affect interpretation, such as current versus\n  historical evidence, unavailable direct signals, fallback evidence used to\n  close currentness, and safety preconditions for remediation.\n- When a finding uses fallback evidence because a direct signal is unavailable,\n  state both parts in the final answer. Do not reduce the conclusion to the\n  fallback metric alone.\n- After a focused SysOM command closes a root cause, answer from it. Do not run\n  extra commands to make the report comprehensive, and do not chase earlier\n  classify anomalies or observations unless they share the same entity and\n  expose a named evidence gap.\n- Do not call backend-only collectors or private helper commands directly.\n- Do not re-check a PID, cgroup, file, limit, or event that SysOM already named\n  in `summary` or `detail`.\n- After a SysOM deep command returns `category=root_cause` with the required\n  entities visible, answer from that envelope. Raw Linux checks are only for\n  contradictions, command errors, or a clearly missing entity.\n- In the final answer, do not turn already-closed entities into extra raw Linux\n  verification commands. Express remediation as dependency-aware action targets\n  and change-window plans unless the envelope itself provides an executable safe\n  next step.\n- Avoid executable shell snippets in the final answer. If a command is useful\n  only for post-change verification, name the SysOM check or metric to re-run\n  instead of raw Linux commands.\n- This includes inline command names for module inspection/removal, memory\n  summary commands, cgroup file writes, cache-drop controls, sysctl changes, and\n  process-kill actions; describe the dependency gate and operational action\n  target in prose.\n- Pivot across domains when the current envelope does not explain the reported\n  symptom and another SysOM domain names a stronger root cause.\n- During diagnosis, do not execute remediation commands that change target\n  state, such as killing processes, removing files, changing sysctl values, or\n  writing to cache-drop controls. Present those as recommendations unless the\n  user explicitly asks you to perform the repair.\n- For non-memory findings, keep the same rule: one focused deep command, then\n  answer when the required entities are visible.\n\n## Error Handling\n\n| `error.code` | Action |\n|--------------|--------|\n| `Sysom.TargetRequired` | Ask for instance ID and region, or explain ECS metadata auto-detection requirements |\n| `Sysom.FallbackClassify` | Present the local classify result and continue only if a focused next step is available |\n| `Sysom.PermissionDenied` | Use `references/ram-policies.md` to explain required RAM permissions |\n| `Sysom.AuthenticationFailure` | Ask the user to configure credentials outside this session |\n| `Sysom.InvalidParameter` | Ask the user to correct the instance, region, or command parameter |\n| `Sysom.DiagnosisVersionNotSupported` | Explain that the target instance diagnosis components need an update |\n| `Sysom.DiagnosisJsonParseFailed` | Retry once only when the user still needs the same evidence |\n| `Sysom.PollError` | Retry the same focused action once when the missing evidence is still required |\n\n### Empty Output Is Not an Envelope\n\nA command can exit non-zero with **no stdout and no stderr at all**. This is not\nan envelope, so do not parse it — parsing empty output as JSON will fail. The\ndominant cause is an unsupported flag: the CLI rejects an undefined flag before\nany envelope is produced, and currently swallows the message.\n\nWhen a command produces no output:\n\n1. Do not retry the same command unchanged, and do not report a diagnosis result.\n2. Check the flags you passed against `references/parameter-guide.md`, and\n   confirm with `sysom-osops <group> <command> --help`. Note that `io`, `load`,\n   and `net` commands accept only `--region`, `--instance`, and `--scope`.\n3. Re-run once with the unsupported flags removed.\n4. If the output is still empty, tell the user the command failed without a\n   diagnosable error, name the command and flags used, and treat it the same as\n   `Sysom.InvalidParameter` instead of inventing findings.\n\n### Help Text Is Not an Envelope\n\nA domain subcommand can be **missing** rather than broken. Remote deep commands\nare discovered at runtime from the SysOM skills catalog, which requires\ncredentials. When credentials are absent the catalog is unreachable, the\nsubcommand is never registered, and the CLI falls back to printing the domain\ngroup's help text — with **exit code 0**.\n\nTreat output that begins with `Commands under \"<domain>\" are discovered at\nruntime` as a missing command, never as a diagnosis result:\n\n1. Do not parse it as an envelope and do not report \"no issue found\". Exit code\n   0 here means the command never ran.\n2. Do not conclude that the domain is unsupported, or that this Skill only\n   offers `memory classify`.\n3. Tell the user that deep diagnosis needs credentials. Point them to\n   `sysom-osops precheck` for auth status and `sysom-osops configure` to set it\n   up; credential setup happens outside the conversation.\n4. Re-run the command only after the user confirms credentials are configured.\n\n## References\n\n| Reference | Use when |\n|-----------|----------|\n| `references/classify-output-guide.md` | Reading local memory classify output |\n| `references/memory-triage.md` | Choosing a memory deep action or checking memory entity completeness |\n| `references/non-memory-triage.md` | Routing IO, load/CPU, network diagnosis |\n| `references/deep-actions.md` | Looking up SysOM commands by domain |\n| `references/parameter-guide.md` | Validating command parameters |\n| `references/report-interpretation.md` | Interpreting envelope fields and answer shape |\n| `references/ram-policies.md` | Explaining RAM permissions |\n| `references/supported-environments.md` | Checking OS, architecture, and region support |\n\n### Java Analysis References\n\n| Reference | Use when |\n|-----------|----------|\n| `references/java/README.md` | **Primary Java entry point**: symptom triage, parameter guide, sub-domain index |\n| `references/java/gc/gc-guide.md` | Running or interpreting `--type gc` results |\n| `references/java/cpu/cpu-guide.md` | Running or interpreting `--type cpu` results |\n| `references/java/memory/memory-guide.md` | `--type memory` interpretation and discovery-first flow entry |\n| `references/java/memory/glossary.md` | Java memory terminology |\n| `references/java/memory/javamem-envelope-guide.md` | Interpreting `--type memory` envelope structure |\n| `references/java/memory/profiling-playbook.md` | Preparation and expected behavior before `--duration` collection |\n| `references/java/memory/decision-tree.md` | Following backend `next_steps` in Java multi-hop sessions |\n| `references/java/memory/case-library.md` | Case library and anti-patterns |\n\nFile v0.0.7:references/java/README.md\n\n# Java Application Diagnosis Reference\n\nUnified command for all Java diagnostics:\n\n```\nsysom-osops java analyze --type <gc|memory|cpu>\n```\n\nThis document is the entry point for Java application diagnosis —\ncovering symptom triage, sub-domain guides, parameter quick-reference,\nand combined-diagnosis strategies. Each sub-domain (gc / memory / cpu)\nhas its own folder under `references/java/`.\n\n---\n\n## Symptom Triage\n\n### When the user's intent is unclear\n\nIf the user describes a vague Java problem (e.g., \"is my Java process healthy\",\n\"something seems wrong with my app\") without mentioning specific symptoms from\nthe table below, ask **one** clarifying question:\n\n> What is the primary symptom you are observing?\n> 1. Application response is slow or has intermittent pauses\n> 2. CPU usage is abnormally high\n> 3. Memory keeps growing or OOM has occurred\n\nThen route based on the answer:\n- 1 → `--type gc` (latency-related symptoms are most often GC-induced)\n- 2 → `--type cpu`\n- 3 → `--type memory`\n\n### Symptom → Diagnosis Type Mapping\n\nWhen the symptom is already clear, route directly:\n\n| Symptom | Recommended type | Description |\n|---------|-----------------|-------------|\n| Long GC pause / STW jitter | gc | Analyze GC event time-series and pause distribution |\n| Frequent Full GC / old generation full | gc | Detect memory pressure source and promotion patterns |\n| Low GC throughput / high application pause ratio | gc | Evaluate GC algorithm efficiency and tuning recommendations |\n| OOM / sustained heap growth | memory | Deep diagnosis of heap/non-heap leaks |\n| Native memory leak | memory | Analyze JNI/DirectBuffer usage |\n| High thread CPU usage | cpu | Flame graph to locate hotspot method stacks |\n| Slow application response (non-GC cause) | cpu | Analyze on-CPU time distribution |\n\n---\n\n## Sub-Domain Reference Index\n\n| type | Folder / entry guide | Scope |\n|------|----------------------|-------|\n| gc | [gc/gc-guide.md](gc/gc-guide.md) | GC pauses, throughput, heap trend analysis |\n| memory | [memory/memory-guide.md](memory/memory-guide.md) (+ `glossary`, `javamem-envelope-guide`, `decision-tree`, `profiling-playbook`, `profiling-interpretation`, `case-library` in `memory/`) | Heap/non-heap diagnosis, snapshots, allocation profiling |\n| cpu | [cpu/cpu-guide.md](cpu/cpu-guide.md) | Flame graph, hotspot method stacks, on-CPU analysis |\n\n---\n\n## Parameter Quick-Reference\n\n| Parameter | gc | memory | cpu |\n|-----------|:---:|:------:|:---:|\n| --pid | Optional | Optional | Optional (omit to get candidate list) |\n| --duration | Seconds (default 60) | Minutes (0=snapshot) | N/A |\n| --pod | Optional (container scenarios) | Optional | Not supported |\n\n---\n\n## Combined Diagnosis\n\nWhen a single type cannot locate the root cause, combine multiple types:\n\n1. **GC causing high CPU**: First `--type gc` to confirm GC frequency, then `--type cpu` to check GC thread proportion\n2. **Memory leak causing frequent GC**: First `--type gc` to observe Full GC patterns, then `--type memory` for deeper heap analysis\n3. **Heavy object allocation found in CPU hotspots**: First `--type cpu` to locate allocation hotspots, then `--type memory --duration 5` to track allocation trends\n\n---\n\n## Prerequisites\n\n- `--type cpu` requires the instance to be managed in the Alibaba Cloud Linux console (the system auto-detects and provides guidance)\n\n---\n\nFor memory-specific interpretation, the discovery-first flow, and answer\ndiscipline, see [memory/memory-guide.md](memory/memory-guide.md).\n\nFile v0.0.7:_meta.json\n\n{\n  \"ownerId\": \"kn74p5w8ywv6prh40g0s82gmqh83nw54\",\n  \"slug\": \"alibabacloud-sysom-diagnosis\",\n  \"version\": \"0.0.7\",\n  \"publishedAt\": 1788839391292\n}\n\nFile v0.0.7:references/classify-output-guide.md\n\n# Classify Output Guide\n\nUse `sysom-osops memory classify` as the local entry point for unclear memory\nsymptoms. It returns the same default Agent envelope as remote deep actions.\n\n## What To Read\n\n| Field | Use |\n|-------|-----|\n| `agent.summary` | Overall memory verdict and dominant issue. |\n| `agent.findings[].category` | Prefer `root_cause` when present. |\n| `agent.findings[].detail` | Agent-visible root-cause entities and missing evidence. |\n| `agent.next_steps[]` | Focused remote actions that can fill missing evidence. |\n\nDo not call backend-only collectors from the Skill. Do not depend on legacy or\ninternal fields outside the default Agent envelope.\n\n## Choosing The First Follow-Up\n\n1. If a root-cause finding already contains the required entities, answer from\n   classify.\n2. If the dominant finding lacks a required entity, run the first relevant\n   `kind=command` next step.\n3. If the visible envelope names an ownership, currentness, attribution, or\n   cleanup gap, follow the focused SysOM action that closes that entity gap.\n4. If a later deep action returns concrete holder or owner entities, combine\n   them with classify instead of re-running broad local checks.\n\n## Next Steps May Not Be Runnable Yet\n\n`agent.next_steps[]` can recommend remote deep commands such as\n`memory filecache` or `memory memcgoffline` even though classify itself ran\nlocally without credentials. Those commands are registered only while the skills\ncatalog is reachable, so on an unconfigured machine they do not exist yet.\n\nA `status: normal` classify result therefore does not imply the recommended\nfollow-up is available. If the follow-up returns domain help text or empty\noutput, treat it as a missing command per the rules in `SKILL.md`, tell the user\ndeep analysis needs credentials, and do not fall back to inventing findings or to\nbroad manual shell probing.\n\n## When No Memory Action Is Needed\n\nIf classify returns no memory finding, or all findings are informational and do\nnot match the user's symptom, report that current SysOM memory evidence is\nhealthy or inconclusive. Pivot to IO, load, network, or Java only when the\nvisible envelope or the user's symptom supports that domain.\n\nFile v0.0.7:references/deep-actions.md\n\n# Deep Actions Reference\n\nThis file lists public SysOM commands that this Skill may route to. On an ECS\ninstance, sysom-osops can auto-detect `--instance` and `--region`; add them only\nfor cross-instance diagnosis or when auto-detection fails.\n\n## Memory\n\n| Command | Mode | Use when |\n|---------|------|----------|\n| `sysom-osops memory classify` | Local | First route for unclear memory symptoms, OOM hints, high RSS, cache growth, shmem/tmpfs, memcg residue, or kernel memory suspicion |\n| `sysom-osops memory memgraph` | Remote | Full memory landscape is missing after classify, or the issue is broad kernel/userspace memory composition |\n| `sysom-osops memory memgraph --enable-socket` | Remote | Socket buffer pressure is visible and socket holder, state, PID, cgroup, or service attribution is missing |\n| `sysom-osops memory process` | Remote | A process is the suspected holder but identity, real executable, cgroup, or service attribution is missing |\n| `sysom-osops memory oom` | Remote | OOM killer event or memcg/host OOM evidence is visible |\n| `sysom-osops memory filecache` | Remote | File/page cache is the dominant unresolved entity and file/holder attribution is missing |\n| `sysom-osops memory shmem` | Remote | shmem, tmpfs, memfd, or SysV shared memory holder attribution is missing |\n| `sysom-osops memory memleak` | Remote | Kernel-hidden growth (vmalloc/page-allocator/slab/percpu) is the dominant unresolved entity and the leaking call point/function/module is missing after memgraph closed only to a candidate (`--type slab\\|page\\|vmalloc\\|percpu`, default vmalloc) |\n| `sysom-osops memory javamem` | Remote | **DEPRECATED** — Use `sysom-osops java analyze --type memory` instead (see Java section below) |\n| `sysom-osops memory memcgoffline` | Remote | Cgroup ownership-transition evidence is visible in SysOM output and original ownership, residue/refcount, or cleanup order needs attribution |\n\n## IO\n\n| Command | Mode | Use when |\n|---------|------|----------|\n| `sysom-osops io iofsstat` | Remote | Disk IO overview is needed for high iowait, slow disk, or IO saturation |\n| `sysom-osops io iodiagnose` | Remote | Slow IO root-cause attribution is needed after the overview or when latency is the primary symptom |\n\n## Load and CPU Scheduling\n\n| Command | Mode | Use when |\n|---------|------|----------|\n| `sysom-osops load loadtask` | Remote | Load average, runqueue, or task composition is the primary symptom |\n| `sysom-osops load delay` | Remote | Runnable tasks are not getting CPU time, scheduling delay is reported, or processes appear stuck without IO evidence |\n\n## Network\n\n| Command | Mode | Use when |\n|---------|------|----------|\n| `sysom-osops net packetdrop` | Remote | Packet loss, retransmits, drops, connection resets, or timeout symptoms are primary |\n| `sysom-osops net netjitter` | Remote | Latency fluctuation, jitter, or intermittent connectivity degradation is primary |\n\n## Java\n\nRoute by symptom through the unified `sysom-osops java analyze` entry point. See `references/java/README.md` for details.\n\n| Command | Scenario |\n|---------|----------|\n| `sysom-osops java analyze --type gc` | GC pause analysis, frequent GC, throughput diagnosis |\n| `sysom-osops java analyze --type gc --duration 120` | Extend collection duration (default 60 seconds) |\n| `sysom-osops java analyze --type memory` | Java heap/non-heap memory diagnosis (equivalent to memory javamem) |\n| `sysom-osops java analyze --type memory --pid <PID> --duration 5` | Continuous collection of a specific process for 5 minutes |\n| `sysom-osops java analyze --type cpu --pid <PID>` | CPU hotspot flame graph analysis (pid optional; if omitted, returns the Java process candidate list first) |\n\n> Note: `--type cpu` locates the target by `--pid` only and **does not support `--pod`**; if `--pid` is omitted it returns the candidate list first. `--pid` is also optional for `memory` and `gc`.\n> duration units: gc = seconds (default 60), memory = minutes (0 = snapshot mode).\n\n## Common Notes\n\n- All public commands return the same envelope shape: `ok`, `error`,\n  `command`, and `agent`.\n- Continue from `agent.next_steps[]` only when a required entity is missing or\n  the next action changes remediation.\n- Do not call backend-only collectors from the Skill. Public commands above are\n  the supported interface.\n\nFile v0.0.7:references/java-triage.md\n\n# Java Problem Triage Guide\n\n> This document has been merged into [references/java/README.md](java/README.md).\n> Please refer to the Java Diagnosis Reference for the full symptom triage,\n> parameter guide, and sub-domain index.\n\nFile v0.0.7:references/java/cpu/cpu-guide.md\n\n<!--\n * @Descripttion: \n * @version: \n * @Author: Jietao Xiao\n * @Date: 2026-07-16 12:12:42\n * @LastEditors: Jietao Xiao\n * @LastEditTime: 2026-07-17 11:05:03\n-->\n# CPU Hotspot Analysis Guide\n\n## Applicable Scenarios\n- High thread CPU usage\n- Slow application response (GC already ruled out)\n- Flame graph needed to locate hotspot method stacks\n\n## Command\n\n`sysom-osops java analyze --type cpu --pid <PID>`\n\n| Parameter | Description |\n|-----------|-------------|\n| --pid | Recommended — target Java process PID. If omitted, the call returns a Java process candidate list first |\n\n> If you are unsure of the PID, run `--type cpu` **without**\n> a pid directly — the CPU path performs Java process discovery itself and\n> returns the candidate list. After the\n> user picks a PID, re-run `--type cpu --pid <PID>` for the actual flame graph.\n\n## Execution Model\n\n`--type cpu` is a **read-only query** over profiling data that is **already being\ncollected continuously**. On-CPU profiling starts automatically once the instance\nis managed in the console, and samples are stored server-side (the `prof_on`\ndataset). The command retrieves roughly the **last 5 minutes** of already-collected\nsamples for the target PID and runs LLM analysis on them.\n\nBecause of this:\n- It does **not** launch an on-demand perf run against the target, does **not**\n  inject any agent, and adds **no measurable CPU overhead** to the process.\n- It returns quickly (query + analysis) — there is **no ~60s sampling wait**.\n- There is **no `--duration`** for cpu (see parameter table), and it does **not**\n  need the pre-profiling overhead confirmation from the skill's general profiling\n  guidance.\n\nWhen confirming this action with the user, describe it as \"view the last ~5\nminutes of continuously-collected CPU flame graph data\". Do **not** say it\nperforms ~60s perf sampling or imposes CPU load on the target — that describes\non-demand profiling (e.g. `--type memory --duration N`), not this path.\n\n## Prerequisites\n\n- **The instance must be managed in the Alibaba Cloud Linux console** (the system auto-detects; returns guidance when not managed)\n\n## Result Interpretation\n\nReturns flame graph LLM analysis conclusions:\n- Hotspot method stacks and call chains\n- On-CPU time proportion per method\n- Potential performance bottleneck identification\n\n### Common Patterns\n\n| Pattern | Symptoms | Recommendation |\n|---------|----------|----------------|\n| Spin lock contention | Significant time spent on lock/CAS operations | Check concurrent data structure usage |\n| Frequent object allocation | Hotspots concentrated on new/allocate paths | Combine with `--type memory` to analyze allocation trends |\n| System call blocking | Hotspots on read/write/poll system calls | Check IO or network blocking |\n| Serialization/deserialization | High proportion of JSON/Protobuf processing | Evaluate serialization strategy or data volume |\n\n## Cross-Domain Diagnosis\n\n- **Heavy object allocation found in CPU hotspots** → Combine with `--type memory --duration 5` to track allocation trends\n- **Suspected GC threads consuming CPU** → First `--type gc` to confirm GC frequency and pause duration\n- **High CPU but no obvious flame graph hotspot** → Likely an off-CPU issue (waiting on IO/locks); check IO domain\n\nFile v0.0.7:references/java/gc/gc-guide.md\n\n# Java GC Diagnosis Guide\n\n## Diagnosis Mode Selection\n\n### Log Analysis Mode (gclog_only)\n\n**Applicable scenarios**: Java process has GC logging configured (JDK8: `-XX:+PrintGCDetails -Xloggc:<path>`; JDK11+: `-Xlog:gc*:file=<path>`)\n\n**Characteristics**:\n- Returns results immediately (no collection wait)\n- Analyzes historical GC logs for retrospective troubleshooting\n- No additional performance overhead\n\n**Limitations**: Requires an existing GC log file; unavailable if JDK8 has no logging configured\n\n**Parameters**: `diagMode=gclog_only`, duration parameter is ignored\n\n### Incremental Collection Mode (collect)\n\n**Applicable scenarios**: No GC logging configured, or precise JFR time-series data needed, or JDK11+ environment\n\n**Characteristics**:\n- Collects JFR + GC logs + OS metrics (5-10 minutes)\n- Does not require pre-configured GC logging (JDK11+ can enable dynamically)\n- Provides complete time-series, heap trends, and OS correlation analysis\n\n**Limitations**: Requires waiting for the duration period; JDK8 cannot enable JFR dynamically and will fall back to GC log collection\n\n**Parameters**: `diagMode=collect`, `duration=5` (default; recommended 5-10 minutes)\n\n## Mode Selection Decision Tree\n\n1. User explicitly mentions \"GC logs\" or \"logging is enabled\" → `gclog_only`\n2. User describes performance issues without mentioning logs → Ask whether GC logging is enabled\n3. User is unsure or has not configured logging → `collect` (more comprehensive, no prerequisites)\n4. JDK8 environment + no logging configured → `collect` (automatically falls back to log collection mode)\n\n## Agent Reply Template\n\nWhen the user has not specified a mode, ask:\n\n> GC diagnosis supports two modes:\n> 1. **Log analysis**: directly analyze existing GC logs and return results immediately. Suitable when the Java process is already configured with `-Xlog:gc*` or `-XX:+PrintGCDetails`.\n> 2. **Incremental collection**: collect 5-10 minutes of JFR + GC data in real time for deep analysis. No prior GC log configuration required, but you must wait for collection to complete.\n>\n> Is GC logging already enabled on your Java process?\n\n## Command\n\n`sysom-osops java analyze --type gc [--diagMode <gclog_only|collect>] [--duration <minutes>] [--pid <PID>]`\n\n| Parameter | Description |\n|-----------|-------------|\n| --diagMode | Diagnosis mode: gclog_only / collect, default collect |\n| --duration | Collection duration (minutes), default 5 for collect mode, ignored for gclog_only |\n| --pid | Optional; auto-detects Java process when not specified |\n| --pod | Optional; specifies pod name in container scenarios |\n\n## Collection execution discipline (collect mode)\n\n`collect` mode samples JFR + GC logs for **5–10 minutes** (minutes-scale, like\nmemory profiling):\n\n- Tell the user the expected wait before starting; they can continue other work.\n- Size the shell/tool timeout to at least **`(duration + 3)` minutes**; do not\n  use 120–180 second timeouts for a 5-minute collection.\n- Launch **once**. On client timeout with no envelope, do **not** auto-re-fire\n  the same command; say it may still be running and ask whether to wait or check\n  task status.\n\n## Result Interpretation\n\nThe envelope contains GC diagnosis analysis results:\n- GC event time-series and pause distribution\n- Throughput assessment and optimization recommendations\n- Heap memory usage trends\n\n### Common Patterns\n\n| Pattern | Symptoms | Recommendation |\n|---------|----------|----------------|\n| Frequent Full GC | Old generation full, promotion rate too high | Combine with `--type memory` for deeper heap analysis |\n| Young GC jitter | Object allocation rate fluctuations causing unstable STW frequency | Check application-layer caching or batch processing logic |\n| Mixed GC reclamation lag | G1 Mixed GC frequency cannot keep up with object promotion | Adjust InitiatingHeapOccupancyPercent |\n| Excessive GC overhead | Application pause ratio > 10% | Evaluate GC algorithm selection and heap size configuration |\n\n\n## Cross-Domain Diagnosis\n\n- **GC causing high CPU** → Combine with `--type cpu` to analyze GC thread stack proportion\n- **Root cause of frequent Full GC** → Combine with `--type memory` for heap analysis to locate leak source\n- **Memory not released after GC** → Combine with `--type memory --duration 5` to track allocation trends\n\nFile v0.0.7:references/java/memory/case-library.md\n\n# Java Memory Case Library\n\nStructured examples for Agent narrative regression. Envelope fields are\nsummarized; numbers are illustrative of real patterns.\n\n## Case template\n\nEach case includes:\n\n- **Context**: user symptom\n- **Envelope summary**: key summary / findings / next_steps\n- **Correct narrative**: what to tell a non-expert user\n- **Incorrect narrative**: patterns to avoid\n\n---\n\n## Case A: High off-heap native, empty profiling (anti-pattern replay)\n\n### Context\n\nUser: \"Java process memory high on instance i-xxx.\"\n\nAgent ran snapshot javamem, then user agreed to profiling.\n\n### Snapshot envelope (summary)\n\n- `summary`: \"Java process 3307705 has memory issues\"\n- `findings`:\n  - `javamem.native_memory`: JNI/Other usage high; NMT not enabled\n  - `javamem.rss_gap`: RSS−JVM gap ~3.97 GB; split: heap gap ~10 MB, JNI/Other ~3.96 GB, glibc ~6 MB\n- `next_steps`:\n  - `command`: `sysom-osops memory javamem --pid 3307705 --duration 5`\n  - reason: capture native allocation path\n\n### Profiling envelope (summary)\n\n- First attempt: **client timeout 180s** (no envelope)\n- Second attempt: `ok: true`, `summary`: profiling did not get usable Top stacks\n- `findings`: `data_quality` — no alloc/leak Top stacks parsed\n- Optional: `job_id` present in backend evidence\n\n### Correct narrative\n\n> The Java process **3307705** is using about **3.96 GB** in **off-heap native\n> memory (JNI/Other)**, not in the Java object heap (only ~10 MB heap accounting\n> gap). The OS sees roughly **3.97 GB** more RSS than the JVM reports; that matches\n> native/off-heap dominance. glibc retained memory is only ~6 MB—not the main cause.\n>\n> We ran **5-minute** profiling to find **new** allocation paths. It did not return\n> Top stacks—this often happens when the memory is **already resident** and the\n> process had **little new allocation** during the window. That does **not** cancel\n> the snapshot conclusion: native memory is still high; we just lack the exact\n> calling code path.\n>\n> Next options: re-sample profiling during **peak load** (one run, full wait), or\n> enable **NMT** and restart to split native categories—not proof the problem went away.\n\n### Incorrect narrative (avoid)\n\n- Pasting a table of \"JNI/Other / heap gap / glibc\" without explanation\n- Treating `--duration 5` as **5 seconds** and using 180s timeout\n- **Re-running** profiling immediately after timeout\n- Saying \"profiling completed but no data → memory may be fine\"\n- Jumping to only \"enable NMT / check JVM args\" without retaining snapshot conclusion\n\n### Regression checklist (Case A)\n\n- [ ] Explained JNI/Other in plain language\n- [ ] Named dominant magnitude (~3.96 GB native vs ~10 MB heap gap)\n- [ ] Distinguished quiet profiling window vs \"no problem\"\n- [ ] Did not re-run profiling on timeout without user consent\n- [ ] Offered peak re-sample or NMT as structured options\n\n---\n\n## Case B: Snapshot → profiling with Top stacks (happy path)\n\n### Context\n\nUser: Java RSS high; snapshot shows JNI/Other elevated; user agrees to profiling.\n\n### Snapshot envelope (summary)\n\n- `javamem.native_memory` warning; `next_steps` command with `--pid 1234 --duration 5`\n\n### Agent before profiling\n\n> Profiling will run for about **5 minutes** on PID **1234**, plus analysis time.\n> Please expect roughly **8–10 minutes** total. I will run it once and wait.\n\nTool timeout: ≥ 600s (prefer 480–600s minimum for duration 5).\n\n### Profiling envelope (summary)\n\n- `summary`: native allocation hotspots located\n- `findings`: `javamem.profiling` with nativealloc Top stacks and share percentages in `detail`\n\n### Correct narrative\n\n> Snapshot showed high off-heap native usage. Five-minute profiling found the\n> hottest allocation path: **[top frames from detail]**, accounting for **[share]%**\n> of sampled native allocations. Focus remediation on that component or JNI bridge\n> in a change window. No need to repeat the same profiling command.\n\n### Incorrect narrative (avoid)\n\n- Asking user to open a flame graph UI when `detail` already has stacks\n- Running `--duration 5` again after success\n- Ignoring snapshot and only listing stacks without connecting to user's RSS symptom\n\n### Regression checklist (Case B)\n\n- [ ] User warned about ~5 minute wait before command\n- [ ] Adequate tool timeout\n- [ ] Single profiling run\n- [ ] Narrated Top stack from `detail`\n- [ ] No duplicate profiling\n\n---\n\n## Case C: Snapshot only — heap dominant\n\n### Snapshot envelope (summary)\n\n- `javamem.heap` warning: old gen high, GC pressure in `detail`\n- No large JNI/Other in `rss_gap`\n\n### Correct narrative\n\n> Memory pressure is mainly in the **Java object heap**, not off-heap native.\n> [Sizes from detail]. Next step is heap dump or object analysis—not native profiling.\n\n### Incorrect narrative (avoid)\n\n- Recommending `--duration` profiling for JNI when heap dominates\n\n---\n\n## Case D: Client timeout with no envelope\n\n### Context\n\nProfiling command started; tool times out at 180s; no JSON returned.\n\n### Correct narrative\n\n> The profiling command likely needs the full **5-minute** sampling window plus\n> backend time; a **3-minute** client timeout is too short. The job may still be\n> running. I should **not** start a duplicate profiling run automatically. We can\n> wait longer with a proper timeout or check whether the first job completed.\n\n### Incorrect narrative (avoid)\n\n- Immediate identical re-run\n- Declaring profiling unsupported without checking wait time\n\n---\n\n## Acceptance scenarios (P0)\n\nUse these four scenarios to validate Skill + references:\n\n1. **Snapshot only** (Case C-like or A snapshot phase): plain-language dominant region\n2. **Empty Top + high JNI snapshot** (Case A): quiet window; snapshot stands\n3. **Profiling with Top stacks** (Case B): narrate stacks; no repeat\n4. **CLI timeout** (Case D): no auto retry; explain minutes vs seconds\n\nFile v0.0.7:references/java/memory/decision-tree.md\n\n# Javamem Follow-up: Following Backend next_steps\n\n> **Role change**: this document no longer decides follow-ups. Diagnosis\n> pacing — what to collect next, when to converge, and cross-domain pivots —\n> is owned by the backend Java diagnosis agent and delivered through\n> `envelope.agent.status` + `envelope.agent.next_steps[]`. This file only\n> explains how to *follow* those instructions correctly.\n\n## Entry: no pid → discovery first\n\nIf the user requests Java memory diagnosis **without** a pid or pod, the backend\nreturns a **process-discovery envelope** (candidate list), not a snapshot:\n\n- **STOP and present the candidates as a table** with PID / cmdline / RSS and a **pod (`namespace/name`) column** for containerized candidates (fall back to service/cgroup for host procs) so container users can filter — the `pod:` value is already in each finding's `detail`. The discovery `next_steps` are `info` **options, not a run list**.\n- **Do not auto-run any diagnosis — even with a single candidate.** Wait for the user to pick a PID, then run `sysom-osops java analyze --type memory --pid <PID>` (legacy `sysom-osops memory javamem --pid <PID>`).\n- No Java process found → ask the user to confirm the target or supply a pod.\n- **Never invent a PID.** See `javamem-envelope-guide.md` → \"Process-discovery envelope\".\n\nOnce you have a real **snapshot** envelope, follow the rules below.\n\n## Following agent.next_steps (the only rule)\n\n1. **Read `agent.status` first**:\n   - `concluded` (or missing/unrecognized) → answer from `summary`/`findings`;\n     the loop ends.\n   - `in_progress` → the backend requests another collection hop; take the\n     `kind=command` entry from `agent.next_steps[]`.\n2. **Confirm before heavy hops**: profiling / long-running commands\n   (`--duration N`, minutes) must be confirmed with the user first — see\n   `profiling-playbook.md`. Read-only snapshot commands may run directly.\n3. **Run the command exactly as shown** — the gateway has already injected\n   `--session-id`. Never rewrite, add, or remove flags; if the command fails,\n   relay the error envelope as-is.\n4. **The command output is the next hop envelope** → go back to step 1.\n5. **Hard limits**:\n   - Max **4 hops** per session. If the backend still says `in_progress` at\n     the ceiling, stop, present the conclusions so far, and state the\n     evidence limits (the backend force-concludes at the same ceiling).\n   - If the user declines a proposed command, stop the loop, summarize from\n     the evidence already collected, and state explicitly which conclusions\n     remain unconfirmed because that hop was not run.\n\n## What moved where\n\n- Heap / native / glibc branch decisions (profiling vs NMT vs heap dump) →\n  made by the backend memory-domain skill; its rationale arrives in\n  `findings[]` and the requested hop in `next_steps[].command`.\n- Cross-domain pivots (memory ↔ gc ↔ cpu) → declared by the backend via\n  `next_steps[].command`; this skill keeps no pivot rules of its own.\n- Multi-hop evidence chaining → the backend quotes prior-hop evidence in\n  later hops; relay those references verbatim when narrating progress.\n\n## Pair with\n\n- Terms: `glossary.md`\n- Envelope reading: `javamem-envelope-guide.md`\n- Profiling execution & timeouts: `profiling-playbook.md`\n- Examples: `case-library.md`\n\nFile v0.0.7:references/java/memory/glossary.md\n\n# Java Memory Glossary\n\nUse when narrating `memory javamem` findings to non-expert users. Each entry:\n**definition → plain language → common mistake → what to check next**.\n\n## Core memory regions\n\n### Java Heap\n\n- **Definition**: Memory for Java objects, managed by the JVM garbage collector.\n- **Plain language**: \"The part of JVM memory where your Java objects live.\"\n- **Common mistake**: Assuming high process RSS always means heap leak.\n- **Next**: If heap dominates findings (`javamem.heap`), consider heap dump / ATP—not another javamem snapshot.\n\n### Non-heap (Metaspace, CodeCache, DirectBuffer)\n\n- **Definition**: JVM memory outside the object heap but still JVM-accounted.\n- **Plain language**: \"Class metadata, JIT code, and direct buffers—still 'JVM inside' but not regular objects.\"\n- **Common mistake**: Confusing Metaspace growth with native JNI leak.\n- **Next**: Read `javamem.nonheap` detail; Metaspace vs DirectBuffer need different remediation.\n\n### JNI/Other\n\n- **Definition**: Process-resident memory attributed to native/off-heap usage outside classic heap accounting—often JNI libraries, thread stacks, other native buffers.\n- **Plain language**: \"Memory the OS sees in the Java process that is **not** mainly your Java object heap—often native libraries or off-heap buffers.\"\n- **Common mistake**: Calling it \"heap leak\" or \"GC problem.\"\n- **Next**: Profiling (`--duration`, minutes) for **incremental** allocation paths, or NMT after restart for **resident** breakdown.\n\n### RSS vs JVM gap\n\n- **Definition**: Process RSS (OS view) minus JVM-reported used memory.\n- **Plain language**: \"The OS thinks the process uses more RAM than the JVM's own ledger explains—something is outside normal JVM heap/non-heap reporting.\"\n- **Common mistake**: Ignoring the gap and only talking about heap usage percentage.\n- **Next**: Split the gap using `javamem.rss_gap` detail (JNI/Other, heap gap, glibc); follow dominant line item.\n\n### Heap gap\n\n- **Definition**: Difference between OS-seen heap-related RSS and JVM heap `used`.\n- **Plain language**: \"Small differences between how the OS and JVM count heap pages—often normal padding/accounting.\"\n- **Common mistake**: Treating a ~10 MB heap gap as the main cause when JNI/Other is multi-GB.\n- **Next**: Only emphasize if it dominates the RSS−JVM split.\n\n### glibc resident / fragmentation\n\n- **Definition**: Estimated memory retained by the C library allocator (arena cache, fragmentation) not returned to the OS.\n- **Plain language**: \"Leftover pages from the C memory allocator—sometimes tens of MB, rarely the main story when native is multi-GB.\"\n- **Common mistake**: Blaming glibc for a 3+ GB RSS when glibc line is only a few MB.\n- **Next**: If `javamem.glibc_fragmentation` warns, discuss allocator tuning in a change window—not emergency cache drops.\n\n## Diagnostic tools (not the same thing)\n\n### Profiling (`--duration N`)\n\n- **Definition**: Minutes-long sampling of **new** native/heap allocations; returns Top stacks in `javamem.profiling`.\n- **Plain language**: \"Watch **new** memory allocations for N minutes to see **which code paths** are allocating.\"\n- **Common mistake**: Treating `--duration 5` as five **seconds**; expecting Top stacks when memory is already fully resident with no new alloc.\n- **Next**: Read `profiling-playbook.md`.\n\n### NMT (NativeMemoryTracking)\n\n- **Definition**: JVM flag `-XX:NativeMemoryTracking=summary|detail`; breaks down native regions after **restart**.\n- **Plain language**: \"Turn on JVM native memory accounting and restart—helps split thread/GC/compiler/native **already resident**.\"\n- **Common mistake**: Offering NMT as the only fix when profiling was empty but snapshot already shows large JNI/Other.\n- **Next**: Use when you need resident category split, not incremental call paths.\n\n## Finding `category` quick reference\n\n| category | Domain | Agent focus |\n|----------|--------|-------------|\n| `javamem.heap` | Java heap usage / GC pressure | Heap dump, ATP—not repeat javamem |\n| `javamem.nonheap` | Metaspace, DirectBuffer, CodeCache | Class leak vs buffer leak |\n| `javamem.native_memory` | JNI/Other, NMT other | Off-heap native; profiling or NMT |\n| `javamem.rss_gap` | RSS − JVM split | Name dominant contributor in gap |\n| `javamem.glibc_fragmentation` | glibc arena/frag | Allocator tuning; not primary if MB-scale |\n| `javamem.profiling` | Top allocation/leak stacks | Narrate stacks; no flame UI needed |\n| `data_quality` | Missing data or empty profiling | Do not claim \"all clear\"; see playbook |\n| `analyzer_error` | Plugin failure | Say analysis incomplete for that domain |\n\nFile v0.0.7:references/java/memory/javamem-envelope-guide.md\n\n# Javamem Envelope Interpretation Guide\n\nUse after `sysom-osops memory javamem` returns an envelope. Pair with\n`glossary.md` for term definitions.\n\n## Read order\n\n1. `ok` and `error` (if any)\n2. `agent.summary`\n3. `agent.findings[]` — sort by: `root_cause` category if present, then severity, then match to user symptom\n   - **Snapshot findings analysed by java_agent use `root_cause` / `observation` / `config` as `category`.** Pick the primary cause from the `root_cause` entries; `config` entries are configuration risks (usually not the trigger), `observation` entries are supporting context.\n   - Older/local-analyser envelopes instead use `javamem.{domain}` categories (`javamem.heap`, `javamem.rss_gap`, …) and encode strength in `severity` only. In that case pick the primary cause(s) by **`severity` high first + largest magnitude in `detail` + user-symptom match** — there may be more than one.\n4. `agent.next_steps[]` — priority plan, not a checklist\n5. `agent.session_id` (if present) — the multi-hop session handle; see \"Multi-hop sessions\"\n\nRequired entities (PID, sizes, mechanism) must appear in `summary` or\n`findings[].detail`. Do not invent numbers.\n\n## Presenting quantitative evidence\n\nEvery number you show the user must come from `summary` or `findings[].detail` —\nthose are the only fields carrying quantitative evidence. (There is no separate\nstructured `evidence` object in this envelope; do not look for one.)\n\nWhen a `detail` states a metric, keep these three parts together in your answer:\n\n1. **Magnitude with unit** — e.g. \"Metaspace 85 MB\". Convert bytes to MB/GB for\n   readability, but never re-derive a number the envelope did not state.\n2. **The stated ratio basis** — if `detail` says `used/max`, keep that framing.\n   **If `detail` says there is no limit (`max = -1`, e.g. `-XX:MaxMetaspaceSize`\n   not set), do not present any \"usage percent\" for that region** — report the\n   absolute size plus \"no configured ceiling\" instead. `used` sitting close to\n   the committed size is normal JVM behaviour, not a risk signal.\n3. **What is missing** — if `detail` declares a premise as unknown (JDK version,\n   thread count, cgroup limit, growth trend), repeat that caveat. A single\n   snapshot cannot prove growth over time; do not upgrade a \"suspicion\" into a\n   confirmed leak.\n\nPresent numbers inline in prose or a small table; do not dump raw field paths at\nthe user unless they ask where a number came from.\n\n## Multi-hop sessions\n\nJava diagnosis is session based: `agent.session_id` ties follow-up hops to the\nsame backend conversation, so the next hop can reference the previous hop's\nevidence instead of re-collecting it.\n\n- When `agent.session_id` is present, **carry it into the next java command**:\n  append `--session-id <id>` (the backend also injects it into any\n  `next_steps[].command` it generates, and injection is idempotent — never add\n  it twice).\n- This applies to follow-up questions too, not just to the commands listed in\n  `next_steps`: if the user asks a new java-memory question about the same\n  process, reuse the session id so context is preserved.\n- Sessions are capped (4 hops) and expire; if the backend returns a new\n  `session_id`, switch to it. If `session_id` is absent, simply omit the flag.\n- Never invent or hand-edit a session id.\n\n## Snapshot vs profiling envelopes\n\n| Command shape | Backend path | Findings expected |\n|---------------|--------------|-------------------|\n| `javamem` (no `--duration`) | Snapshot: sysak `-g`, analysed by java_agent | `root_cause` / `observation` / `config` (legacy path: `javamem.heap`, `javamem.native_memory`, `javamem.rss_gap`, …) |\n| `javamem --pid P --duration N` | Profiling only: Top-N stacks | `javamem.profiling` and/or `data_quality` |\n\nProfiling envelope does **not** replace snapshot domain findings. If you only\nhave a profiling envelope, recall the **previous snapshot** when narrating.\n\n> Snapshot data is now collected through `sysom-osops collect memory-javamem`\n> (envelope mode). The user-readable entities still live in `summary` /\n> `findings[].detail`; nothing changes in how you read them.\n\n## Process-discovery envelope (no pid / no pod)\n\nWhen the user asks for Java memory diagnosis **without** naming a pid or pod, the\nbackend first runs a lightweight discovery collector (`memory-javaproc`, a pure\n`/proc` scan — no sysak) and returns a **candidate list** instead of a diagnosis:\n\n- `agent.findings[]` are `info` (category `observation`), one per candidate,\n  carrying `PID`, a command-line summary, `RSS`, and — when containerized —\n  `pod` (`namespace/name`); otherwise `service`/`cgroup`.\n- `agent.next_steps[]` are `info` options (one per candidate); the matching\n  `sysom-osops memory javamem --pid <PID>` string is shown in the option's\n  `reason` for the user to pick — it is **not** an auto-runnable command.\n- If no Java process is found, a single `info` / `data_quality` finding says so\n  and there are **no** `next_steps` (ask the user to confirm target or supply a pod).\n\nAgent behavior:\n\n1. **STOP and present the candidate list to the user** as a table that includes\n   PID, command summary, RSS, and a **pod (`namespace/name`) column** for\n   containerized candidates (fall back to service/cgroup for host processes) so\n   container users can filter quickly — the `pod:` value is already in each\n   finding's `detail`. The discovery `next_steps` are `info` options (no\n   auto-runnable `command`) — **choices, not a to-do list**. Do **not** auto-run\n   a diagnosis, **even if there is only one candidate**.\n2. Let the **user pick** the target PID (largest RSS is listed first, but the\n   user may want a specific service/pod). Only after the user chooses, run\n   `sysom-osops memory javamem --pid <PID>` for that PID.\n3. **Never invent a PID.** Only diagnose a PID that appears in the candidate list\n   (or one the user explicitly provides).\n\n## Three-part answer template (snapshot)\n\nFor each user-facing answer after snapshot:\n\n1. **Dominant contributor**: Heap vs off-heap (JNI/Other) vs glibc—use magnitudes in `detail`.\n2. **Scale**: PID and approximate GB/MB from envelope text.\n3. **Evidence gap**: Missing allocation path? Missing NMT split? Point to `next_steps`.\n\n**Do not** output findings as a jargon table without translation.\n\n### Example shape (illustrative)\n\n> Process **3307705** uses most of its RAM in **off-heap native memory (~3.96 GB)**,\n> not the Java object heap (heap accounting gap ~10 MB). The OS sees ~3.97 GB more\n> RSS than the JVM ledger explains; **JNI/Other** is the main part of that gap.\n> glibc retained memory is only ~6 MB—not the primary cause. We still lack the\n> **specific native allocation call path**; profiling during load can fill that gap.\n\n## By `category`\n\n### `javamem.heap`\n\n- Focus: heap utilization, young/old gen, GC hints in `detail`.\n- User message: \"Object heap pressure\" vs \"native\" if both present—state which dominates.\n- Follow-up: heap dump / ATP per `next_steps`; not another snapshot javamem.\n\n### `javamem.nonheap`\n\n- Focus: Metaspace, DirectBuffer, CodeCache in `detail`.\n- Distinguish class/metadata growth vs direct buffer leak suspicion.\n\n### `javamem.native_memory`\n\n- Focus: JNI/Other or NMT-other high usage.\n- Always translate JNI/Other (see glossary).\n- If NMT disabled in `detail`, say native **categories** cannot be split further without restart.\n\n### `javamem.rss_gap`\n\n- Focus: RSS − JVM total and split lines (heap gap, JNI/Other, glibc, NMT other).\n- Identify **largest line item** before recommending actions.\n- Small heap gap + large JNI/Other → narrative centers on native, not heap.\n\n### `javamem.glibc_fragmentation`\n\n- Focus: arena/fragmentation mechanism in `detail`.\n- Only treat as primary if magnitude supports it (usually not when JNI/Other is GB-scale).\n\n### `javamem.profiling` (profiling hop only)\n\n- Focus: Top stacks in `detail` (share % and reversed frame list).\n- Explain what the top frame **means** operationally (e.g. JNI bridge, allocator).\n- Do not request flame UI.\n\n### `data_quality`\n\n- Snapshot: missing fields → \"this dimension unavailable,\" not \"normal.\"\n- Profiling: empty Top stacks → read `profiling-playbook.md`; **preserve snapshot conclusion**.\n\n## Profiling-only envelope with no prior snapshot in thread\n\nIf the user jumped straight to `--duration` without snapshot:\n\n- Answer from profiling findings only for **incremental** path.\n- Note that **resident** breakdown may still need a prior or follow-up snapshot.\n\n## `next_steps` kinds\n\n| kind | Agent action |\n|------|--------------|\n| `command` | Run exact CLI string when user agrees and entity gap remains |\n| `info` | Explain in prose; no automatic extra command |\n\nIn a **process-discovery** envelope the candidate `next_steps` are `info`\noptions (choices for the user to pick). **Never auto-run them**; wait for the\nuser to select a PID, even when only one candidate is listed.\n\nWhen `command` includes `--duration`, read `profiling-playbook.md` **before** executing.\n\nArchive v0.0.6: 22 files, 43285 bytes\n\nFiles: references/classify-output-guide.md (1517b), references/deep-actions.md (4337b), references/java-triage.md (228b), references/java/cpu/cpu-guide.md (3286b), references/java/gc/gc-guide.md (4342b), references/java/memory/case-library.md (5842b), references/java/memory/decision-tree.md (3340b), references/java/memory/glossary.md (4645b), references/java/memory/javamem-envelope-guide.md (9044b), references/java/memory/memory-guide.md (3625b), references/java/memory/profiling-interpretation.md (5829b), references/java/memory/profiling-playbook.md (3432b), references/java/README.md (3515b), references/memory-triage.md (5472b), references/non-memory-triage.md (2153b), references/parameter-guide.md (3120b), references/ram-policies.md (1221b), references/report-interpretation.md (2788b), references/supported-environments.md (1051b), skill-card.md (2671b), SKILL.md (17349b), _meta.json (147b)\n\nFile v0.0.6:SKILL.md\n\n---\nname: alibabacloud-sysom-diagnosis\ndescription: >\n  Use when troubleshooting Linux server performance or stability issues —\n  CPU saturation, high load, scheduling delay, memory pressure, OOM events,\n  high RSS, page cache / shared memory growth, memory cgroup residue, Java\n  heap issues, disk IO saturation or latency, packet loss, network jitter,\n  or a server that is slow, stuck, or unstable. Performs diagnosis and\n  surfaces recommendations; does not apply fixes automatically.\nlicense: Apache-2.0\ncompatibility: >\n  Requires sysom-osops CLI. Remote diagnosis requires Alibaba Cloud credentials\n  through AK/SK or an ECS RAM Role, an online Cloud Assistant on the target ECS,\n  and a supported China Mainland or Hong Kong region.\nmetadata:\n  domain: aiops\n  product: sysom\n  supported_domains:\n    - cpu\n    - io\n    - memory\n    - network\n    - java\n  owner: sysom-team\n  contact: sysom-team@alibaba-inc.com\nallowed-tools: Bash Read\n---\n\n# alibabacloud-sysom-diagnosis\n\nUse SysOM CLI and backend envelopes as the diagnosis source of truth. This Skill\nreplaces the older SysOM diagnosis Skill and is the single entry point for SysOM\nECS performance and stability diagnosis.\n\n## Immediate Route\n\nWhen the user reports a symptom and has not provided fresh SysOM envelope output,\nrun the matching SysOM command from **Domain Routing** below before ad hoc Linux\ninspection or manual probing. Then follow the returned `agent.summary`,\n`agent.findings[].detail/category`, and `agent.next_steps[]`. Raw Linux commands\nare bounded fallbacks only when a SysOM command is unavailable, outputs\ncontradict each other, or a required entity remains missing after the focused\nSysOM command.\n\n## Credential Security\n\nNever print, echo, or ask for AccessKey ID or AccessKey Secret values. Remote\ncommands perform their own authentication checks. If a command returns an\nauthentication or permission error, explain the error and point the user to\n`references/ram-policies.md`; credential setup must happen outside the\nconversation.\n\n## CLI Setup\n\nCheck whether the CLI is available:\n\n```bash\ncommand -v sysom-osops\n```\n\nIf it is missing, install it:\n\n```bash\ncurl -fsSL --connect-timeout 1000 https://sysom-prd-cn-hangzhou.oss-cn-hangzhou.aliyuncs.com/sysom_prd/skill_cli/install.sh | sudo bash\n```\n\nThen verify only the binary:\n\n```bash\ncommand -v sysom-osops\n```\n\n## Core Workflow\n\n1. Classify the user's symptom into one SysOM domain: memory, IO, load/CPU,\n   network, or Java (GC/memory/CPU).\n2. Run the smallest SysOM command that matches that domain. Prefer a local\n   memory classify for unclear memory symptoms; for other domains, use the\n   matching documented remote action.\n3. Read only the default envelope fields: `ok`, `error`, `command`, and\n   `agent`.\n4. **Load domain references before building the answer.** This step is mandatory\n   and must not be skipped even when `agent.findings` and `agent.next_steps`\n   appear complete. Which references to load depends on the domain:\n   - Java (any type: gc/memory/cpu) → read `references/java/README.md` first\n     for symptom routing and parameter validation; then by type:\n     - gc: `references/java/gc/gc-guide.md`\n     - memory: `references/java/memory/memory-guide.md` (then glossary, envelope\n       guide, profiling playbook, decision tree under `references/java/memory/`)\n     - cpu: `references/java/cpu/cpu-guide.md`\n   - Other domains → load the matching reference from the References table below.\n   References add interpretation rules, entity definitions, and answer-shaping\n   guidance that the envelope alone does not convey. Do not infer Java terms,\n   native memory categories, or profiling semantics from raw envelope text.\n5. Relay the hop as visible progress: present `agent.summary` (plus key\n   findings) to the user, interpreted through the reference material loaded in\n   step 4. Keep evidence qualifiers that change interpretation, including\n   currentness, unavailable direct signals, fallback evidence, and remediation\n   preconditions.\n6. Branch on `agent.status`:\n   - `concluded` (or missing/unrecognized) → build the final answer from\n     `agent.summary`, `agent.findings[].detail/category`, and\n     `agent.next_steps[]`, then stop the loop.\n   - `in_progress` → the backend is requesting another collection hop: take\n     the `kind=command` entry from `agent.next_steps[]`, apply the\n     confirmation rules below, run it, and feed the new envelope back into\n     step 4.\n\n**Guided-diagnosis loop hard rules** (Java multi-hop sessions):\n\n- Pace ownership: never skip `agent.next_steps[]` to decide collection on\n  your own, and never run diagnostic commands outside the envelope.\n- Run commands **exactly as shown** — the gateway has already injected\n  `--session-id`; never rewrite, add, or remove flags. If a command fails,\n  relay the error envelope as-is instead of retrying with tweaked parameters.\n- Hop limit: stop the loop after 4 hops in the same session even if the\n  backend still says `in_progress`; present the conclusions so far and state\n  the evidence limits (the backend force-concludes at the same ceiling —\n  double safety).\n- User refusal: if the user declines a proposed command, stop the loop,\n  summarize from the evidence already collected, and state explicitly which\n  conclusions remain unconfirmed because that hop was not run.\n\n**Before executing a profiling or long-running follow-up command** (e.g.,\n`java analyze --type memory --duration N`, any command that injects an agent\ninto the target process, or any command expected to run for multiple minutes):\n- Tell the user what the command does, how long it takes, and what performance\n  impact it may have on the target process (e.g., CPU overhead from sampling,\n  extra memory from the injected agent, potential safepoint pauses).\n- Ask the user whether to proceed. Do not run the command until the user\n  confirms, or until the user has previously given a standing instruction to\n  auto-run follow-ups.\n- Once the command starts, tell the user the expected wait time and keep them\n  informed if the operation is still in progress.\n\nRead-only query commands (e.g. `java analyze --type cpu`) do not apply here —\nsee each domain guide's Execution Model for specifics.\n\nWhen classify returns a command in `agent.next_steps[]` and no root-cause\nfinding already contains enough evidence to answer, run the first command next.\nDo not replace an Agent-visible SysOM next step with manual shell probing. Raw\nLinux checks are bounded fallbacks after the SysOM next step succeeds, fails, or\ntimes out.\n\nUse the documented commands exactly as shown by default. Do not add raw,\ndebug, or backend evidence expansion flags unless the user explicitly asks for\nthat view.\n\nFinal answers should name evidence, root cause, owner/scope, and operational\naction targets. Do not add shell snippets for verification or remediation unless\nthe user explicitly asks for commands. Prefer phrases such as \"review dependency\nand disable or upgrade the leaking component in a change window\" over raw module,\ncgroup, sysctl, cache-drop, or process-kill commands.\nDo not include command-looking inline snippets such as module inspection/removal,\nmemory summary commands, cgroup file writes, cache-drop controls, sysctl changes,\nor process-kill commands as default final-answer steps.\n\nThe `agent` view must be self-contained for diagnosis. Structured evidence is a\nbackend/UI view and must not be treated as the default Agent source for required\nentities.\n\n## Domain Routing\n\n| User symptom | First route |\n|--------------|-------------|\n| Unclear memory issue, OOM, high RSS, file cache, shmem/tmpfs, memory cgroup, socket memory, kernel memory | `sysom-osops memory classify` |\n| Java issue (symptom unclear) | Follow the Symptom Triage rules in `references/java/README.md` — ask the user about the symptom, then route to the matching type |\n| Java GC pause / frequent GC / low GC throughput | `sysom-osops java analyze --type gc` |\n| Java heap / OOM / heap leak / native leak | `sysom-osops java analyze --type memory` — without a pid/pod it returns a candidate list; STOP and wait for the user to choose before retrying |\n| Java CPU hotspot / high thread CPU / flame graph | `sysom-osops java analyze --type cpu --pid <PID>` |\n| Slow disk, high iowait, disk latency, blocked IO | `sysom-osops io iofsstat`, then `io iodiagnose` if the overview points to slow IO |\n| High load, runqueue backlog, task stuck waiting for CPU | `sysom-osops load loadtask` or `load delay` based on the visible symptom |\n| Packet loss, retransmits, network timeout, jitter | `sysom-osops net packetdrop` for loss/drop symptoms; `net netjitter` for latency fluctuation |\n\nFor command parameters, read `references/deep-actions.md` and\n`references/parameter-guide.md`. For OS and region support, read\n`references/supported-environments.md`. These references are Skill material; do\nnot use remote target file tools to open `.claude/skills` paths on the diagnosed\nhost.\nFor Java symptom-to-type routing, consult `references/java/README.md`.\n\n## Memory Routing\n\nMemory follows the same Core Workflow and Follow-up Rules as every domain: start\nfrom `sysom-osops memory classify`, then pick the next action from visible output\nor `agent.next_steps[]`. For choosing among memory deep actions or checking which\nentity is still missing, load `references/memory-triage.md` (parallel to\n`references/non-memory-triage.md` for other domains).\n\nNote: Java-related memory symptoms (OOM in Java process, heap leak, native leak)\nroute to Java domain via `sysom-osops java analyze --type memory`, not through\nmemory classify. See Domain Routing table above.\n\nChoose the next memory action from visible SysOM output. Do not infer a memory mechanism from symptom wording alone.\n\n## Envelope Contract\n\nDefault command output is the Agent contract:\n\n```json\n{\n  \"ok\": true,\n  \"command\": \"sysom-osops memory classify\",\n  \"agent\": {\n    \"status\": \"concluded\",\n    \"session_id\": \"a1b2c3d4e5f6\",\n    \"summary\": \"Concise diagnosis summary.\",\n    \"findings\": [\n      {\n        \"severity\": \"high\",\n        \"title\": \"Short finding title\",\n        \"detail\": \"Root cause, key entities, and evidence summary.\",\n        \"category\": \"root_cause\"\n      }\n    ],\n    \"next_steps\": [\n      {\n        \"kind\": \"command\",\n        \"label\": \"Run focused deep diagnosis\",\n        \"command\": \"sysom-osops memory oom\",\n        \"reason\": \"The missing entity this command can fill.\"\n      }\n    ]\n  }\n}\n```\n\n`agent.findings[]` may contain only `severity`, `title`, `detail`, and\n`category`. Required entities such as PID, cgroup, service, file path, OOM\nvictim, limit/current, residue, holder, or cleanup target must be written in\n`agent.summary` or `agent.findings[].detail`.\n\nField semantics for guided diagnosis sessions:\n\n- `agent.status`: `in_progress` means the backend diagnosis agent requests\n  another collection hop; `concluded` means diagnosis has converged. Treat a\n  missing or unrecognized value as `concluded`. Legacy collector-level states\n  (`success`, `warning`, ...) may still appear on non-Java actions; interpret\n  them as before.\n- `agent.session_id`: backend-generated multi-hop session identifier. Never\n  generate or modify it; the gateway already injects it into `command`\n  strings, so run them verbatim.\n- `agent.next_steps[].kind`: `command` = backend-requested collection command\n  (subject to the confirmation rules above); `info` = user-side suggestion —\n  present it but never auto-run; `warning` = evidence or data-quality caveat.\n\n## Follow-up Rules\n\n- Prefer `category=root_cause`, then highest severity, then the finding that\n  best matches the user's reported symptom.\n- Treat `root_cause` as stop-ready when visible `detail` contains the entities\n  needed to explain the symptom and a safe next action.\n- Treat `agent.next_steps[]` as a priority plan, not a checklist.\n- Run another SysOM command only when it can fill a named missing entity or\n  change remediation.\n- For long-running Java collection commands — `java analyze --type memory\n  --duration N` (profiling; legacy `memory javamem --duration N`) and `java\n  analyze --type gc` in `collect` mode (5–10 min JFR/GC collection) — the wait is\n  minutes-scale: tell the user, size the tool timeout accordingly, and never\n  re-fire the same command on client timeout. See\n  `references/java/memory/profiling-playbook.md` (memory) and\n  `references/java/gc/gc-guide.md` (gc).\n- Preserve visible qualifiers that affect interpretation, such as current versus\n  historical evidence, unavailable direct signals, fallback evidence used to\n  close currentness, and safety preconditions for remediation.\n- When a finding uses fallback evidence because a direct signal is unavailable,\n  state both parts in the final answer. Do not reduce the conclusion to the\n  fallback metric alone.\n- After a focused SysOM command closes a root cause, answer from it. Do not run\n  extra commands to make the report comprehensive, and do not chase earlier\n  classify anomalies or observations unless they share the same entity and\n  expose a named evidence gap.\n- Do not call backend-only collectors or private helper commands directly.\n- Do not re-check a PID, cgroup, file, limit, or event that SysOM already named\n  in `summary` or `detail`.\n- After a SysOM deep command returns `category=root_cause` with the required\n  entities visible, answer from that envelope. Raw Linux checks are only for\n  contradictions, command errors, or a clearly missing entity.\n- In the final answer, do not turn already-closed entities into extra raw Linux\n  verification commands. Express remediation as dependency-aware action targets\n  and change-window plans unless the envelope itself provides an executable safe\n  next step.\n- Avoid executable shell snippets in the final answer. If a command is useful\n  only for post-change verification, name the SysOM check or metric to re-run\n  instead of raw Linux commands.\n- This includes inline command names for module inspection/removal, memory\n  summary commands, cgroup file writes, cache-drop controls, sysctl changes, and\n  process-kill actions; describe the dependency gate and operational action\n  target in prose.\n- Pivot across domains when the current envelope does not explain the reported\n  symptom and another SysOM domain names a stronger root cause.\n- During diagnosis, do not execute remediation commands that change target\n  state, such as killing processes, removing files, changing sysctl values, or\n  writing to cache-drop controls. Present those as recommendations unless the\n  user explicitly asks you to perform the repair.\n- For non-memory findings, keep the same rule: one focused deep command, then\n  answer when the required entities are visible.\n\n## Error Handling\n\n| `error.code` | Action |\n|--------------|--------|\n| `Sysom.TargetRequired` | Ask for instance ID and region, or explain ECS metadata auto-detection requirements |\n| `Sysom.FallbackClassify` | Present the local classify result and continue only if a focused next step is available |\n| `Sysom.PermissionDenied` | Use `references/ram-policies.md` to explain required RAM permissions |\n| `Sysom.AuthenticationFailure` | Ask the user to configure credentials outside this session |\n| `Sysom.InvalidParameter` | Ask the user to correct the instance, region, or command parameter |\n| `Sysom.DiagnosisVersionNotSupported` | Explain that the target instance diagnosis components need an update |\n| `Sysom.DiagnosisJsonParseFailed` | Retry once only when the user still needs the same evidence |\n| `Sysom.PollError` | Retry the same focused action once when the missing evidence is still required |\n\n## References\n\n| Reference | Use when |\n|-----------|----------|\n| `references/classify-output-guide.md` | Reading local memory classify output |\n| `references/memory-triage.md` | Choosing a memory deep action or checking memory entity completeness |\n| `references/non-memory-triage.md` | Routing IO, load/CPU, network diagnosis |\n| `references/deep-actions.md` | Looking up SysOM commands by domain |\n| `references/parameter-guide.md` | Validating command parameters |\n| `references/report-interpretation.md` | Interpreting envelope fields and answer shape |\n| `references/ram-policies.md` | Explaining RAM permissions |\n| `references/supported-environments.md` | Checking OS, architecture, and region support |\n\n### Java Analysis References\n\n| Reference | Use when |\n|-----------|----------|\n| `references/java/README.md` | **Primary Java entry point**: symptom triage, parameter guide, sub-domain index |\n| `references/java/gc/gc-guide.md` | Running or interpreting `--type gc` results |\n| `references/java/cpu/cpu-guide.md` | Running or interpreting `--type cpu` results |\n| `references/java/memory/memory-guide.md` | `--type memory` interpretation and discovery-first flow entry |\n| `references/java/memory/glossary.md` | Java memory terminology |\n| `references/java/memory/javamem-envelope-guide.md` | Interpreting `--type memory` envelope structure |\n| `references/java/memory/profiling-playbook.md` | Preparation and expected behavior before `--duration` collection |\n| `references/java/memory/decision-tree.md` | Following backend `next_steps` in Java multi-hop sessions |\n| `references/java/memory/case-library.md` | Case library and anti-patterns |\n\nFile v0.0.6:references/java/README.md\n\n# Java Application Diagnosis Reference\n\nUnified command for all Java diagnostics:\n\n```\nsysom-osops java analyze --type <gc|memory|cpu>\n```\n\nThis document is the entry point for Java application diagnosis —\ncovering symptom triage, sub-domain guides, parameter quick-reference,\nand combined-diagnosis strategies. Each sub-domain (gc / memory / cpu)\nhas its own folder under `references/java/`.\n\n---\n\n## Symptom Triage\n\n### When the user's intent is unclear\n\nIf the user describes a vague Java problem (e.g., \"is my Java process healthy\",\n\"something seems wrong with my app\") without mentioning specific symptoms from\nthe table below, ask **one** clarifying question:\n\n> What is the primary symptom you are observing?\n> 1. Application response is slow or has intermittent pauses\n> 2. CPU usage is abnormally high\n> 3. Memory keeps growing or OOM has occurred\n\nThen route based on the answer:\n- 1 → `--type gc` (latency-related symptoms are most often GC-induced)\n- 2 → `--type cpu`\n- 3 → `--type memory`\n\n### Symptom → Diagnosis Type Mapping\n\nWhen the symptom is already clear, route directly:\n\n| Symptom | Recommended type | Description |\n|---------|-----------------|-------------|\n| Long GC pause / STW jitter | gc | Analyze GC event time-series and pause distribution |\n| Frequent Full GC / old generation full | gc | Detect memory pressure source and promotion patterns |\n| Low GC throughput / high application pause ratio | gc | Evaluate GC algorithm efficiency and tuning recommendations |\n| OOM / sustained heap growth | memory | Deep diagnosis of heap/non-heap leaks |\n| Native memory leak | memory | Analyze JNI/DirectBuffer usage |\n| High thread CPU usage | cpu | Flame graph to locate hotspot method stacks |\n| Slow application response (non-GC cause) | cpu | Analyze on-CPU time distribution |\n\n---\n\n## Sub-Domain Reference Index\n\n| type | Folder / entry guide | Scope |\n|------|----------------------|-------|\n| gc | [gc/gc-guide.md](gc/gc-guide.md) | GC pauses, throughput, heap trend analysis |\n| memory | [memory/memory-guide.md](memory/memory-guide.md) (+ `glossary`, `javamem-envelope-guide`, `decision-tree`, `profiling-playbook`, `profiling-interpretation`, `case-library` in `memory/`) | Heap/non-heap diagnosis, snapshots, allocation profiling |\n| cpu | [cpu/cpu-guide.md](cpu/cpu-guide.md) | Flame graph, hotspot method stacks, on-CPU analysis |\n\n---\n\n## Parameter Quick-Reference\n\n| Parameter | gc | memory | cpu |\n|-----------|:---:|:------:|:---:|\n| --pid | Optional | Optional | Optional (omit to get candidate list) |\n| --duration | Seconds (default 60) | Minutes (0=snapshot) | N/A |\n| --pod | Optional (container scenarios) | Optional | Not supported |\n\n---\n\n## Combined Diagnosis\n\nWhen a single type cannot locate the root cause, combine multiple types:\n\n1. **GC causing high CPU**: First `--type gc` to confirm GC frequency, then `--type cpu` to check GC thread proportion\n2. **Memory leak causing frequent GC**: First `--type gc` to observe Full GC patterns, then `--type memory` for deeper heap analysis\n3. **Heavy object allocation found in CPU hotspots**: First `--type cpu` to locate allocation hotspots, then `--type memory --duration 5` to track allocation trends\n\n---\n\n## Prerequisites\n\n- `--type cpu` requires the instance to be managed in the Alibaba Cloud Linux console (the system auto-detects and provides guidance)\n\n---\n\nFor memory-specific interpretation, the discovery-first flow, and answer\ndiscipline, see [memory/memory-guide.md](memory/memory-guide.md).\n\nFile v0.0.6:_meta.json\n\n{\n  \"ownerId\": \"kn74p5w8ywv6prh40g0s82gmqh83nw54\",\n  \"slug\": \"alibabacloud-sysom-diagnosis\",\n  \"version\": \"0.0.6\",\n  \"publishedAt\": 1787362884292\n}\n\nFile v0.0.6:references/classify-output-guide.md\n\n# Classify Output Guide\n\nUse `sysom-osops memory classify` as the local entry point for unclear memory\nsymptoms. It returns the same default Agent envelope as remote deep actions.\n\n## What To Read\n\n| Field | Use |\n|-------|-----|\n| `agent.summary` | Overall memory verdict and dominant issue. |\n| `agent.findings[].category` | Prefer `root_cause` when present. |\n| `agent.findings[].detail` | Agent-visible root-cause entities and missing evidence. |\n| `agent.next_steps[]` | Focused remote actions that can fill missing evidence. |\n\nDo not call backend-only collectors from the Skill. Do not depend on legacy or\ninternal fields outside the default Agent envelope.\n\n## Choosing The First Follow-Up\n\n1. If a root-cause finding already contains the required entities, answer from\n   classify.\n2. If the dominant finding lacks a required entity, run the first relevant\n   `kind=command` next step.\n3. If the visible envelope names an ownership, currentness, attribution, or\n   cleanup gap, follow the focused SysOM action that closes that entity gap.\n4. If a later deep action returns concrete holder or owner entities, combine\n   them with classify instead of re-running broad local checks.\n\n## When No Memory Action Is Needed\n\nIf classify returns no memory finding, or all findings are informational and do\nnot match the user's symptom, report that current SysOM memory evidence is\nhealthy or inconclusive. Pivot to IO, load, network, or Java only when the\nvisible envelope or the user's symptom supports that domain.\n\nFile v0.0.6:references/deep-actions.md\n\n# Deep Actions Reference\n\nThis file lists public SysOM commands that this Skill may route to. On an ECS\ninstance, sysom-osops can auto-detect `--instance` and `--region`; add them only\nfor cross-instance diagnosis or when auto-detection fails.\n\n## Memory\n\n| Command | Mode | Use when |\n|---------|------|----------|\n| `sysom-osops memory classify` | Local | First route for unclear memory symptoms, OOM hints, high RSS, cache growth, shmem/tmpfs, memcg residue, or kernel memory suspicion |\n| `sysom-osops memory memgraph` | Remote | Full memory landscape is missing after classify, or the issue is broad kernel/userspace memory composition |\n| `sysom-osops memory memgraph --enable-socket` | Remote | Socket buffer pressure is visible and socket holder, state, PID, cgroup, or service attribution is missing |\n| `sysom-osops memory process` | Remote | A process is the suspected holder but identity, real executable, cgroup, or service attribution is missing |\n| `sysom-osops memory oom` | Remote | OOM killer event or memcg/host OOM evidence is visible |\n| `sysom-osops memory filecache` | Remote | File/page cache is the dominant unresolved entity and file/holder attribution is missing |\n| `sysom-osops memory shmem` | Remote | shmem, tmpfs, memfd, or SysV shared memory holder attribution is missing |\n| `sysom-osops memory memleak` | Remote | Kernel-hidden growth (vmalloc/page-allocator/slab/percpu) is the dominant unresolved entity and the leaking call point/function/module is missing after memgraph closed only to a candidate (`--type slab\\|page\\|vmalloc\\|percpu`, default vmalloc) |\n| `sysom-osops memory javamem` | Remote | **DEPRECATED** — Use `sysom-osops java analyze --type memory` instead (see Java section below) |\n| `sysom-osops memory memcgoffline` | Remote | Cgroup ownership-transition evidence is visible in SysOM output and original ownership, residue/refcount, or cleanup order needs attribution |\n\n## IO\n\n| Command | Mode | Use when |\n|---------|------|----------|\n| `sysom-osops io iofsstat` | Remote | Disk IO overview is needed for high iowait, slow disk, or IO saturation |\n| `sysom-osops io iodiagnose` | Remote | Slow IO root-cause attribution is needed after the overview or when latency is the primary symptom |\n\n## Load and CPU Scheduling\n\n| Command | Mode | Use when |\n|---------|------|----------|\n| `sysom-osops load loadtask` | Remote | Load average, runqueue, or task composition is the primary symptom |\n| `sysom-osops load delay` | Remote | Runnable tasks are not getting CPU time, scheduling delay is reported, or processes appear stuck without IO evidence |\n\n## Network\n\n| Command | Mode | Use when |\n|---------|------|----------|\n| `sysom-osops net packetdrop` | Remote | Packet loss, retransmits, drops, connection resets, or timeout symptoms are primary |\n| `sysom-osops net netjitter` | Remote | Latency fluctuation, jitter, or intermittent connectivity degradation is primary |\n\n## Java\n\nRoute by symptom through the unified `sysom-osops java analyze` entry point. See `references/java/README.md` for details.\n\n| Command | Scenario |\n|---------|----------|\n| `sysom-osops java analyze --type gc` | GC pause analysis, frequent GC, throughput diagnosis |\n| `sysom-osops java analyze --type gc --duration 120` | Extend collection duration (default 60 seconds) |\n| `sysom-osops java analyze --type memory` | Java heap/non-heap memory diagnosis (equivalent to memory javamem) |\n| `sysom-osops java analyze --type memory --pid <PID> --duration 5` | Continuous collection of a specific process for 5 minutes |\n| `sysom-osops java analyze --type cpu --pid <PID>` | CPU hotspot flame graph analysis (pid optional; if omitted, returns the Java process candidate list first) |\n\n> Note: `--type cpu` locates the target by `--pid` only and **does not support `--pod`**; if `--pid` is omitted it returns the candidate list first. `--pid` is also optional for `memory` and `gc`.\n> duration units: gc = seconds (default 60), memory = minutes (0 = snapshot mode).\n\n## Common Notes\n\n- All public commands return the same envelope shape: `ok`, `error`,\n  `command`, and `agent`.\n- Continue from `agent.next_steps[]` only when a required entity is missing or\n  the next action changes remediation.\n- Do not call backend-only collectors from the Skill. Public commands above are\n  the supported interface.\n\nFile v0.0.6:references/java-triage.md\n\n# Java Problem Triage Guide\n\n> This document has been merged into [references/java/README.md](java/README.md).\n> Please refer to the Java Diagnosis Reference for the full symptom triage,\n> parameter guide, and sub-domain index.\n\nFile v0.0.6:references/java/cpu/cpu-guide.md\n\n<!--\n * @Descripttion: \n * @version: \n * @Author: Jietao Xiao\n * @Date: 2026-07-16 12:12:42\n * @LastEditors: Jietao Xiao\n * @LastEditTime: 2026-07-17 11:05:03\n-->\n# CPU Hotspot Analysis Guide\n\n## Applicable Scenarios\n- High thread CPU usage\n- Slow application response (GC already ruled out)\n- Flame graph needed to locate hotspot method stacks\n\n## Command\n\n`sysom-osops java analyze --type cpu --pid <PID>`\n\n| Parameter | Description |\n|-----------|-------------|\n| --pid | Recommended — target Java process PID. If omitted, the call returns a Java process candidate list first |\n\n> If you are unsure of the PID, run `--type cpu` **without**\n> a pid directly — the CPU path performs Java process discovery itself and\n> returns the candidate list. After the\n> user picks a PID, re-run `--type cpu --pid <PID>` for the actual flame graph.\n\n## Execution Model\n\n`--type cpu` is a **read-only query** over profiling data that is **already being\ncollected continuously**. On-CPU profiling starts automatically once the instance\nis managed in the console, and samples are stored server-side (the `prof_on`\ndataset). The command retrieves roughly the **last 5 minutes** of already-collected\nsamples for the target PID and runs LLM analysis on them.\n\nBecause of this:\n- It does **not** launch an on-demand perf run against the target, does **not**\n  inject any agent, and adds **no measurable CPU overhead** to the process.\n- It returns quickly (query + analysis) — there is **no ~60s sampling wait**.\n- There is **no `--duration`** for cpu (see parameter table), and it does **not**\n  need the pre-profiling overhead confirmation from the skill's general profiling\n  guidance.\n\nWhen confirming this action with the user, describe it as \"view the last ~5\nminutes of continuously-collected CPU flame graph data\". Do **not** say it\nperforms ~60s perf sampling or imposes CPU load on the target — that describes\non-demand profiling (e.g. `--type memory --duration N`), not this path.\n\n## Prerequisites\n\n- **The instance must be managed in the Alibaba Cloud Linux console** (the system auto-detects; returns guidance when not managed)\n\n## Result Interpretation\n\nReturns flame graph LLM analysis conclusions:\n- Hotspot method stacks and call chains\n- On-CPU time proportion per method\n- Potential performance bottleneck identification\n\n### Common Patterns\n\n| Pattern | Symptoms | Recommendation |\n|---------|----------|----------------|\n| Spin lock contention | Significant time spent on lock/CAS operations | Check concurrent data structure usage |\n| Frequent object allocation | Hotspots concentrated on new/allocate paths | Combine with `--type memory` to analyze allocation trends |\n| System call blocking | Hotspots on read/write/poll system calls | Check IO or network blocking |\n| Serialization/deserialization | High proportion of JSON/Protobuf processing | Evaluate serialization strategy or data volume |\n\n## Cross-Domain Diagnosis\n\n- **Heavy object allocation found in CPU hotspots** → Combine with `--type memory --duration 5` to track allocation trends\n- **Suspected GC threads consuming CPU** → First `--type gc` to confirm GC frequency and pause duration\n- **High CPU but no obvious flame graph hotspot** → Likely an off-CPU issue (waiting on IO/locks); check IO domain\n\nFile v0.0.6:references/java/gc/gc-guide.md\n\n# Java GC Diagnosis Guide\n\n## Diagnosis Mode Selection\n\n### Log Analysis Mode (gclog_only)\n\n**Applicable scenarios**: Java process has GC logging configured (JDK8: `-XX:+PrintGCDetails -Xloggc:<path>`; JDK11+: `-Xlog:gc*:file=<path>`)\n\n**Characteristics**:\n- Returns results immediately (no collection wait)\n- Analyzes historical GC logs for retrospective troubleshooting\n- No additional performance overhead\n\n**Limitations**: Requires an existing GC log file; unavailable if JDK8 has no logging configured\n\n**Parameters**: `diagMode=gclog_only`, duration parameter is ignored\n\n### Incremental Collection Mode (collect)\n\n**Applicable scenarios**: No GC logging configured, or precise JFR time-series data needed, or JDK11+ environment\n\n**Characteristics**:\n- Collects JFR + GC logs + OS metrics (5-10 minutes)\n- Does not require pre-configured GC logging (JDK11+ can enable dynamically)\n- Provides complete time-series, heap trends, and OS correlation analysis\n\n**Limitations**: Requires waiting for the duration period; JDK8 cannot enable JFR dynamically and will fall back to GC log collection\n\n**Parameters**: `diagMode=collect`, `duration=5` (default; recommended 5-10 minutes)\n\n## Mode Selection Decision Tree\n\n1. User explicitly mentions \"GC logs\" or \"logging is enabled\" → `gclog_only`\n2. User describes performance issues without mentioning logs → Ask whether GC logging is enabled\n3. User is unsure or has not configured logging → `collect` (more comprehensive, no prerequisites)\n4. JDK8 environment + no logging configured → `collect` (automatically falls back to log collection mode)\n\n## Agent Reply Template\n\nWhen the user has not specified a mode, ask:\n\n> GC diagnosis supports two modes:\n> 1. **Log analysis**: directly analyze existing GC logs and return results immediately. Suitable when the Java process is already configured with `-Xlog:gc*` or `-XX:+PrintGCDetails`.\n> 2. **Incremental collection**: collect 5-10 minutes of JFR + GC data in real time for deep analysis. No prior GC log configuration required, but you must wait for collection to complete.\n>\n> Is GC logging already enabled on your Java process?\n\n## Command\n\n`sysom-osops java analyze --type gc [--diagMode <gclog_only|collect>] [--duration <minutes>] [--pid <PID>]`\n\n| Parameter | Description |\n|-----------|-------------|\n| --diagMode | Diagnosis mode: gclog_only / collect, default collect |\n| --duration | Collection duration (minutes), default 5 for collect mode, ignored for gclog_only |\n| --pid | Optional; auto-detects Java process when not specified |\n| --pod | Optional; specifies pod name in container scenarios |\n\n## Collection execution discipline (collect mode)\n\n`collect` mode samples JFR + GC logs for **5–10 minutes** (minutes-scale, like\nmemory profiling):\n\n- Tell the user the expected wait before starting; they can continue other work.\n- Size the shell/tool timeout to at least **`(duration + 3)` minutes**; do not\n  use 120–180 second timeouts for a 5-minute collection.\n- Launch **once**. On client timeout with no envelope, do **not** auto-re-fire\n  the same command; say it may still be running and ask whether to wait or check\n  task status.\n\n## Result Interpretation\n\nThe envelope contains GC diagnosis analysis results:\n- GC event time-series and pause distribution\n- Throughput assessment and optimization recommendations\n- Heap memory usage trends\n\n### Common Patterns\n\n| Pattern | Symptoms | Recommendation |\n|---------|----------|----------------|\n| Frequent Full GC | Old generation full, promotion rate too high | Combine with `--type memory` for deeper heap analysis |\n| Young GC jitter | Object allocation rate fluctuations causing unstable STW frequency | Check application-layer caching or batch processing logic |\n| Mixed GC reclamation lag | G1 Mixed GC frequency cannot keep up with object promotion | Adjust InitiatingHeapOccupancyPercent |\n| Excessive GC overhead | Application pause ratio > 10% | Evaluate GC algorithm selection and heap size configuration |\n\n\n## Cross-Domain Diagnosis\n\n- **GC causing high CPU** → Combine with `--type cpu` to analyze GC thread stack proportion\n- **Root cause of frequent Full GC** → Combine with `--type memory` for heap analysis to locate leak source\n- **Memory not released after GC** → Combine with `--type memory --duration 5` to track allocation trends\n\nFile v0.0.6:references/java/memory/case-library.md\n\n# Java Memory Case Library\n\nStructured examples for Agent narrative regression. Envelope fields are\nsummarized; numbers are illustrative of real patterns.\n\n## Case template\n\nEach case includes:\n\n- **Context**: user symptom\n- **Envelope summary**: key summary / findings / next_steps\n- **Correct narrative**: what to tell a non-expert user\n- **Incorrect narrative**: patterns to avoid\n\n---\n\n## Case A: High off-heap native, empty profiling (anti-pattern replay)\n\n### Context\n\nUser: \"Java process memory high on instance i-xxx.\"\n\nAgent ran snapshot javamem, then user agreed to profiling.\n\n### Snapshot envelope (summary)\n\n- `summary`: \"Java process 3307705 has memory issues\"\n- `findings`:\n  - `javamem.native_memory`: JNI/Other usage high; NMT not enabled\n  - `javamem.rss_gap`: RSS−JVM gap ~3.97 GB; split: heap gap ~10 MB, JNI/Other ~3.96 GB, glibc ~6 MB\n- `next_steps`:\n  - `command`: `sysom-osops memory javamem --pid 3307705 --duration 5`\n  - reason: capture native allocation path\n\n### Profiling envelope (summary)\n\n- First attempt: **client timeout 180s** (no envelope)\n- Second attempt: `ok: true`, `summary`: profiling did not get usable Top stacks\n- `findings`: `data_quality` — no alloc/leak Top stacks parsed\n- Optional: `job_id` present in backend evidence\n\n### Correct narrative\n\n> The Java process **3307705** is using about **3.96 GB** in **off-heap native\n> memory (JNI/Other)**, not in the Java object heap (only ~10 MB heap accounting\n> gap). The OS sees roughly **3.97 GB** more RSS than the JVM reports; that matches\n> native/off-heap dominance. glibc retained memory is only ~6 MB—not the main cause.\n>\n> We ran **5-minute** profiling to find **new** allocation paths. It did not return\n> Top stacks—this often happens when the memory is **already resident** and the\n> process had **little new allocation** during the window. That does **not** cancel\n> the snapshot conclusion: native memory is still high; we just lack the exact\n> calling code path.\n>\n> Next options: re-sample profiling during **peak load** (one run, full wait), or\n> enable **NMT** and restart to split native categories—not proof the problem went away.\n\n### Incorrect narrative (avoid)\n\n- Pasting a table of \"JNI/Other / heap gap / glibc\" without explanation\n- Treating `--duration 5` as **5 seconds** and using 180s timeout\n- **Re-running** profiling immediately after timeout\n- Saying \"profiling completed but no data → memory may be fine\"\n- Jumping to only \"enable NMT / check JVM args\" without retaining snapshot conclusion\n\n### Regression checklist (Case A)\n\n- [ ] Explained JNI/Other in plain language\n- [ ] Named dominant magnitude (~3.96 GB native vs ~10 MB heap gap)\n- [ ] Distinguished quiet profiling window vs \"no problem\"\n- [ ] Did not re-run profiling on timeout without user consent\n- [ ] Offered peak re-sample or NMT as structured options\n\n---\n\n## Case B: Snapshot → profiling with Top stacks (happy path)\n\n### Context\n\nUser: Java RSS high; snapshot shows JNI/Other elevated; user agrees to profiling.\n\n### Snapshot envelope (summary)\n\n- `javamem.native_memory` warning; `next_steps` command with `--pid 1234 --duration 5`\n\n### Agent before profiling\n\n> Profiling will run for about **5 minutes** on PID **1234**, plus analysis time.\n> Please expect roughly **8–10 minutes** total. I will run it once and wait.\n\nTool timeout: ≥ 600s (prefer 480–600s minimum for duration 5).\n\n### Profiling envelope (summary)\n\n- `summary`: native allocation hotspots located\n- `findings`: `javamem.profiling` with nativealloc Top stacks and share percentages in `detail`\n\n### Correct narrative\n\n> Snapshot showed high off-heap native usage. Five-minute profiling found the\n> hottest allocation path: **[top frames from detail]**, accounting for **[share]%**\n> of sampled native allocations. Focus remediation on that component or JNI bridge\n> in a change window. No need to repeat the same profiling command.\n\n### Incorrect narrative (avoid)\n\n- Asking user to open a flame graph UI when `detail` already has stacks\n- Running `--duration 5` again after success\n- Ignoring snapshot and only listing stacks without connecting to user's RSS symptom\n\n### Regression checklist (Case B)\n\n- [ ] User warned about ~5 minute wait before command\n- [ ] Adequate tool timeout\n- [ ] Single profiling run\n- [ ] Narrated Top stack from `detail`\n- [ ] No duplicate profiling\n\n---\n\n## Case C: Snapshot only — heap dominant\n\n### Snapshot envelope (summary)\n\n- `javamem.heap` warning: old gen high, GC pressure in `detail`\n- No large JNI/Other in `rss_gap`\n\n### Correct narrative\n\n> Memory pressure is mainly in the **Java object heap**, not off-heap native.\n> [Sizes from detail]. Next step is heap dump or object analysis—not native profiling.\n\n### Incorrect narrative (avoid)\n\n- Recommending `--duration` profiling for JNI when heap dominates\n\n---\n\n## Case D: Client timeout with no envelope\n\n### Context\n\nProfiling command started; tool times out at 180s; no JSON returned.\n\n### Correct narrative\n\n> The profiling command likely needs the full **5-minute** sampling window plus\n> backend time; a **3-minute** client timeout is too short. The job may still be\n> running. I should **not** start a duplicate profiling run automatically. We can\n> wait longer with a proper timeout or check whether the first job completed.\n\n### Incorrect narrative (avoid)\n\n- Immediate identical re-run\n- Declaring profiling unsupported without checking wait time\n\n---\n\n## Acceptance scenarios (P0)\n\nUse these four scenarios to validate Skill + references:\n\n1. **Snapshot only** (Case C-like or A snapshot phase): plain-language dominant region\n2. **Empty Top + high JNI snapshot** (Case A): quiet window; snapshot stands\n3. **Profiling with Top stacks** (Case B): narrate stacks; no repeat\n4. **CLI timeout** (Case D): no auto retry; explain minutes vs seconds\n\nFile v0.0.6:references/java/memory/decision-tree.md\n\n# Javamem Follow-up: Following Backend next_steps\n\n> **Role change**: this document no longer decides follow-ups. Diagnosis\n> pacing — what to collect next, when to converge, and cross-domain pivots —\n> is owned by the backend Java diagnosis agent and delivered through\n> `envelope.agent.status` + `envelope.agent.next_steps[]`. This file only\n> explains how to *follow* those instructions correctly.\n\n## Entry: no pid → discovery first\n\nIf the user requests Java memory diagnosis **without** a pid or pod, the backend\nreturns a **process-discovery envelope** (candidate list), not a snapshot:\n\n- **STOP and present the candidates as a table** with PID / cmdline / RSS and a **pod (`namespace/name`) column** for containerized candidates (fall back to service/cgroup for host procs) so container users can filter — the `pod:` value is already in each finding's `detail`. The discovery `next_steps` are `info` **options, not a run list**.\n- **Do not auto-run any diagnosis — even with a single candidate.** Wait for the user to pick a PID, then run `sysom-osops java analyze --type memory --pid <PID>` (legacy `sysom-osops memory javamem --pid <PID>`).\n- No Java process found → ask the user to confirm the target or supply a pod.\n- **Never invent a PID.** See `javamem-envelope-guide.md` → \"Process-discovery envelope\".\n\nOnce you have a real **snapshot** envelope, follow the rules below.\n\n## Following agent.next_steps (the only rule)\n\n1. **Read `agent.status` first**:\n   - `concluded` (or missing/unrecognized) → answer from `summary`/`findings`;\n     the loop ends.\n   - `in_progress` → the backend requests another collection hop; take the\n     `kind=command` entry from `agent.next_steps[]`.\n2. **Confirm before heavy hops**: profiling / long-running commands\n   (`--duration N`, minutes) must be confirmed with the user first — see\n   `profiling-playbook.md`. Read-only snapshot commands may run directly.\n3. **Run the command exactly as shown** — the gateway has already injected\n   `--session-id`. Never rewrite, add, or remove flags; if the command fails,\n   relay the error envelope as-is.\n4. **The command output is the next hop envelope** → go back to step 1.\n5. **Hard limits**:\n   - Max **4 hops** per session. If the backend still says `in_progress` at\n     the ceiling, stop, present the conclusions so far, and state the\n     evidence limits (the backend force-concludes at the same ceiling).\n   - If the user declines a proposed command, stop the loop, summarize from\n     the evidence already collected, and state explicitly which conclusions\n     remain unconfirmed because that hop was not run.\n\n## What moved where\n\n- Heap / native / glibc branch decisions (profiling vs NMT vs heap dump) →\n  made by the backend memory-domain skill; its rationale arrives in\n  `findings[]` and the requested hop in `next_steps[].command`.\n- Cross-domain pivots (memory ↔ gc ↔ cpu) → declared by the backend via\n  `next_steps[].command`; this skill keeps no pivot rules of its own.\n- Multi-hop evidence chaining → the backend quotes prior-hop evidence in\n  later hops; relay those references verbatim when narrating progress.\n\n## Pair with\n\n- Terms: `glossary.md`\n- Envelope reading: `javamem-envelope-guide.md`\n- Profiling execution & timeouts: `profiling-playbook.md`\n- Examples: `case-library.md`\n\nFile v0.0.6:references/java/memory/glossary.md\n\n# Java Memory Glossary\n\nUse when narrating `memory javamem` findings to non-expert users. Each entry:\n**definition → plain language → common mistake → what to check next**.\n\n## Core memory regions\n\n### Java Heap\n\n- **Definition**: Memory for Java objects, managed by the JVM garbage collector.\n- **Plain language**: \"The part of JVM memory where your Java objects live.\"\n- **Common mistake**: Assuming high process RSS always means heap leak.\n- **Next**: If heap dominates findings (`javamem.heap`), consider heap dump / ATP—not another javamem snapshot.\n\n### Non-heap (Metaspace, CodeCache, DirectBuffer)\n\n- **Definition**: JVM memory outside the object heap but still JVM-accounted.\n- **Plain language**: \"Class metadata, JIT code, and direct buffers—still 'JVM inside' but not regular objects.\"\n- **Common mistake**: Confusing Metaspace growth with native JNI leak.\n- **Next**: Read `javamem.nonheap` detail; Metaspace vs DirectBuffer need different remediation.\n\n### JNI/Other\n\n- **Definition**: Process-resident memory attributed to native/off-heap usage outside classic heap accounting—often JNI libraries, thread stacks, other native buffers.\n- **Plain language**: \"Memory the OS sees in the Java process that is **not** mainly your Java object heap—often native libraries or off-heap buffers.\"\n- **Common mistake**: Calling it \"heap leak\" or \"GC problem.\"\n- **Next**: Profiling (`--duration`, minutes) for **incremental** allocation paths, or NMT after restart for **resident** breakdown.\n\n### RSS vs JVM gap\n\n- **Definition**: Process RSS (OS view) minus JVM-reported used memory.\n- **Plain language**: \"The OS thinks the process uses more RAM than the JVM's own ledger explains—something is outside normal JVM heap/non-heap reporting.\"\n- **Common mistake**: Ignoring the gap and only talking about heap usage percentage.\n- **Next**: Split the gap using `javamem.rss_gap` detail (JNI/Other, heap gap, glibc); follow dominant line item.\n\n### Heap gap\n\n- **Definition**: Difference between OS-seen heap-related RSS and JVM heap `used`.\n- **Plain language**: \"Small differences between how the OS and JVM count heap pages—often normal padding/accounting.\"\n- **Common mistake**: Treating a ~10 MB heap gap as the main cause when JNI/Other is multi-GB.\n- **Next**: Only emphasize if it dominates the RSS−JVM split.\n\n### glibc resident / fragmentation\n\n- **Definition**: Estimated memory retained by the C library allocator (arena cache, fragmentation) not returned to the OS.\n- **Plain language**: \"Leftover pages from the C memory allocator—sometimes tens of MB, rarely the main story when native is multi-GB.\"\n- **Common mistake**: Blaming glibc for a 3+ GB RSS when glibc line is only a few MB.\n- **Next**: If `javamem.glibc_fragmentation` warns, discuss allocator tuning in a change window—not emergency cache drops.\n\n## Diagnostic tools (not the same thing)\n\n### Profiling (`--duration N`)\n\n- **Definition**: Minutes-long sampling of **new** native/heap allocations; returns Top stacks in `javamem.profiling`.\n- **Plain language**: \"Watch **new** memory allocations for N minutes to see **which code paths** are allocating.\"\n- **Common mistake**: Treating `--duration 5` as five **seconds**; expecting Top stacks when memory is already fully resident with no new alloc.\n- **Next**: Read `profiling-playbook.md`.\n\n### NMT (NativeMemoryTracking)\n\n- **Definition**: JVM flag `-XX:NativeMemoryTracking=summary|detail`; breaks down native regions after **restart**.\n- **Plain language**: \"Turn on JVM native memory accounting and restart—helps split thread/GC/compiler/native **already resident**.\"\n- **Common mistake**: Offering NMT as the only fix when profiling was empty but snapshot already shows large JNI/Other.\n- **Next**: Use when you need resident category split, not incremental call paths.\n\n## Finding `category` quick reference\n\n| category | Domain | Agent focus |\n|----------|--------|-------------|\n| `javamem.heap` | Java heap usage / GC pressure | Heap dump, ATP—not repeat javamem |\n| `javamem.nonheap` | Metaspace, DirectBuffer, CodeCache | Class leak vs buffer leak |\n| `javamem.native_memory` | JNI/Other, NMT other | Off-heap native; profiling or NMT |\n| `javamem.rss_gap` | RSS − JVM split | Name dominant contributor in gap |\n| `javamem.glibc_fragmentation` | glibc arena/frag | Allocator tuning; not primary if MB-scale |\n| `javamem.profiling` | Top allocation/leak stacks | Narrate stacks; no flame UI needed |\n| `data_quality` | Missing data or empty profiling | Do not claim \"all clear\"; see playbook |\n| `analyzer_error` | Plugin failure | Say analysis incomplete for that domain |\n\nFile v0.0.6:references/java/memory/javamem-envelope-guide.md\n\n# Javamem Envelope Interpretation Guide\n\nUse after `sysom-osops memory javamem` returns an envelope. Pair with\n`glossary.md` for term definitions.\n\n## Read order\n\n1. `ok` and `error` (if any)\n2. `agent.summary`\n3. `agent.findings[]` — sort by: `root_cause` category if present, then severity, then match to user symptom\n   - **Snapshot findings analysed by java_agent use `root_cause` / `observation` / `config` as `category`.** Pick the primary cause from the `root_cause` entries; `config` entries are configuration risks (usually not the trigger), `observation` entries are supporting context.\n   - Older/local-analyser envelopes instead use `javamem.{domain}` categories (`javamem.heap`, `javamem.rss_gap`, …) and encode strength in `severity` only. In that case pick the primary cause(s) by **`severity` high first + largest magnitude in `detail` + user-symptom match** — there may be more than one.\n4. `agent.next_steps[]` — priority plan, not a checklist\n5. `agent.session_id` (if present) — the multi-hop session handle; see \"Multi-hop sessions\"\n\nRequired entities (PID, sizes, mechanism) must appear in `summary` or\n`findings[].detail`. Do not invent numbers.\n\n## Presenting quantitative evidence\n\nEvery number you show the user must come from `summary` or `findings[].detail` —\nthose are the only fields carrying quantitative evidence. (There is no separate\nstructured `evidence` object in this envelope; do not look for one.)\n\nWhen a `detail` states a metric, keep these three parts together in your answer:\n\n1. **Magnitude with unit** — e.g. \"Metaspace 85 MB\". Convert bytes to MB/GB for\n   readability, but never re-derive a number the envelope did not state.\n2. **The stated ratio basis** — if `detail` says `used/max`, keep that framing.\n   **If `detail` says there is no limit (`max = -1`, e.g. `-XX:MaxMetaspaceSize`\n   not set), do not present any \"usage percent\" for that region** — report the\n   absolute size plus \"no configured ceiling\" instead. `used` sitting close to\n   the committed size is normal JVM behaviour, not a risk signal.\n3. **What is missing** — if `detail` declares a premise as unknown (JDK version,\n   thread count, cgroup limit, growth trend), repeat that caveat. A single\n   snapshot cannot prove growth over time; do not upgrade a \"suspicion\" into a\n   confirmed leak.\n\nPresent numbers inline in prose or a small table; do not dump raw field paths at\nthe user unless they ask where a number came from.\n\n## Multi-hop sessions\n\nJava diagnosis is session based: `agent.session_id` ties follow-up hops to the\nsame backend conversation, so the next hop can reference the previous hop's\nevidence instead of re-collecting it.\n\n- When `agent.session_id` is present, **carry it into the next java command**:\n  append `--session-id <id>` (the backend also injects it into any\n  `next_steps[].command` it generates, and injection is idempotent — never add\n  it twice).\n- This applies to follow-up questions too, not just to the commands listed in\n  `next_steps`: if the user asks a new java-memory question about the same\n  process, reuse the session id so context is preserved.\n- Sessions are capped (4 hops) and expire; if the backend returns a new\n  `session_id`, switch to it. If `session_id` is absent, simply omit the flag.\n- Never invent or hand-edit a session id.\n\n## Snapshot vs profiling envelopes\n\n| Command shape | Backend path | Findings expected |\n|---------------|--------------|-------------------|\n| `javamem` (no `--duration`) | Snapshot: sysak `-g`, analysed by java_agent | `root_cause` / `observation` / `config` (legacy path: `javamem.heap`, `javamem.native_memory`, `javamem.rss_gap`, …) |\n| `javamem --pid P --duration N` | Profiling only: Top-N stacks | `javamem.profiling` and/or `data_quality` |\n\nProfiling envelope does **not** replace snapshot domain findings. If you only\nhave a profiling envelope, recall the **previous snapshot** when narrating.\n\n> Snapshot data is now collected through `sysom-osops collect memory-javamem`\n> (envelope mode). The user-readable entities still live in `summary` /\n> `findings[].detail`; nothing changes in how you read them.\n\n## Process-discovery envelope (no pid / no pod)\n\nWhen the user asks for Java memory diagnosis **without** naming a pid or pod, the\nbackend first runs a lightweight discovery collector (`memory-javaproc`, a pure\n`/proc` scan — no sysak) and returns a **candidate list** instead of a diagnosis:\n\n- `agent.findings[]` are `info` (category `observation`), one per candidate,\n  carrying `PID`, a command-line summary, `RSS`, and — when containerized —\n  `pod` (`namespace/name`); otherwise `service`/`cgroup`.\n- `agent.next_steps[]` are `info` options (one per candidate); the matching\n  `sysom-osops memory javamem --pid <PID>` string is shown in the option's\n  `reason` for the user to pick — it is **not** an auto-runnable command.\n- If no Java process is found, a single `info` / `data_quality` finding says so\n  and there are **no** `next_steps` (ask the user to confirm target or supply a pod).\n\nAgent behavior:\n\n1. **STOP and present the candidate list to the user** as a table that includes\n   PID, command summary, RSS, and a **pod (`namespace/name`) column** for\n   containerized candidates (fall back to service/cgroup for host processes) so\n   container users can filter quickly — the `pod:` value is already in each\n   finding's `detail`. The discovery `next_steps` are `info` options (no\n   auto-runnable `command`) — **choices, not a to-do list**. Do **not** auto-run\n   a diagnosis, **even if there is only one candidate**.\n2. Let the **user pick** the target PID (largest RSS is listed first, but the\n   user may want a specific service/pod). Only after the user chooses, run\n   `sysom-osops memory javamem --pid <PID>` for that PID.\n3. **Never invent a PID.** Only diagnose a PID that appears in the candidate list\n   (or one the user explicitly provides).\n\n## Three-part answer template (snapshot)\n\nFor each user-facing answer after snapshot:\n\n1. **Dominant contributor**: Heap vs off-heap (JNI/Other) vs glibc—use magnitudes in `detail`.\n2. **Scale**: PID and approximate GB/MB from envelope text.\n3. **Evidence gap**: Missing allocation path? Missing NMT split? Point to `next_steps`.\n\n**Do not** output findings as a jargon table without translation.\n\n### Example shape (illustrative)\n\n> Process **3307705** uses most of its RAM in **off-heap native memory (~3.96 GB)**,\n> not the Java object heap (heap accounting gap ~10 MB). The OS sees ~3.97 GB more\n> RSS than the JVM ledger explains; **JNI/Other** is the main part of that gap.\n> glibc retained memory is only ~6 MB—not the primary cause. We still lack the\n> **specific native allocation call path**; profiling during load can fill that gap.\n\n## By `category`\n\n### `javamem.heap`\n\n- Focus: heap utilization, young/old gen, GC hints in `detail`.\n- User message: \"Object heap pressure\" vs \"native\" if both present—state which dominates.\n- Follow-up: heap dump / ATP per `next_steps`; not another snapshot javamem.\n\n### `javamem.nonheap`\n\n- Focus: Metaspace, DirectBuffer, CodeCache in `detail`.\n- Distinguish class/metadata growth vs direct buffer leak suspicion.\n\n### `javamem.native_memory`\n\n- Focus: JNI/Other or NMT-other high usage.\n- Always translate JNI/Other (see glossary).\n- If NMT disabled in `detail`, say native **categories** cannot be split further without restart.\n\n### `javamem.rss_gap`\n\n- Focus: RSS − JVM total and split lines (heap gap, JNI/Other, glibc, NMT other).\n- Identify **largest line item** before recommending actions.\n- Small heap gap + large JNI/Other → narrative centers on native, not heap.\n\n### `javamem.glibc_fragmentation`\n\n- Focus: arena/fragmentation mechanism in `detail`.\n- Only treat as primary if magnitude supports it (usually not when JNI/Other is GB-scale).\n\n### `javamem.profiling` (profiling hop only)\n\n- Focus: Top stacks in `detail` (share % and reversed frame list).\n- Explain what the top frame **means** operationally (e.g. JNI bridge, allocator).\n- Do not request flame UI.\n\n### `data_quality`\n\n- Snapshot: missing fields → \"this dimension unavailable,\" not \"normal.\"\n- Profiling: empty Top stacks → read `profiling-playbook.md`; **preserve snapshot conclusion**.\n\n## Profiling-only envelope with no prior snapshot in thread\n\nIf the user jumped straight to `--duration` without snapshot:\n\n- Answer from profiling findings only for **incremental** path.\n- Note that **resident** breakdown may still need a prior or follow-up snapshot.\n\n## `next_steps` kinds\n\n| kind | Agent action |\n|------|--------------|\n| `command` | Run exact CLI string when user agrees and entity gap remains |\n| `info` | Explain in prose; no automatic extra command |\n\nIn a **process-discovery** envelope the candidate `next_steps` are `info`\noptions (choices for the user to pick). **Never auto-run them**; wait for the\nuser to select a PID, even when only one candidate is listed.\n\nWhen `command` includes `--duration`, read `profiling-playbook.md` **before** executing.\n\nArchive v0.0.5: 22 files, 43492 bytes\n\nFiles: references/classify-output-guide.md (1517b), references/deep-actions.md (4337b), references/java-triage.md (228b), references/java/cpu/cpu-guide.md (3286b), references/java/gc/gc-guide.md (4342b), references/java/memory/case-library.md (5842b), references/java/memory/decision-tree.md (3340b), references/java/memory/glossary.md (4645b), references/java/memory/javamem-envelope-guide.md (9044b), references/java/memory/memory-guide.md (3625b), references/java/memory/profiling-interpretation.md (5829b), references/java/memory/profiling-playbook.md (3432b), references/java/README.md (3515b), references/memory-triage.md (5472b), references/non-memory-triage.md (2153b), references/parameter-guide.md (3120b), references/ram-policies.md (1221b), references/report-interpretation.md (2788b), references/supported-environments.md (1051b), skill-card.md (3296b), SKILL.md (17349b), _meta.json (147b)\n\nFile v0.0.5:SKILL.md\n\n---\nname: alibabacloud-sysom-diagnosis\ndescription: >\n  Use when troubleshooting Linux server performance or stability issues —\n  CPU saturation, high load, scheduling delay, memory pressure, OOM events,\n  high RSS, page cache / shared memory growth, memory cgroup residue, Java\n  heap issues, disk IO saturation or latency, packet loss, network jitter,\n  or a server that is slow, stuck, or unstable. Performs diagnosis and\n  surfaces recommendations; does not apply fixes automatically.\nlicense: Apache-2.0\ncompatibility: >\n  Requires sysom-osops CLI. Remote diagnosis requires Alibaba Cloud credentials\n  through AK/SK or an ECS RAM Role, an online Cloud Assistant on the target ECS,\n  and a supported China Mainland or Hong Kong region.\nmetadata:\n  domain: aiops\n  product: sysom\n  supported_domains:\n    - cpu\n    - io\n    - memory\n    - network\n    - java\n  owner: sysom-team\n  contact: sysom-team@alibaba-inc.com\nallowed-tools: Bash Read\n---\n\n# alibabacloud-sysom-diagnosis\n\nUse SysOM CLI and backend envelopes as the diagnosis source of truth. This Skill\nreplaces the older SysOM diagnosis Skill and is the single entry point for SysOM\nECS performance and stability diagnosis.\n\n## Immediate Route\n\nWhen the user reports a symptom and has not provided fresh SysOM envelope output,\nrun the matching SysOM command from **Domain Routing** below before ad hoc Linux\ninspection or manual probing. Then follow the returned `agent.summary`,\n`agent.findings[].detail/category`, and `agent.next_steps[]`. Raw Linux commands\nare bounded fallbacks only when a SysOM command is unavailable, outputs\ncontradict each other, or a required entity remains missing after the focused\nSysOM command.\n\n## Credential Security\n\nNever print, echo, or ask for AccessKey ID or AccessKey Secret values. Remote\ncommands perform their own authentication checks. If a command returns an\nauthentication or permission error, explain the error and point the user to\n`references/ram-policies.md`; credential setup must happen outside the\nconversation.\n\n## CLI Setup\n\nCheck whether the CLI is available:\n\n```bash\ncommand -v sysom-osops\n```\n\nIf it is missing, install it:\n\n```bash\ncurl -fsSL --connect-timeout 1000 https://sysom-prd-cn-hangzhou.oss-cn-hangzhou.aliyuncs.com/sysom_prd/skill_cli/install.sh | sudo bash\n```\n\nThen verify only the binary:\n\n```bash\ncommand -v sysom-osops\n```\n\n## Core Workflow\n\n1. Classify the user's symptom into one SysOM domain: memory, IO, load/CPU,\n   network, or Java (GC/memory/CPU).\n2. Run the smallest SysOM command that matches that domain. Prefer a local\n   memory classify for unclear memory symptoms; for other domains, use the\n   matching documented remote action.\n3. Read only the default envelope fields: `ok`, `error`, `command`, and\n   `agent`.\n4. **Load domain references before building the answer.** This step is mandatory\n   and must not be skipped even when `agent.findings` and `agent.next_steps`\n   appear complete. Which references to load depends on the domain:\n   - Java (any type: gc/memory/cpu) → read `references/java/README.md` first\n     for symptom routing and parameter validation; then by type:\n     - gc: `references/java/gc/gc-guide.md`\n     - memory: `references/java/memory/memory-guide.md` (then glossary, envelope\n       guide, profiling playbook, decision tree under `references/java/memory/`)\n     - cpu: `references/java/cpu/cpu-guide.md`\n   - Other domains → load the matching reference from the References table below.\n   References add interpretation rules, entity definitions, and answer-shaping\n   guidance that the envelope alone does not convey. Do not infer Java terms,\n   native memory categories, or profiling semantics from raw envelope text.\n5. Relay the hop as visible progress: present `agent.summary` (plus key\n   findings) to the user, interpreted through the reference material loaded in\n   step 4. Keep evidence qualifiers that change interpretation, including\n   currentness, unavailable direct signals, fallback evidence, and remediation\n   preconditions.\n6. Branch on `agent.status`:\n   - `concluded` (or missing/unrecognized) → build the final answer from\n     `agent.summary`, `agent.findings[].detail/category`, and\n     `agent.next_steps[]`, then stop the loop.\n   - `in_progress` → the backend is requesting another collection hop: take\n     the `kind=command` entry from `agent.next_steps[]`, apply the\n     confirmation rules below, run it, and feed the new envelope back into\n     step 4.\n\n**Guided-diagnosis loop hard rules** (Java multi-hop sessions):\n\n- Pace ownership: never skip `agent.next_steps[]` to decide collection on\n  your own, and never run diagnostic commands outside the envelope.\n- Run commands **exactly as shown** — the gateway has already injected\n  `--session-id`; never rewrite, add, or remove flags. If a command fails,\n  relay the error envelope as-is instead of retrying with tweaked parameters.\n- Hop limit: stop the loop after 4 hops in the same session even if the\n  backend still says `in_progress`; present the conclusions so far and state\n  the evidence limits (the backend force-concludes at the same ceiling —\n  double safety).\n- User refusal: if the user declines a proposed command, stop the loop,\n  summarize from the evidence already collected, and state explicitly which\n  conclusions remain unconfirmed because that hop was not run.\n\n**Before executing a profiling or long-running follow-up command** (e.g.,\n`java analyze --type memory --duration N`, any command that injects an agent\ninto the target process, or any command expected to run for multiple minutes):\n- Tell the user what the command does, how long it takes, and what performance\n  impact it may have on the target process (e.g., CPU overhead from sampling,\n  extra memory from the injected agent, potential safepoint pauses).\n- Ask the user whether to proceed. Do not run the command until the user\n  confirms, or until the user has previously given a standing instruction to\n  auto-run follow-ups.\n- Once the command starts, tell the user the expected wait time and keep them\n  informed if the operation is still in progress.\n\nRead-only query commands (e.g. `java analyze --type cpu`) do not apply here —\nsee each domain guide's Execution Model for specifics.\n\nWhen classify returns a command in `agent.next_steps[]` and no root-cause\nfinding already contains enough evidence to answer, run the first command next.\nDo not replace an Agent-visible SysOM next step with manual shell probing. Raw\nLinux checks are bounded fallbacks after the SysOM next step succeeds, fails, or\ntimes out.\n\nUse the documented commands exactly as shown by default. Do not add raw,\ndebug, or backend evidence expansion flags unless the user explicitly asks for\nthat view.\n\nFinal answers should name evidence, root cause, owner/scope, and operational\naction targets. Do not add shell snippets for verification or remediation unless\nthe user explicitly asks for commands. Prefer phrases such as \"review dependency\nand disable or upgrade the leaking component in a change window\" over raw module,\ncgroup, sysctl, cache-drop, or process-kill commands.\nDo not include command-looking inline snippets such as module inspection/removal,\nmemory summary commands, cgroup file writes, cache-drop controls, sysctl changes,\nor process-kill commands as default final-answer steps.\n\nThe `agent` view must be self-contained for diagnosis. Structured evidence is a\nbackend/UI view and must not be treated as the default Agent source for required\nentities.\n\n## Domain Routing\n\n| User symptom | First route |\n|--------------|-------------|\n| Unclear memory issue, OOM, high RSS, file cache, shmem/tmpfs, memory cgroup, socket memory, kernel memory | `sysom-osops memory classify` |\n| Java issue (symptom unclear) | Follow the Symptom Triage rules in `references/java/README.md` — ask the user about the symptom, then route to the matching type |\n| Java GC pause / frequent GC / low GC throughput | `sysom-osops java analyze --type gc` |\n| Java heap / OOM / heap leak / native leak | `sysom-osops java analyze --type memory` — without a pid/pod it returns a candidate list; STOP and wait for the user to choose before retrying |\n| Java CPU hotspot / high thread CPU / flame graph | `sysom-osops java analyze --type cpu --pid <PID>` |\n| Slow disk, high iowait, disk latency, blocked IO | `sysom-osops io iofsstat`, then `io iodiagnose` if the overview points to slow IO |\n| High load, runqueue backlog, task stuck waiting for CPU | `sysom-osops load loadtask` or `load delay` based on the visible symptom |\n| Packet loss, retransmits, network timeout, jitter | `sysom-osops net packetdrop` for loss/drop symptoms; `net netjitter` for latency fluctuation |\n\nFor command parameters, read `references/deep-actions.md` and\n`references/parameter-guide.md`. For OS and region support, read\n`references/supported-environments.md`. These references are Skill material; do\nnot use remote target file tools to open `.claude/skills` paths on the diagnosed\nhost.\nFor Java symptom-to-type routing, consult `references/java/README.md`.\n\n## Memory Routing\n\nMemory follows the same Core Workflow and Follow-up Rules as every domain: start\nfrom `sysom-osops memory classify`, then pick the next action from visible output\nor `agent.next_steps[]`. For choosing among memory deep actions or checking which\nentity is still missing, load `references/memory-triage.md` (parallel to\n`references/non-memory-triage.md` for other domains).\n\nNote: Java-related memory symptoms (OOM in Java process, heap leak, native leak)\nroute to Java domain via `sysom-osops java analyze --type memory`, not through\nmemory classify. See Domain Routing table above.\n\nChoose the next memory action from visible SysOM output. Do not infer a memory mechanism from symptom wording alone.\n\n## Envelope Contract\n\nDefault command output is the Agent contract:\n\n```json\n{\n  \"ok\": true,\n  \"command\": \"sysom-osops memory classify\",\n  \"agent\": {\n    \"status\": \"concluded\",\n    \"session_id\": \"a1b2c3d4e5f6\",\n    \"summary\": \"Concise diagnosis summary.\",\n    \"findings\": [\n      {\n        \"severity\": \"high\",\n        \"title\": \"Short finding title\",\n        \"detail\": \"Root cause, key entities, and evidence summary.\",\n        \"category\": \"root_cause\"\n      }\n    ],\n    \"next_steps\": [\n      {\n        \"kind\": \"command\",\n        \"label\": \"Run focused deep diagnosis\",\n        \"command\": \"sysom-osops memory oom\",\n        \"reason\": \"The missing entity this command can fill.\"\n      }\n    ]\n  }\n}\n```\n\n`agent.findings[]` may contain only `severity`, `title`, `detail`, and\n`category`. Required entities such as PID, cgroup, service, file path, OOM\nvictim, limit/current, residue, holder, or cleanup target must be written in\n`agent.summary` or `agent.findings[].detail`.\n\nField semantics for guided diagnosis sessions:\n\n- `agent.status`: `in_progress` means the backend diagnosis agent requests\n  another collection hop; `concluded` means diagnosis has converged. Treat a\n  missing or unrecognized value as `concluded`. Legacy collector-level states\n  (`success`, `warning`, ...) may still appear on non-Java actions; interpret\n  them as before.\n- `agent.session_id`: backend-generated multi-hop session identifier. Never\n  generate or modify it; the gateway already injects it into `command`\n  strings, so run them verbatim.\n- `agent.next_steps[].kind`: `command` = backend-requested collection command\n  (subject to the confirmation rules above); `info` = user-side suggestion —\n  present it but never auto-run; `warning` = evidence or data-quality caveat.\n\n## Follow-up Rules\n\n- Prefer `category=root_cause`, then highest severity, then the finding that\n  best matches the user's reported symptom.\n- Treat `root_cause` as stop-ready when visible `detail` contains the entities\n  needed to explain the symptom and a safe next action.\n- Treat `agent.next_steps[]` as a priority plan, not a checklist.\n- Run another SysOM command only when it can fill a named missing entity or\n  change remediation.\n- For long-running Java collection commands — `java analyze --type memory\n  --duration N` (profiling; legacy `memory javamem --duration N`) and `java\n  analyze --type gc` in `collect` mode (5–10 min JFR/GC collection) — the wait is\n  minutes-scale: tell the user, size the tool timeout accordingly, and never\n  re-fire the same command on client timeout. See\n  `references/java/memory/profiling-playbook.md` (memory) and\n  `references/java/gc/gc-guide.md` (gc).\n- Preserve visible qualifiers that affect interpretation, such as current versus\n  historical evidence, unavailable direct signals, fallback evidence used to\n  close currentness, and safety preconditions for remediation.\n- When a finding uses fallback evidence because a direct signal is unavailable,\n  state both parts in the final answer. Do not reduce the conclusion to the\n  fallback metric alone.\n- After a focused SysOM command closes a root cause, answer from it. Do not run\n  extra commands to make the report comprehensive, and do not chase earlier\n  classify anomalies or observations unless they share the same entity and\n  expose a named evidence gap.\n- Do not call backend-only collectors or private helper commands directly.\n- Do not re-check a PID, cgroup, file, limit, or event that SysOM already named\n  in `summary` or `detail`.\n- After a SysOM deep command returns `category=root_cause` with the required\n  entities visible, answer from that envelope. Raw Linux checks are only for\n  contradictions, command errors, or a clearly missing entity.\n- In the final answer, do not turn already-closed entities into extra raw Linux\n  verification commands. Express remediation as dependency-aware action targets\n  and change-window plans unless the envelope itself provides an executable safe\n  next step.\n- Avoid executable shell snippets in the final answer. If a command is useful\n  only for post-change verification, name the SysOM check or metric to re-run\n  instead of raw Linux commands.\n- This includes inline command names for module inspection/removal, memory\n  summary commands, cgroup file writes, cache-drop controls, sysctl changes, and\n  process-kill actions; describe the dependency gate and operational action\n  target in prose.\n- Pivot across domains when the current envelope does not explain the reported\n  symptom and another SysOM domain names a stronger root cause.\n- During diagnosis, do not execute remediation commands that change target\n  state, such as killing processes, removing files, changing sysctl values, or\n  writing to cache-drop controls. Present those as recommendations unless the\n  user explicitly asks you to perform the repair.\n- For non-memory findings, keep the same rule: one focused deep command, then\n  answer when the required entities are visible.\n\n## Error Handling\n\n| `error.code` | Action |\n|--------------|--------|\n| `Sysom.TargetRequired` | Ask for instance ID and region, or explain ECS metadata auto-detection requirements |\n| `Sysom.FallbackClassify` | Present the local classify result and continue only if a focused next step is available |\n| `Sysom.PermissionDenied` | Use `references/ram-policies.md` to explain required RAM permissions |\n| `Sysom.AuthenticationFailure` | Ask the user to configure credentials outside this session |\n| `Sysom.InvalidParameter` | Ask the user to correct the instance, region, or command parameter |\n| `Sysom.DiagnosisVersionNotSupported` | Explain that the target instance diagnosis components need an update |\n| `Sysom.DiagnosisJsonParseFailed` | Retry once only when the user still needs the same evidence |\n| `Sysom.PollError` | Retry the same focused action once when the missing evidence is still required |\n\n## References\n\n| Reference | Use when |\n|-----------|----------|\n| `references/classify-output-guide.md` | Reading local memory classify output |\n| `references/memory-triage.md` | Choosing a memory deep action or checking memory entity completeness |\n| `references/non-memory-triage.md` | Routing IO, load/CPU, network diagnosis |\n| `references/deep-actions.md` | Looking up SysOM commands by domain |\n| `references/parameter-guide.md` | Validating command parameters |\n| `references/report-interpretation.md` | Interpreting envelope fields and answer shape |\n| `references/ram-policies.md` | Explaining RAM permissions |\n| `references/supported-environments.md` | Checking OS, architecture, and region support |\n\n### Java Analysis References\n\n| Reference | Use when |\n|-----------|----------|\n| `references/java/README.md` | **Primary Java entry point**: symptom triage, parameter guide, sub-domain index |\n| `references/java/gc/gc-guide.md` | Running or interpreting `--type gc` results |\n| `references/java/cpu/cpu-guide.md` | Running or interpreting `--type cpu` results |\n| `references/java/memory/memory-guide.md` | `--type memory` interpretation and discovery-first flow entry |\n| `references/java/memory/glossary.md` | Java memory terminology |\n| `references/java/memory/javamem-envelope-guide.md` | Interpreting `--type memory` envelope structure |\n| `references/java/memory/profiling-playbook.md` | Preparation and expected behavior before `--duration` collection |\n| `references/java/memory/decision-tree.md` | Following backend `next_steps` in Java multi-hop sessions |\n| `references/java/memory/case-library.md` | Case library and anti-patterns |\n\nFile v0.0.5:references/java/README.md\n\n# Java Application Diagnosis Reference\n\nUnified command for all Java diagnostics:\n\n```\nsysom-osops java analyze --type <gc|memory|cpu>\n```\n\nThis document is the entry point for Java application diagnosis —\ncovering symptom triage, sub-domain guides, parameter quick-reference,\nand combined-diagnosis strategies. Each sub-domain (gc / memory / cpu)\nhas its own folder under `references/java/`.\n\n---\n\n## Symptom Triage\n\n### When the user's intent is unclear\n\nIf the user describes a vague Java problem (e.g., \"is my Java process healthy\",\n\"something seems wrong with my app\") without mentioning specific symptoms from\nthe table below, ask **one** clarifying question:\n\n> What is the primary symptom you are observing?\n> 1. Application response is slow or has intermittent pauses\n> 2. CPU usage is abnormally high\n> 3. Memory keeps growing or OOM has occurred\n\nThen route based on the answer:\n- 1 → `--type gc` (latency-related symptoms are most often GC-induced)\n- 2 → `--type cpu`\n- 3 → `--type memory`\n\n### Symptom → Diagnosis Type Mapping\n\nWhen the symptom is already clear, route directly:\n\n| Symptom | Recommended type | Description |\n|---------|-----------------|-------------|\n| Long GC pause / STW jitter | gc | Analyze GC event time-series and pause distribution |\n| Frequent Full GC / old generation full | gc | Detect memory pressure source and promotion patterns |\n| Low GC throughput / high application pause ratio | gc | Evaluate GC algorithm efficiency and tuning recommendations |\n| OOM / sustained heap growth | memory | Deep diagnosis of heap/non-heap leaks |\n| Native memory leak | memory | Analyze JNI/DirectBuffer usage |\n| High thread CPU usage | cpu | Flame graph to locate hotspot method stacks |\n| Slow application response (non-GC cause) | cpu | Analyze on-CPU time distribution |\n\n---\n\n## Sub-Domain Reference Index\n\n| type | Folder / entry guide | Scope |\n|------|----------------------|-------|\n| gc | [gc/gc-guide.md](gc/gc-guide.md) | GC pauses, throughput, heap trend analysis |\n| memory | [memory/memory-guide.md](memory/memory-guide.md) (+ `glossary`, `javamem-envelope-guide`, `decision-tree`, `profiling-playbook`, `profiling-interpretation`, `case-library` in `memory/`) | Heap/non-heap diagnosis, snapshots, allocation profiling |\n| cpu | [cpu/cpu-guide.md](cpu/cpu-guide.md) | Flame graph, hotspot method stacks, on-CPU analysis |\n\n---\n\n## Parameter Quick-Reference\n\n| Parameter | gc | memory | cpu |\n|-----------|:---:|:------:|:---:|\n| --pid | Optional | Optional | Optional (omit to get candidate list) |\n| --duration | Seconds (default 60) | Minutes (0=snapshot) | N/A |\n| --pod | Optional (container scenarios) | Optional | Not supported |\n\n---\n\n## Combined Diagnosis\n\nWhen a single type cannot locate the root cause, combine multiple types:\n\n1. **GC causing high CPU**: First `--type gc` to confirm GC frequency, then `--type cpu` to check GC thread proportion\n2. **Memory leak causing frequent GC**: First `--type gc` to observe Full GC patterns, then `--type memory` for deeper heap analysis\n3. **Heavy object allocation found in CPU hotspots**: First `--type cpu` to locate allocation hotspots, then `--type memory --duration 5` to track allocation trends\n\n---\n\n## Prerequisites\n\n- `--type cpu` requires the instance to be managed in the Alibaba Cloud Linux console (the system auto-dete\n\nArchive v0.0.4: 11 files, 16692 bytes\n\nFiles: references/classify-output-guide.md (1517b), references/deep-actions.md (3212b), references/memory-triage.md (5398b), references/non-memory-triage.md (1975b), references/parameter-guide.md (2848b), references/ram-policies.md (1221b), references/report-interpretation.md (2346b), references/supported-environments.md (1051b), skill-card.md (3466b), SKILL.md (11430b), _meta.json (147b)\n\nArchive v0.0.3: 102 files, 147084 bytes\n\nFiles: references/agent-conventions.md (3217b), references/authentication.md (5722b), references/cli-development-guide.md (5916b), references/diagnoses/delay.md (1768b), references/diagnoses/iodiagnose.md (1421b), references/diagnoses/iofsstat.md (1349b), references/diagnoses/javamem.md (1509b), references/diagnoses/loadtask.md (1281b), references/diagnoses/memgraph.md (2206b), references/diagnoses/netjitter.md (2113b), references/diagnoses/oomcheck.md (2272b), references/diagnoses/packetdrop.md (2060b), references/diagnoses/README.md (2814b), references/invoke-diagnosis.md (4276b), references/memory-routing.md (2744b), references/metadata-api.md (5811b), references/non-memory-routing.md (2078b), references/openapi-permission-guide.md (8072b), references/output-format.md (1979b), references/permission-guide.md (1925b), references/ram-policies.md (942b), references/service-linked-role-subaccount.md (1026b), scripts/init.sh (3410b), scripts/osops.sh (875b), scripts/pyproject.toml (918b), scripts/requirements.txt (243b), scripts/sysom_cli/__init__.py (115b), scripts/sysom_cli/__main__.py (9835b), scripts/sysom_cli/configure/__init__.py (86b), scripts/sysom_cli/configure/command.py (7913b), scripts/sysom_cli/core/__init__.py (602b), scripts/sysom_cli/core/base.py (5121b), scripts/sysom_cli/core/executor.py (1592b), scripts/sysom_cli/core/registry.py (8704b), scripts/sysom_cli/diagnosis/__init__.py (49b), scripts/sysom_cli/diagnosis/invoke/__init__.py (19b), scripts/sysom_cli/diagnosis/invoke/command.py (11852b), scripts/sysom_cli/io/__init__.py (91b), scripts/sysom_cli/io/iodiagnose/__init__.py (24b), scripts/sysom_cli/io/iodiagnose/command.py (547b), scripts/sysom_cli/io/iofsstat/__init__.py (24b), scripts/sysom_cli/io/iofsstat/command.py (538b), scripts/sysom_cli/lib/__init__.py (732b), scripts/sysom_cli/lib/auth.py (33998b), scripts/sysom_cli/lib/diagnosis_backend.py (1923b), scripts/sysom_cli/lib/diagnosis_helper.py (12263b), scripts/sysom_cli/lib/diagnosis_source.py (919b), scripts/sysom_cli/lib/ecs_metadata.py (4145b), scripts/sysom_cli/lib/guidance.py (18066b), scripts/sysom_cli/lib/invoke_envelope_finalize.py (5661b), scripts/sysom_cli/lib/kernel_log.py (1035b), scripts/sysom_cli/lib/log_parser.py (11206b), scripts/sysom_cli/lib/log_plugin.py (1799b), scripts/sysom_cli/lib/openapi_client.py (15905b), scripts/sysom_cli/lib/precheck_envelope.py (10244b), scripts/sysom_cli/lib/precheck_gate.py (2543b), scripts/sysom_cli/lib/precheck_summary.py (10768b), scripts/sysom_cli/lib/schema.py (1688b), scripts/sysom_cli/lib/specialty_args.py (1580b), scripts/sysom_cli/lib/specialty_command.py (860b), scripts/sysom_cli/load/__init__.py (99b), scripts/sysom_cli/load/delay/__init__.py (24b), scripts/sysom_cli/load/delay/command.py (527b), scripts/sysom_cli/load/loadtask/__init__.py (24b), scripts/sysom_cli/load/loadtask/command.py (535b), scripts/sysom_cli/memory/__init__.py (91b), scripts/sysom_cli/memory/classify/__init__.py (18b), scripts/sysom_cli/memory/classify/command.py (5363b), scripts/sysom_cli/memory/javamem/__init__.py (24b), scripts/sysom_cli/memory/javamem/command.py (4486b), scripts/sysom_cli/memory/lib/__init__.py (64b), scripts/sysom_cli/memory/lib/classify_engine.py (8088b), scripts/sysom_cli/memory/lib/envelope_memory.py (6832b), scripts/sysom_cli/memory/lib/invoke_bridge.py (1021b), scripts/sysom_cli/memory/lib/memory_envelope_finalize.py (7142b), scripts/sysom_cli/memory/lib/memory_remote_helpers.py (3183b), scripts/sysom_cli/memory/lib/oom_log_extract.py (9906b), scripts/sysom_cli/memory/lib/oom_quick.py (17639b), scripts/sysom_cli/memory/lib/remote_capabilities.py (922b), scripts/sysom_cli/memory/lib/shared_invoke_args.py (1660b)\n\nArchive v0.0.2: 101 files, 159019 bytes\n\nFiles: references/agent-conventions.md (2904b), references/authentication.md (15339b), references/cli-development-guide.md (21687b), references/diagnoses/delay.md (1812b), references/diagnoses/iodiagnose.md (1397b), references/diagnoses/iofsstat.md (1378b), references/diagnoses/javamem.md (1495b), references/diagnoses/loadtask.md (1272b), references/diagnoses/memgraph.md (2616b), references/diagnoses/netjitter.md (2068b), references/diagnoses/oomcheck.md (2212b), references/diagnoses/packetdrop.md (1971b), references/diagnoses/README.md (2565b), references/invoke-diagnosis.md (4091b), references/memory-routing.md (2465b), references/metadata-api.md (5670b), references/non-memory-routing.md (2035b), references/openapi-permission-guide.md (13654b), references/output-format.md (1784b), references/permission-guide.md (1851b), references/ram-policies.md (895b), references/service-linked-role-subaccount.md (890b), scripts/init.sh (3410b), scripts/osops.sh (875b), scripts/pyproject.toml (918b), scripts/requirements.txt (243b), scripts/sysom_cli/__init__.py (115b), scripts/sysom_cli/__main__.py (9835b), scripts/sysom_cli/configure/__init__.py (86b), scripts/sysom_cli/configure/command.py (7913b), scripts/sysom_cli/core/__init__.py (602b), scripts/sysom_cli/core/base.py (5121b), scripts/sysom_cli/core/executor.py (1592b), scripts/sysom_cli/core/registry.py (8704b), scripts/sysom_cli/diagnosis/__init__.py (49b), scripts/sysom_cli/diagnosis/invoke/__init__.py (19b), scripts/sysom_cli/diagnosis/invoke/command.py (11852b), scripts/sysom_cli/io/__init__.py (91b), scripts/sysom_cli/io/iodiagnose/__init__.py (24b), scripts/sysom_cli/io/iodiagnose/command.py (547b), scripts/sysom_cli/io/iofsstat/__init__.py (24b), scripts/sysom_cli/io/iofsstat/command.py (538b), scripts/sysom_cli/lib/__init__.py (732b), scripts/sysom_cli/lib/auth.py (33994b), scripts/sysom_cli/lib/diagnosis_backend.py (1923b), scripts/sysom_cli/lib/diagnosis_helper.py (12263b), scripts/sysom_cli/lib/diagnosis_source.py (3324b), scripts/sysom_cli/lib/ecs_metadata.py (4145b), scripts/sysom_cli/lib/guidance.py (18066b), scripts/sysom_cli/lib/invoke_envelope_finalize.py (5661b), scripts/sysom_cli/lib/kernel_log.py (1035b), scripts/sysom_cli/lib/log_parser.py (11206b), scripts/sysom_cli/lib/log_plugin.py (1799b), scripts/sysom_cli/lib/openapi_client.py (15905b), scripts/sysom_cli/lib/precheck_envelope.py (10244b), scripts/sysom_cli/lib/precheck_gate.py (2543b), scripts/sysom_cli/lib/precheck_summary.py (10768b), scripts/sysom_cli/lib/schema.py (1688b), scripts/sysom_cli/lib/specialty_args.py (1580b), scripts/sysom_cli/lib/specialty_command.py (860b), scripts/sysom_cli/load/__init__.py (99b), scripts/sysom_cli/load/delay/__init__.py (24b), scripts/sysom_cli/load/delay/command.py (527b), scripts/sysom_cli/load/loadtask/__init__.py (24b), scripts/sysom_cli/load/loadtask/command.py (535b), scripts/sysom_cli/memory/__init__.py (91b), scripts/sysom_cli/memory/classify/__init__.py (18b), scripts/sysom_cli/memory/classify/command.py (5363b), scripts/sysom_cli/memory/javamem/__init__.py (24b), scripts/sysom_cli/memory/javamem/command.py (4486b), scripts/sysom_cli/memory/lib/__init__.py (64b), scripts/sysom_cli/memory/lib/classify_engine.py (8088b), scripts/sysom_cli/memory/lib/envelope_memory.py (6832b), scripts/sysom_cli/memory/lib/invoke_bridge.py (1021b), scripts/sysom_cli/memory/lib/memory_envelope_finalize.py (7142b), scripts/sysom_cli/memory/lib/memory_remote_helpers.py (3183b), scripts/sysom_cli/memory/lib/oom_log_extract.py (9906b), scripts/sysom_cli/memory/lib/oom_quick.py (17639b), scripts/sysom_cli/memory/lib/remote_capabilities.py (922b), scripts/sysom_cli/memory/lib/shared_invoke_args.py (1660b)\n\nArchive v0.0.1: 19 files, 25315 bytes\n\nFiles: references/agent-conventions.md (2924b), references/diagnoses/delay.md (1820b), references/diagnoses/iodiagnose.md (1405b), references/diagnoses/iofsstat.md (1386b), references/diagnoses/javamem.md (1503b), references/diagnoses/loadtask.md (1130b), references/diagnoses/memgraph.md (2628b), references/diagnoses/netjitter.md (2076b), references/diagnoses/oomcheck.md (2055b), references/diagnoses/packetdrop.md (1979b), references/diagnoses/README.md (2607b), references/invoke-diagnosis.md (3872b), references/memory-routing.md (2489b), references/non-memory-routing.md (1737b), references/output-format.md (1784b), references/permission-guide.md (1893b), references/ram-policies.md (895b), SKILL.md (5878b), _meta.json (147b)","readmeExcerpt":"Skill: alibabacloud-sysom-diagnosis Owner: sdk-team Summary: Use when troubleshooting Linux server performance or stability issues — CPU saturation, high load, scheduling delay, memory pressure, OOM events, high RSS, page cache / shared memory growth, memory cgroup residue, Java heap issues, disk IO saturation or latency, packet loss, network jitter, or a server that is slow, stuck, or unstable. Performs diagnosis an","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"command -v sysom-osops"},{"language":"bash","snippet":"curl -fsSL --connect-timeout 1000 https://sysom-prd-cn-hangzhou.oss-cn-hangzhou.aliyuncs.com/sysom_prd/skill_cli/install.sh \\"},{"language":"bash","snippet":"curl -fsSL --connect-timeout 1000 https://sysom-prd-cn-hangzhou.oss-cn-hangzhou.aliyuncs.com/sysom_prd/skill_cli/install.sh \\\n  | sudo bash"},{"language":"bash","snippet":"curl -fsSL --connect-timeout 1000 https://sysom-prd-cn-hangzhou.oss-cn-hangzhou.aliyuncs.com/sysom_prd/skill_cli/install.sh \\"},{"language":"bash","snippet":"mkdir -p ~/.local/bin\ncurl -fsSL --connect-timeout 1000 https://sysom-prd-cn-hangzhou.oss-cn-hangzhou.aliyuncs.com/sysom_prd/skill_cli/install.sh \\\n  | bash -s -- -d \"$HOME/.local/bin\""},{"language":"bash","snippet":"export PATH=\"$HOME/.local/bin:$PATH\""}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: alibabacloud-sysom-diagnosis\ndescription: >\n  Use when troubleshooting Linux server performance or stability issues —\n  CPU saturation, high load, scheduling delay, memory pressure, OOM events,\n  high RSS, page cache / shared memory growth, memory cgroup residue, Java\n  heap issues, disk IO saturation or latency, packet loss, network jitter,\n  or a server that is slow, stuck, or unstable. Performs diagnosis and\n  surfaces recommendations; does not apply fixes automatically.\nlicense: Apache-2.0\ncompatibility: >\n  Requires sysom-osops CLI. The control host running the CLI can be Linux or\n  macOS (x86_64 or aarch64); Windows is not supported. Remote diagnosis targets\n  a Linux ECS instance and requires Alibaba Cloud credentials through AK/SK or\n  an ECS RAM Role, an online Cloud Assistant on the target ECS, and a supported\n  China Mainland or Hong Kong region.\nmetadata:\n  domain: aiops\n  product: sysom\n  supported_domains:\n    - cpu\n    - io\n    - memory\n    - network\n    - java\n  owner: sysom-team\n  contact: sysom-team@alibaba-inc.com\nallowed-tools: Bash Read\n---\n\n# alibabacloud-sysom-diagnosis\n\nUse SysOM CLI and backend envelopes as the diagnosis source of truth. This Skill replaces the older SysOM diagnosis Skill and is the single entry point for SysOM ECS performance and stability diagnosis.\n\n## Immediate Route\n\nWhen the user reports a symptom and has not provided fresh SysOM envelope output,\nrun the matching SysOM command from **Domain Routing** below before ad hoc Linux\ninspection or manual probing. Then follow the returned `agent.summary`,\n`agent.findings[].detail/category`, and `agent.next_steps[]`. Raw Linux commands\nare bounded fallbacks only when a SysOM command is unavailable, outputs\ncontradict each other, or a required entity remains missing after the focused\nSysOM command.\n\n## Credential Security\n\nNever print, echo, or ask for AccessKey ID or AccessKey Secret values. Remote\ncommands perform their own authentication checks. If a command returns an\nauthentication or permission error, explain the error and point the user to\n`references/ram-policies.md`; credential setup must happen outside the\nconversation.\n\n## CLI Setup\n\nCheck whether the CLI is available:\n\n```bash\ncommand -v sysom-osops\n```\n\nIf it is missing, install it. The installer runs on **Linux or macOS** — it does\nnot run on Windows. The target ECS being diagnosed must be Linux (see\n`references/supported-environments.md`), but the control host can be either OS.\n\nSystem-wide install (needs write access to `/usr/local/bin`, typically via\n`sudo`):\n\n```bash\ncurl -fsSL --connect-timeout 1000 https://sysom-prd-cn-hangzhou.oss-cn-hangzhou.aliyuncs.com/sysom_prd/skill_cli/install.sh \\\n  | sudo bash\n```\n\nUser-local install — no sudo, no root-owned paths. Works on both Linux and\nmacOS, and is the recommended path when you do not have administrator\nprivileges:\n\n```bash\nmkdir -p ~/.local/bin\ncurl -fsSL --connect-timeout 1000 https://sysom-prd-cn-hangzhou.oss-cn-hangzhou.aliyuncs.c"},{"path":"references/java/README.md","content":"# Java Application Diagnosis Reference\n\nUnified command for all Java diagnostics:\n\n```\nsysom-osops java analyze --type <gc|memory|cpu>\n```\n\nThis document is the entry point for Java application diagnosis —\ncovering symptom triage, sub-domain guides, parameter quick-reference,\nand combined-diagnosis strategies. Each sub-domain (gc / memory / cpu)\nhas its own folder under `references/java/`.\n\n---\n\n## Symptom Triage\n\n### When the user's intent is unclear\n\nIf the user describes a vague Java problem (e.g., \"is my Java process healthy\",\n\"something seems wrong with my app\") without mentioning specific symptoms from\nthe table below, ask **one** clarifying question:\n\n> What is the primary symptom you are observing?\n> 1. Application response is slow or has intermittent pauses\n> 2. CPU usage is abnormally high\n> 3. Memory keeps growing or OOM has occurred\n\nThen route based on the answer:\n- 1 → `--type gc` (latency-related symptoms are most often GC-induced)\n- 2 → `--type cpu`\n- 3 → `--type memory`\n\n### Symptom → Diagnosis Type Mapping\n\nWhen the symptom is already clear, route directly:\n\n| Symptom | Recommended type | Description |\n|---------|-----------------|-------------|\n| Long GC pause / STW jitter | gc | Analyze GC event time-series and pause distribution |\n| Frequent Full GC / old generation full | gc | Detect memory pressure source and promotion patterns |\n| Low GC throughput / high application pause ratio | gc | Evaluate GC algorithm efficiency and tuning recommendations |\n| OOM / sustained heap growth | memory | Deep diagnosis of heap/non-heap leaks |\n| Native memory leak | memory | Analyze JNI/DirectBuffer usage |\n| High thread CPU usage | cpu | Flame graph to locate hotspot method stacks |\n| Slow application response (non-GC cause) | cpu | Analyze on-CPU time distribution |\n\n---\n\n## Sub-Domain Reference Index\n\n| type | Folder / entry guide | Scope |\n|------|----------------------|-------|\n| gc | [gc/gc-guide.md](gc/gc-guide.md) | GC pauses, throughput, heap trend analysis |\n| memory | [memory/memory-guide.md](memory/memory-guide.md) (+ `glossary`, `javamem-envelope-guide`, `decision-tree`, `profiling-playbook`, `profiling-interpretation`, `case-library` in `memory/`) | Heap/non-heap diagnosis, snapshots, allocation profiling |\n| cpu | [cpu/cpu-guide.md](cpu/cpu-guide.md) | Flame graph, hotspot method stacks, on-CPU analysis |\n\n---\n\n## Parameter Quick-Reference\n\n| Parameter | gc | memory | cpu |\n|-----------|:---:|:------:|:---:|\n| --pid | Optional | Optional | Optional (omit to get candidate list) |\n| --duration | Seconds (default 60) | Minutes (0=snapshot) | N/A |\n| --pod | Optional (container scenarios) | Optional | Not supported |\n\n---\n\n## Combined Diagnosis\n\nWhen a single type cannot locate the root cause, combine multiple types:\n\n1. **GC causing high CPU**: First `--type gc` to confirm GC frequency, then `--type cpu` to check GC thread proportion\n2. **Memory leak causing frequent GC**: First `--type gc` to observe Full GC patterns, then `--t"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn74p5w8ywv6prh40g0s82gmqh83nw54\",\n  \"slug\": \"alibabacloud-sysom-diagnosis\",\n  \"version\": \"0.0.7\",\n  \"publishedAt\": 1788839391292\n}"},{"path":"references/classify-output-guide.md","content":"# Classify Output Guide\n\nUse `sysom-osops memory classify` as the local entry point for unclear memory\nsymptoms. It returns the same default Agent envelope as remote deep actions.\n\n## What To Read\n\n| Field | Use |\n|-------|-----|\n| `agent.summary` | Overall memory verdict and dominant issue. |\n| `agent.findings[].category` | Prefer `root_cause` when present. |\n| `agent.findings[].detail` | Agent-visible root-cause entities and missing evidence. |\n| `agent.next_steps[]` | Focused remote actions that can fill missing evidence. |\n\nDo not call backend-only collectors from the Skill. Do not depend on legacy or\ninternal fields outside the default Agent envelope.\n\n## Choosing The First Follow-Up\n\n1. If a root-cause finding already contains the required entities, answer from\n   classify.\n2. If the dominant finding lacks a required entity, run the first relevant\n   `kind=command` next step.\n3. If the visible envelope names an ownership, currentness, attribution, or\n   cleanup gap, follow the focused SysOM action that closes that entity gap.\n4. If a later deep action returns concrete holder or owner entities, combine\n   them with classify instead of re-running broad local checks.\n\n## Next Steps May Not Be Runnable Yet\n\n`agent.next_steps[]` can recommend remote deep commands such as\n`memory filecache` or `memory memcgoffline` even though classify itself ran\nlocally without credentials. Those commands are registered only while the skills\ncatalog is reachable, so on an unconfigured machine they do not exist yet.\n\nA `status: normal` classify result therefore does not imply the recommended\nfollow-up is available. If the follow-up returns domain help text or empty\noutput, treat it as a missing command per the rules in `SKILL.md`, tell the user\ndeep analysis needs credentials, and do not fall back to inventing findings or to\nbroad manual shell probing.\n\n## When No Memory Action Is Needed\n\nIf classify returns no memory finding, or all findings are informational and do\nnot match the user's symptom, report that current SysOM memory evidence is\nhealthy or inconclusive. Pivot to IO, load, network, or Java only when the\nvisible envelope or the user's symptom supports that domain."},{"path":"references/deep-actions.md","content":"# Deep Actions Reference\n\nThis file lists public SysOM commands that this Skill may route to. On an ECS\ninstance, sysom-osops can auto-detect `--instance` and `--region`; add them only\nfor cross-instance diagnosis or when auto-detection fails.\n\n## Memory\n\n| Command | Mode | Use when |\n|---------|------|----------|\n| `sysom-osops memory classify` | Local | First route for unclear memory symptoms, OOM hints, high RSS, cache growth, shmem/tmpfs, memcg residue, or kernel memory suspicion |\n| `sysom-osops memory memgraph` | Remote | Full memory landscape is missing after classify, or the issue is broad kernel/userspace memory composition |\n| `sysom-osops memory memgraph --enable-socket` | Remote | Socket buffer pressure is visible and socket holder, state, PID, cgroup, or service attribution is missing |\n| `sysom-osops memory process` | Remote | A process is the suspected holder but identity, real executable, cgroup, or service attribution is missing |\n| `sysom-osops memory oom` | Remote | OOM killer event or memcg/host OOM evidence is visible |\n| `sysom-osops memory filecache` | Remote | File/page cache is the dominant unresolved entity and file/holder attribution is missing |\n| `sysom-osops memory shmem` | Remote | shmem, tmpfs, memfd, or SysV shared memory holder attribution is missing |\n| `sysom-osops memory memleak` | Remote | Kernel-hidden growth (vmalloc/page-allocator/slab/percpu) is the dominant unresolved entity and the leaking call point/function/module is missing after memgraph closed only to a candidate (`--type slab\\|page\\|vmalloc\\|percpu`, default vmalloc) |\n| `sysom-osops memory javamem` | Remote | **DEPRECATED** — Use `sysom-osops java analyze --type memory` instead (see Java section below) |\n| `sysom-osops memory memcgoffline` | Remote | Cgroup ownership-transition evidence is visible in SysOM output and original ownership, residue/refcount, or cleanup order needs attribution |\n\n## IO\n\n| Command | Mode | Use when |\n|---------|------|----------|\n| `sysom-osops io iofsstat` | Remote | Disk IO overview is needed for high iowait, slow disk, or IO saturation |\n| `sysom-osops io iodiagnose` | Remote | Slow IO root-cause attribution is needed after the overview or when latency is the primary symptom |\n\n## Load and CPU Scheduling\n\n| Command | Mode | Use when |\n|---------|------|----------|\n| `sysom-osops load loadtask` | Remote | Load average, runqueue, or task composition is the primary symptom |\n| `sysom-osops load delay` | Remote | Runnable tasks are not getting CPU time, scheduling delay is reported, or processes appear stuck without IO evidence |\n\n## Network\n\n| Command | Mode | Use when |\n|---------|------|----------|\n| `sysom-osops net packetdrop` | Remote | Packet loss, retransmits, drops, connection resets, or timeout symptoms are primary |\n| `sysom-osops net netjitter` | Remote | Latency fluctuation, jitter, or intermittent connectivity degradation is primary |\n\n## Java\n\nRoute by symptom through the unified `sysom-osops java analyze"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":2515,"uniquenessScore":38,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-10T14:45:50.559Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-10T14:45:50.559Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T17:36:55.469Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}