{"id":"50bdedb0-cfa1-4d75-82e8-d79dbaf61d94","entityType":"agent","slug":"clawhub-voronindenis5-failure-forensics","name":"failure-forensics","canonicalUrl":"https://www.xpersona.co/agent/clawhub-voronindenis5-failure-forensics","canonicalPath":"/agent/clawhub-voronindenis5-failure-forensics","generatedAt":"2026-10-10T05:57:52.035Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-09T14:15:58.668Z","emptyReason":null},"description":"Use when an agent task fails or produces unexpected results. Performs structured post-mortem root cause analysis: categorizes the failure, traces the exact failure point through tool-call logs, reconstructs the decision chain, generates a post-mortem report, and saves lessons to prevent recurrence. Skill: failure-forensics Owner: voronindenis5 Summary: Use when an agent task fails or produces unexpected results. Performs structured post-mortem root cause analysis: categorizes the failure, traces the exact failure point through tool-call logs, reconstructs the decision chain, generates a post-mortem report, and saves lessons to prevent recurrence. Tags: latest:0.1.1 Version history: v0.1.1 | 2026-08-11T11:58:54.","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 2.5K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s17b6amkd3wzqgg640v03a9r1n83gxs1:failure-forensics","sourceUrl":"https://clawhub.ai/voronindenis5/failure-forensics","homepage":"https://clawhub.ai/voronindenis5/skills/failure-forensics","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/voronindenis5/failure-forensics","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/voronindenis5/skills/failure-forensics","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":68,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Use when an agent task fails or produces unexpected results. Performs structured post-mortem root cause analysis: categorizes the failure, traces the exact fail"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T14:15:58.668Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T14:15:58.668Z","emptyReason":null},"stars":null,"forks":null,"downloads":2501,"packageName":null,"latestVersion":"0.1.1","tractionLabel":"2.5K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T14:15:58.668Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T14:15:58.668Z","lastCrawledAt":"2026-10-09T14:15:58.668Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T14:15:58.668Z","lastVerifiedAt":null,"highlights":[{"version":"0.1.1","createdAt":"2026-08-11T11:58:54.013Z","changelog":"- Removed the file skill-card.md to clean up redundant or outdated documentation. - No changes to core logic or usage. - All main documentation and workflow details now remain in SKILL.md.","fileCount":11,"zipByteSize":23490},{"version":"0.1.0","createdAt":"2026-08-05T19:49:41.304Z","changelog":"Initial release of failure-forensics. - Provides structured post-mortem root cause analysis for failed agent tasks. - Categorizes failures, reconstructs timelines, and traces causal chains using tool-call logs. - Generates detailed post-mortem reports with lessons learned to prevent recurrence. - Includes a CLI script to analyze logs and classify errors. - Designed to build a persistent knowledge base of past failures for improved agent reliability.","fileCount":11,"zipByteSize":23413}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17b6amkd3wzqgg640v03a9r1n83gxs1:failure-forensics","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-voronindenis5-failure-forensics/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-voronindenis5-failure-forensics/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-voronindenis5-failure-forensics/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-voronindenis5-failure-forensics/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-voronindenis5-failure-forensics/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-voronindenis5-failure-forensics/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T05:57:52.033Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-voronindenis5-failure-forensics/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-voronindenis5-failure-forensics/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-voronindenis5-failure-forensics/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-voronindenis5-failure-forensics/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-09T14:15:58.668Z","emptyReason":null},"readme":"Skill: failure-forensics\n\nOwner: voronindenis5\n\nSummary: Use when an agent task fails or produces unexpected results. Performs structured post-mortem root cause analysis: categorizes the failure, traces the exact failure point through tool-call logs, reconstructs the decision chain, generates a post-mortem report, and saves lessons to prevent recurrence.\n\nTags: latest:0.1.1\n\nVersion history:\n\nv0.1.1 | 2026-08-11T11:58:54.013Z | auto\n\n- Removed the file skill-card.md to clean up redundant or outdated documentation.\n- No changes to core logic or usage.\n- All main documentation and workflow details now remain in SKILL.md.\n\nv0.1.0 | 2026-08-05T19:49:41.304Z | auto\n\nInitial release of failure-forensics.\n\n- Provides structured post-mortem root cause analysis for failed agent tasks.\n- Categorizes failures, reconstructs timelines, and traces causal chains using tool-call logs.\n- Generates detailed post-mortem reports with lessons learned to prevent recurrence.\n- Includes a CLI script to analyze logs and classify errors.\n- Designed to build a persistent knowledge base of past failures for improved agent reliability.\n\nArchive index:\n\nArchive v0.1.1: 11 files, 23490 bytes\n\nFiles: LICENSE (1070b), README.md (3824b), references (0b), references/failure-taxonomy.md (10215b), references/post-mortem-template.md (5800b), scripts (0b), scripts/failure_forensics.py (19949b), scripts/sample_log.jsonl (1293b), skill-card.md (2323b), SKILL.md (10646b), _meta.json (136b)\n\nFile v0.1.1:SKILL.md\n\n---\nname: failure-forensics\ndescription: \"Use when an agent task fails or produces unexpected results. Performs structured post-mortem root cause analysis: categorizes the failure, traces the exact failure point through tool-call logs, reconstructs the decision chain, generates a post-mortem report, and saves lessons to prevent recurrence.\"\nversion: 1.0.0\nauthor: Denis Voronin\nlicense: MIT\nmetadata:\n  hermes:\n    tags: [debugging, post-mortem, forensics, root-cause-analysis, failure-analysis, agent-reliability]\n    related_skills: [systematic-debugging, debugging-hermes-tui-commands]\n---\n\n# Failure Forensics\n\n## Overview\n\nWhen an agent task fails, the default response is to retry — hoping for a different outcome. **Failure Forensics** rejects that reflex. Instead, the agent performs structured root cause analysis *before* retrying, treating every failure as evidence to be collected, categorized, and learned from.\n\nThe workflow has four phases:\n\n1. **Triage** — Categorize the failure using the taxonomy in [`references/failure-taxonomy.md`](references/failure-taxonomy.md).\n2. **Timeline Reconstruction** — Parse tool-call logs and agent decision points to build a chronological failure timeline. The script [`scripts/failure_forensics.py`](scripts/failure_forensics.py) automates this from JSON or JSONL log formats.\n3. **Causal Chain Analysis** — Trace the chain of decisions, assumptions, and actions that led from the task kickoff to the failure point. Identify the *root cause*, not just the proximate symptom.\n4. **Post-Mortem Report** — Generate a structured report from the template in [`references/post-mortem-template.md`](references/post-mortem-template.md) and persist it so future sessions can learn.\n\nThis skill turns a single failure into a permanent, reusable lesson.\n\n## When to Use\n\n- **An agent task failed** and retrying without understanding *why* is risky.\n- **A failure recurs** across attempts — you suspect a systemic cause, not bad luck.\n- **You need an artifact** documenting what went wrong for a team review or audit.\n- **A complex multi-step task** partially completed then broke — you need to understand which step is safe to resume from.\n- **You want to improve agent reliability** by building a corpus of past failure patterns.\n\n### Don't use for:\n\n- **Trivial failures with obvious fixes** (typo in a command, missing flag). Fix and move on.\n- **Live debugging** of an actively failing process — use `systematic-debugging` for that. Run forensics *after* the process is dead or the task is abandoned.\n- **Human performance reviews.** This skill analyzes agent + tool behavior, not people.\n\n## The Forensics Workflow\n\n### Phase 1: Triage — Categorize the Failure\n\nRead the full taxonomy in [`references/failure-taxonomy.md`](references/failure-taxonomy.md). At a high level, every failure falls into one of six categories:\n\n| Category | Signature | First Question |\n|---|---|---|\n| **Network** | Connection refused, timeout, DNS, TLS, 5xx HTTP | \"Is the endpoint reachable *right now*?\" |\n| **Permissions** | 401/403, EACCES, \"permission denied\", \"unauthorized\" | \"Does the credential/token have the needed scope?\" |\n| **Logic** | Code runs but output is wrong; assertions fail; data is corrupt | \"What assumption did the code make that was false?\" |\n| **Environment** | Missing binary, wrong version, missing env var, wrong OS | \"What does `env`/`which`/`uname` say vs. what was expected?\" |\n| **Dependency** | ImportError, version conflict, package not found, ABI mismatch | \"What changed in the dependency graph?\" |\n| **Resource** | OOM, disk full, too many open files, rate limit, quota exhausted | \"What was the ceiling, and what hit it?\" |\n\nRecord the category — it determines the questions you ask next.\n\n### Phase 2: Timeline Reconstruction\n\nCollect the evidence:\n\n1. **Tool-call logs.** If the agent session logged tool calls (JSON or JSONL with timestamps, tool name, args, result/error), feed them to the analyzer:\n\n   ```bash\n   python3 scripts/failure_forensics.py analyze \\\n     --log session.jsonl \\\n     --output timeline.md\n   ```\n\n   The script produces a chronological timeline with:\n   - Each tool call, its timestamp, duration, and outcome (success/failure)\n   - The **first failure point** flagged\n   - Error messages and exit codes extracted\n   - A summary of the failure category (inferred from error signatures)\n\n2. **Manual reconstruction.** If no structured logs exist, reconstruct the timeline from memory of the session. List each decision and action in order. Be honest about uncertainty — mark gaps explicitly.\n\n3. **What to capture for each step:**\n   - Timestamp (or relative ordering)\n   - The action or decision taken\n   - The *intent* behind it (what the agent was trying to achieve)\n   - The actual outcome\n   - Any assumption the agent made\n\n### Phase 3: Causal Chain Analysis\n\nThis is the core of forensics. You're looking for the **causal chain** — the sequence where each link made the next failure more likely.\n\nAsk these questions in order:\n\n1. **What was the immediate (proximate) cause of failure?**\n   - The error message, the crash point, the wrong output. This is the *symptom*.\n\n2. **What was the agent doing when it failed?**\n   - The specific tool call or action. What was the goal of that action?\n\n3. **Why did the agent take that action at that point?**\n   - Trace back one decision. What information did the agent have? What did it assume?\n\n4. **Was that assumption valid?**\n   - Check against logs, file contents, environment state. A failed assumption here is a link in the causal chain.\n\n5. **Continue backward** until you reach either:\n   - A **root cause**: a decision or state that, if different, would have prevented the entire failure cascade. Stop here.\n   - The **task kickoff**: if no single root cause emerges, the failure is *systemic* (multiple contributing factors).\n\n6. **Look for contributing factors** that didn't *cause* the failure but made it worse or harder to recover from:\n   - Missing retry logic\n   - Poor error messages that obscured the real problem\n   - Timeouts set too aggressively or too loosely\n   - Insufficient logging that made diagnosis harder\n\n**Anti-pattern: the \"five whys\" that stops at one.** The first \"why\" almost always produces the symptom, not the cause. Keep going. The root cause is usually 3-5 links back.\n\n### Phase 4: Post-Mortem Report\n\nFill out the template in [`references/post-mortem-template.md`](references/post-mortem-template.md). Key sections:\n\n- **Summary** — one paragraph, plain language. A reader who wasn't there should understand it.\n- **Timeline** — the reconstructed failure timeline from Phase 2.\n- **Root Cause** — the terminal link of the causal chain from Phase 3, stated plainly.\n- **Contributing Factors** — the rest of the chain.\n- **Action Items** — concrete, assigned, verifiable. \"Add retry logic\" is bad; \"Add exponential backoff retry (max 3 attempts) to the `web_extract` call in `session.py:142`\" is good.\n- **Lessons Learned** — generalizable insights. These are the durable output.\n\n**Save the report.** Write it to a persistent location (e.g., a `post-mortems/` directory, an issue tracker, or a knowledge base). A post-mortem that isn't saved didn't happen.\n\n## Using the Forensics Script\n\nThe script `scripts/failure_forensics.py` has three subcommands:\n\n### `analyze` — Build a failure timeline from logs\n\n```bash\n# JSONL log (one JSON object per line)\npython3 scripts/failure_forensics.py analyze --log session.jsonl --format jsonl\n\n# JSON array\npython3 scripts/failure_forensics.py analyze --log session.json --format json\n\n# Write report to file\npython3 scripts/failure_forensics.py analyze --log session.jsonl --output report.md\n```\n\n### `categorize` — Classify an error message\n\n```bash\npython3 scripts/failure_forensics.py categorize --error \"ConnectionRefusedError: [Errno 111] Connection refused\"\n# Output: network\n\npython3 scripts/failure_forensics.py categorize --error \"PermissionError: [Errno 13] Permission denied\"\n# Output: permissions\n```\n\n### `report` — Generate a post-mortem template pre-filled with timeline data\n\n```bash\npython3 scripts/failure_forensics.py report --log session.jsonl --title \"Deploy failure 2024-01-15\" --author \"agent\"\n```\n\n### Expected Log Format\n\nThe analyzer accepts JSON/JSONL where each entry is a tool call record:\n\n```json\n{\n  \"timestamp\": \"2024-01-15T10:23:45Z\",\n  \"tool\": \"terminal\",\n  \"args\": {\"command\": \"npm install\"},\n  \"result\": {\"success\": false, \"error\": \"EACCES: permission denied, open '/usr/lib/node_modules'\"},\n  \"duration_ms\": 1200\n}\n```\n\nRequired fields: `timestamp` (ISO 8601), `tool`. The script tolerates missing optional fields (`result`, `duration_ms`, `args`).\n\n## Common Pitfalls\n\n1. **Confusing symptom with cause.** \"The deploy failed because the build failed\" is a symptom. \"The build failed because `package-lock.json` was regenerated with a different Node version than CI\" is a cause. Keep digging past the symptom.\n\n2. **Stopping at human error.** \"I made a mistake\" is never a root cause. Ask *why* the mistake was possible — missing validation? Misleading documentation? Fatigue from context switching? Fix the system, not the human.\n\n3. **Writing action items that aren't actionable.** \"Be more careful\" is worthless. Every action item must specify *what* to change, *where*, and *how to verify* it works.\n\n4. **Skipping the timeline.** Without a chronological timeline, causal chain analysis becomes guessing. Build the timeline first, even if it's rough.\n\n5. **Not saving the post-mortem.** A post-mortem kept in chat history is lost when the session ends. Write it to a durable artifact (file, issue, doc).\n\n6. **Blame-oriented language.** Post-mortems are blameless by design. Describe what *happened*, not who *messed up*. This is especially important when the \"who\" is an agent — focus on the decision and the information available at the time.\n\n7. **Retrying before forensics.** The whole point is to analyze before retrying. If you retry first, you lose the original failure state and may introduce changes that mask the real cause.\n\n## Verification Checklist\n\n- [ ] Failure categorized into one of the six taxonomy categories\n- [ ] Timeline reconstructed (either via script or manually) with timestamps/ordering\n- [ ] Causal chain traced backward to a root cause or identified as systemic\n- [ ] Post-mortem report filled from the template\n- [ ] Report saved to a durable location\n- [ ] At least one concrete, verifiable action item identified\n- [ ] Lessons learned phrased as generalizable insights, not task-specific notes\n\nFile v0.1.1:README.md\n\n# Failure Forensics\n\n> A structured post-mortem analysis skill for AI agents. When a task fails, don't just retry — investigate.\n\n[![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE)\n\n## What It Does\n\n**Failure Forensics** is a skill for AI agent frameworks (Hermes Agent, OpenClaw, and compatible). When an agent task fails, instead of blindly retrying, the agent performs a four-phase structured root cause analysis:\n\n1. **Triage** — Categorize the failure (network, permissions, logic, environment, dependency, resource)\n2. **Timeline Reconstruction** — Parse tool-call logs to build a chronological failure timeline\n3. **Causal Chain Analysis** — Trace the decision chain backward to the root cause\n4. **Post-Mortem Report** — Generate a structured report and save lessons learned\n\n## Why\n\nRetrying a failed task without understanding why it failed is a gamble. You might:\n- Hit the same failure again (wasted effort)\n- Mask the real cause with incidental changes (harder to debug later)\n- Miss a systemic issue that will recur in different forms\n\nFailure Forensics turns each failure into a reusable lesson, building institutional memory that makes the agent more reliable over time.\n\n## Repository Structure\n\n```\nfailure-forensics/\n├── SKILL.md                          # Main skill definition (YAML frontmatter + workflow)\n├── README.md                         # This file\n├── LICENSE                           # MIT\n├── references/\n│   ├── failure-taxonomy.md           # Six-category failure taxonomy with signatures\n│   └── post-mortem-template.md       # Fill-in-the-blanks report template\n└── scripts/\n    └── failure_forensics.py          # Log parser, categorizer, report generator\n```\n\n## Quick Start\n\n### As a Hermes Agent Skill\n\nCopy or symlink this directory to your skills folder:\n\n```bash\ncp -r failure-forensics/ ~/.hermes/skills/\n```\n\nThe skill auto-loads. When a task fails, the agent will follow the forensics workflow described in `SKILL.md`.\n\n### Standalone (Script Only)\n\nThe Python script works independently — no agent required:\n\n```bash\n# Analyze a JSONL log of tool calls\npython3 scripts/failure_forensics.py analyze --log session.jsonl --output timeline.md\n\n# Categorize an error message\npython3 scripts/failure_forensics.py categorize --error \"ConnectionRefusedError: Connection refused\"\n\n# Generate a pre-filled post-mortem report\npython3 scripts/failure_forensics.py report --log session.jsonl --title \"Deploy failure\"\n```\n\n### Log Format\n\nThe analyzer reads JSON or JSONL files where each entry represents a tool call:\n\n```json\n{\n  \"timestamp\": \"2024-01-15T10:23:45Z\",\n  \"tool\": \"terminal\",\n  \"args\": {\"command\": \"npm install\"},\n  \"result\": {\"success\": false, \"error\": \"EACCES: permission denied\"},\n  \"duration_ms\": 1200\n}\n```\n\nSee `scripts/sample_log.jsonl` for a working example.\n\n## Failure Taxonomy (Summary)\n\n| Category | Signature | Example |\n|---|---|---|\n| **Network** | Connection refused, timeout, DNS, TLS | `curl: (7) Failed to connect` |\n| **Permissions** | 401/403, EACCES, unauthorized | `PermissionError: [Errno 13]` |\n| **Logic** | Wrong output, assertion failure | `AssertionError: expected 200, got 404` |\n| **Environment** | Missing binary, wrong version, missing env | `command not found: docker` |\n| **Dependency** | ImportError, version conflict | `ModuleNotFoundError: No module named 'foo'` |\n| **Resource** | OOM, disk full, rate limit | `OSError: [Errno 28] No space left` |\n\nFull details in [`references/failure-taxonomy.md`](references/failure-taxonomy.md).\n\n## Requirements\n\n- Python 3.8+\n- No external dependencies (stdlib only)\n\n## License\n\nMIT — see [LICENSE](LICENSE).\n\n## Author\n\n**Denis Voronin** — [voronindenis5@gmail.com](mailto:voronindenis5@gmail.com)\n\nFile v0.1.1:_meta.json\n\n{\n  \"ownerId\": \"kn75wwn4x6djaf28jbykeamazd81gtdp\",\n  \"slug\": \"failure-forensics\",\n  \"version\": \"0.1.1\",\n  \"publishedAt\": 1786449534013\n}\n\nFile v0.1.1:references/failure-taxonomy.md\n\n# Failure Taxonomy\n\nA reference for categorizing failures during Phase 1 (Triage) of the forensics workflow.\n\n## How to Use This Taxonomy\n\n1. Read the error message / failure signature.\n2. Match it against the patterns below.\n3. If multiple categories match, pick the **most specific** one. A `ModuleNotFoundError` is a *dependency* failure, not an *environment* failure, even though both relate to the system.\n4. If no category fits cleanly, record it as **uncategorized** and note the novel pattern. The taxonomy grows by accretion.\n\n---\n\n## 1. Network Failures\n\n**Core question:** Is the endpoint reachable *right now*, and from this environment?\n\n### Signatures\n\n| Pattern | Meaning |\n|---|---|\n| `Connection refused`, `ConnectionRefusedError` | Port open but nothing listening / rejected |\n| `Connection timed out`, `ETIMEDOUT` | Packet dropped, firewall, or host unreachable |\n| `Name or service not known`, `NXDOMAIN` | DNS resolution failure |\n| `SSL: CERTIFICATE_VERIFY_FAILED` | TLS cert expired, self-signed, or MITM |\n| `HTTP 502 Bad Gateway`, `503 Service Unavailable`, `504 Gateway Timeout` | Server-side failure |\n| `curl: (7) Failed to connect`, `curl: (28) Connection timed out` | CLI-level network failure |\n| `ECONNRESET`, `Connection reset by peer` | Remote end dropped the connection |\n\n### Diagnostic Questions\n\n- Does `curl -v <url>` or `nc -zv <host> <port>` work from the same environment?\n- Is this an internal vs. external endpoint? (internal may need VPN/peering)\n- Is there a proxy or corporate firewall in play?\n- Did this work before? What changed? (network config, DNS, certs)\n\n### Common Root Causes\n\n- Service is down or not started\n- Wrong port number (e.g., `:443` vs `:80`)\n- DNS misconfiguration or stale cache\n- Expired TLS certificate\n- Firewall / security group blocking the port\n- IPv6 vs IPv4 resolution mismatch\n\n---\n\n## 2. Permissions Failures\n\n**Core question:** Does the credential/token/user have the needed scope for this action?\n\n### Signatures\n\n| Pattern | Meaning |\n|---|---|\n| `401 Unauthorized`, `HTTP 401` | No credentials, or credentials rejected |\n| `403 Forbidden`, `HTTP 403` | Credentials valid, but lack permission |\n| `Permission denied`, `EACCES`, `PermissionError` | Filesystem permission denied |\n| `Access denied`, `UnauthorizedAccess` | Cloud API / IAM denial |\n| `insufficient privileges`, `requires elevated permissions` | OS-level privilege denial |\n| `invalid token`, `token expired`, `invalid_grant` | Auth token problem |\n\n### Diagnostic Questions\n\n- What user/service account is the agent running as?\n- What scopes/roles does the token have? (check the token's claims, not assumptions)\n- Is this a filesystem permission issue (check `ls -la`, `id`, `getfacl`)?\n- Is this an API/IAM issue (check the service's permission model)?\n- Did the token expire? Check issuance and expiry timestamps.\n\n### Common Root Causes\n\n- Token expired and wasn't refreshed\n- Token has correct identity but wrong scope/role\n- File owned by a different user; agent running as non-root\n- sudo / privilege escalation required but not available\n- IAM policy missing a specific action (e.g., `s3:GetObject` but not `s3:PutObject`)\n\n---\n\n## 3. Logic Failures\n\n**Core question:** What assumption did the code make that turned out to be false?\n\n### Signatures\n\n| Pattern | Meaning |\n|---|---|\n| `AssertionError` | Explicit assertion violated |\n| `TypeError`, `ValueError`, `KeyError` | Unexpected data shape or type |\n| Output is wrong (no exception) | Silent logic error |\n| `IndexError`, `AttributeError` | Assumed structure that doesn't exist |\n| Unexpected None / null | Missing value not handled |\n| Off-by-one, race condition | Classic algorithmic bugs |\n\n### Diagnostic Questions\n\n- What did the code *expect* the input/output to be? What was it actually?\n- Is this a data-dependent failure? (works on test data, fails on real data)\n- Is there an implicit assumption about ordering, types, or state?\n- Did the logic work before? What input changed?\n\n### Common Root Causes\n\n- Assumed data shape that the real data doesn't match (e.g., missing optional field)\n- Assumed a side effect completed (e.g., file written) when it hadn't\n- Assumed an operation was atomic when it wasn't (race condition)\n- Hardcoded value that was correct in one environment but not another\n- Logic that handles the happy path but not edge cases (empty list, None, concurrent access)\n\n---\n\n## 4. Environment Failures\n\n**Core question:** What does the runtime environment look like vs. what the task assumed?\n\n### Signatures\n\n| Pattern | Meaning |\n|---|---|\n| `command not found`, `No such file or directory` | Binary/tool not installed or not on PATH |\n| `env: ‘FOO’: No such file or directory` | Missing required environment variable |\n| Wrong version output | Binary present but wrong version |\n| `uname` mismatch | Wrong OS or architecture |\n| `too large section header offset` | Binary compiled for different arch |\n| `/usr/bin/python3: No module named pip` | Toolchain component missing |\n\n### Diagnostic Questions\n\n- `which <tool>` / `command -v <tool>` — is the binary on PATH?\n- `<tool> --version` — is it the expected version?\n- `echo $VAR` — are required env vars set?\n- `uname -a` — correct OS and architecture?\n- Is this running in a container, VM, or bare metal? What's the base image?\n\n### Common Root Causes\n\n- Tool installed in a different environment (e.g., dev machine but not CI)\n- PATH doesn't include the install directory\n- Wrong Docker base image\n- Missing environment variable that was set in `.bashrc` but not in the agent's shell\n- Architecture mismatch (e.g., arm64 binary on x86_64)\n\n---\n\n## 5. Dependency Failures\n\n**Core question:** What changed in the dependency graph?\n\n### Signatures\n\n| Pattern | Meaning |\n|---|---|\n| `ModuleNotFoundError`, `ImportError` | Python package not installed |\n| `Cannot find module`, `MODULE_NOT_FOUND` | Node.js package not installed |\n| `NoClassDefFoundError`, `ClassNotFoundException` | Java class not on classpath |\n| `pkg: unresolved dependency`, version conflict | Conflicting version requirements |\n| `ABI mismatch`, `undefined symbol` | Compiled extension built for wrong version |\n| `Package 'foo' not found` | Package not available in the repo |\n\n### Diagnostic Questions\n\n- `pip list` / `npm ls` / `gem list` — is the package installed?\n- Is there a `requirements.txt` / `package-lock.json` / `Gemfile.lock` that pins versions?\n- Was a dependency recently upgraded? Check `pip list --outdated` or git diff on lock files.\n- Are there conflicting version requirements from different packages?\n\n### Common Root Causes\n\n- Package not installed in the active virtualenv / node_modules\n- Version conflict: package A needs `lib>=2.0`, package B needs `lib<2.0`\n- Transitive dependency changed without updating lock file\n- Package installed for Python 3.11 but running on 3.9\n- Compiled C-extension built against a different version of the shared library\n\n---\n\n## 6. Resource Failures\n\n**Core question:** What was the ceiling, and what hit it?\n\n### Signatures\n\n| Pattern | Meaning |\n|---|---|\n| `Out of memory`, `OOMKilled`, `MemoryError` | RAM exhausted |\n| `No space left on device`, `ENOSPC` | Disk full |\n| `Too many open files`, `EMFILE`, `ENFILE` | File descriptor limit |\n| `429 Too Many Requests`, rate limit headers | API rate limit |\n| `Quota exceeded` | Cloud quota / billing limit |\n| `Cannot allocate memory` | Fork/thread allocation failure |\n| `SIGKILL` (exit code 137) | Process killed (often OOM killer) |\n\n### Diagnostic Questions\n\n- `free -h` / `df -h` — current memory and disk state\n- `ulimit -n` / `cat /proc/<pid>/limits` — file descriptor limit\n- For API limits: check the `X-RateLimit-Remaining` and `X-RateLimit-Reset` headers\n- Is this a spike or a steady-state exhaustion?\n- Was a resource leaked? (file handles not closed, memory not freed)\n\n### Common Root Causes\n\n- Memory leak (growing RSS over time)\n- Unbounded queue / cache without eviction\n- Processing a file or dataset larger than expected\n- File descriptor leak (opened connections not closed)\n- Too many concurrent operations hitting an API rate limit\n- Log file filling the disk\n- Cloud account hit a soft quota (e.g., max instances)\n\n---\n\n## Edge Cases and Overlaps\n\n### Network vs. Permissions\n\nAn HTTP 401/403 from an API *looks* like a network call but is a **permissions** failure. The network worked fine — the server responded. The issue is authorization.\n\n**Rule:** If the server responded at all (even with an error code), the network layer succeeded. Classify by the *meaning* of the response, not the fact that a request was made.\n\n### Environment vs. Dependency\n\n`command not found: python3` is an **environment** failure (Python itself isn't installed). `ModuleNotFoundError: No module named 'requests'` is a **dependency** failure (Python is there, but a package isn't).\n\n**Rule:** If the *runtime* (interpreter, package manager) is missing, it's environment. If the runtime exists but a *package* within it is missing, it's dependency.\n\n### Logic vs. Environment\n\nCode produces wrong output. Is the code buggy (logic) or is it running in an environment where an assumption doesn't hold (environment)?\n\n**Rule:** If the same code produces correct output in another environment, it's likely environment. If it's wrong everywhere, it's logic. When unsure, test in a clean environment first.\n\n### Resource vs. Network\n\nA request times out. Is it network latency or the server being overloaded?\n\n**Rule:** Check server-side metrics if available. A timeout under normal network conditions often indicates server-side resource exhaustion (a resource failure on the *server*, even though it manifests as a network failure on the *client*).\n\n---\n\n## Adding to the Taxonomy\n\nWhen you encounter a failure that doesn't fit any category:\n\n1. Document it with its signature and diagnostic questions.\n2. Check if it's genuinely new or an unmapped edge case of an existing category.\n3. If new, add it as a sub-category or propose a new top-level category.\n4. Update this file and note the addition in the post-mortem.\n\nThe taxonomy is not exhaustive by design — it's a living document that grows with experience.\n\nFile v0.1.1:references/post-mortem-template.md\n\n# Post-Mortem Report Template\n\nFill in every section. If a section doesn't apply, write \"N/A — [reason]\" rather than deleting it. A blank section is information; a missing section is ambiguity.\n\n---\n\n# Post-Mortem: [TITLE]\n\n**Date:** YYYY-MM-DD\n**Author:** [agent name / human name]\n**Task:** [one-line description of what the agent was trying to do]\n**Status:** [Failed / Partially completed / Recovered after intervention]\n\n## Summary\n\n[One paragraph, plain language. Describe what happened, not just that it failed. A reader who wasn't present should understand the failure and its impact from this paragraph alone. Aim for 3-5 sentences.]\n\n**Failure category:** [network / permissions / logic / environment / dependency / resource / uncategorized]\n\n## Timeline\n\nReconstruct the sequence of events leading to the failure. Use timestamps from logs where available. Mark the failure point explicitly with **[FAILURE]**.\n\n| Time (UTC) | Event | Outcome |\n|---|---|---|\n| 10:23:01 | Agent received task: \"Deploy service to staging\" | Task started |\n| 10:23:15 | Agent ran `git pull origin main` | Success |\n| 10:23:45 | Agent ran `npm install` | **[FAILURE]** — EACCES: permission denied |\n| 10:24:02 | Agent retried `npm install` with `sudo` | Different error: EACCES on different path |\n| 10:24:30 | Agent abandoned task | Task failed |\n\nIf using the forensics script, paste the generated timeline here.\n\n## Impact\n\n- **What was affected:** [services, data, users, downstream tasks]\n- **Severity:** [low / medium / high / critical]\n- **Duration of impact:** [how long the system was in a bad state, if applicable]\n- **Data loss:** [yes/no — if yes, what and how much]\n- **Recovery actions taken:** [what was done to restore service, if anything]\n\n## Root Cause\n\n[The terminal link of the causal chain. State this plainly and specifically. This should be a single, clear sentence that explains *why* the failure happened at the deepest level you could trace.]\n\n**Example (bad):** \"npm install failed.\"\n**Example (good):** \"The agent ran as a non-root user in a container where the global npm directory (`/usr/lib/node_modules`) was owned by root with no write permission for others, and `npm install` without `--prefix` defaults to global installation.\"\n\n## Causal Chain\n\nTrace backward from the failure point. Each entry should answer \"why did the previous step happen/ matter?\"\n\n1. **[FAILURE]** `npm install` returned EACCES on `/usr/lib/node_modules`\n2. **Because:** npm attempted a global install (no `--prefix` or local `package.json`)\n3. **Because:** The agent assumed the install target was local, but the working directory had no `package.json`\n4. **Because:** The agent didn't verify the working directory contents before running the install\n5. **Because:** The task description referenced a project at a path the agent assumed existed without checking ← **ROOT CAUSE**\n\n**Root cause:** The agent operated on an unverified assumption about the filesystem state (project path) and cascaded into a permissions failure that looked like an environment problem.\n\n## Contributing Factors\n\nFactors that didn't *cause* the failure but made it worse or harder to diagnose:\n\n- **Retry without analysis:** The agent retried with `sudo` before understanding the error, introducing a new failure mode and obscuring the original cause.\n- **Poor error context:** npm's error message mentioned the path but not that it was a global vs. local install distinction.\n- **No pre-flight check:** No step verified the working directory contained the expected project files.\n\n## What Went Well\n\n[Post-mortems that only list problems create a blame culture. Note what worked — fast detection, good logging, graceful degradation, etc.]\n\n- The agent correctly abandoned the task after two failures rather than continuing to cascade.\n- Tool-call logs captured the full error messages, enabling this analysis.\n\n## Action Items\n\nEach item must be **specific, assigned, and verifiable**.\n\n| # | Action | Owner | Verification | Priority |\n|---|---|---|---|---|\n| 1 | Add a pre-flight check: verify `package.json` exists in the working directory before running `npm install` | agent framework team | Unit test: `test_npm_install_requires_package_json` passes | High |\n| 2 | Add `--prefix` flag to npm install commands by default, or detect global vs. local context | agent framework team | Manual: run in a dir without package.json, confirm error is actionable | Medium |\n| 3 | Add retry-with-analysis rule to agent: after first failure, perform Phase 1 triage before retrying | this skill | Verify `failure-forensics` skill is loaded and triggered on retry | High |\n\n## Lessons Learned\n\nGeneralizable insights. These are the durable output of the post-mortem — they should apply beyond this specific incident.\n\n1. **Verify assumptions about filesystem state before acting.** \"The project is at this path\" is an assumption, not a fact. `ls` or `test -f` is cheap; cascading failures are expensive.\n2. **Retry is not a debugging strategy.** Retrying without analysis can introduce new failure modes and destroy evidence of the original cause.\n3. **Permission errors often mask environment/context errors.** A permissions failure at the symptom level may have a logic or assumption failure at the root.\n4. **Error messages rarely point at the root cause directly.** They point at the *symptom*. Always read them as clues, not diagnoses.\n\n## Appendix\n\n### Full Error Output\n\n```\n[Paste the complete error message / stack trace / log output here]\n```\n\n### Environment\n\n- **OS:** [e.g., Ubuntu 22.04 x86_64]\n- **Agent runtime:** [e.g., Hermes Agent v2.3]\n- **Key dependencies:** [versions of relevant tools]\n- **Working directory:** [path]\n\n### References\n\n- [Links to related issues, PRs, prior post-mortems, docs]\n\nFile v0.1.1:skill-card.md\n\n## Description:\n\nUse when an agent task fails or produces unexpected results. Performs structured post-mortem root cause analysis: categorizes the failure, traces the exact failure point through tool-call logs, reconstructs the decision chain, generates a post-mortem report, and saves lessons to prevent recurrence.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[voronindenis5](https://clawhub.ai/user/voronindenis5)\n\n### License/Terms of Use:\n\nMIT\n\n## Use Case:\n\nDevelopers and engineers use this skill after an agent task fails to classify the failure, reconstruct the tool-call timeline, trace the causal chain, and produce a post-mortem report with reusable lessons.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Failure logs and post-mortem reports may contain credentials, private paths, customer data, or confidential prompts.\n\nMitigation: Analyze only selected logs, redact sensitive values before processing or saving reports, and store generated reports in approved access-controlled locations.\n\nRisk: Markdown generated from external or attacker-influenced logs may include misleading content.\n\nMitigation: Treat generated Markdown as untrusted until reviewed, and do not execute copied commands or follow links from reports without validation.\n\n## Reference(s):\n\n- [Failure taxonomy](references/failure-taxonomy.md)\n- [Post-mortem report template](references/post-mortem-template.md)\n- [Source repository](https://github.com/voronindenis5/failure-forensics)\n- [ClawHub skill page](https://clawhub.ai/voronindenis5/skills/failure-forensics)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, shell commands, guidance]\n\n**Output Format:** [Markdown reports, text classifications, and command-line guidance]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Can analyze selected JSON or JSONL tool-call logs and write timeline or post-mortem Markdown files.]\n\n## Skill Version(s):\n\n0.1.1 (source: ClawHub release evidence; source frontmatter reports 1.0.0)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v0.1.1:LICENSE\n\nMIT License\n\nCopyright (c) 2026 Denis Voronin\n\nPermission is hereby granted, free of charge, to any person obtaining a copy\nof this software and associated documentation files (the \"Software\"), to deal\nin the Software without restriction, including without limitation the rights\nto use, copy, modify, merge, publish, distribute, sublicense, and/or sell\ncopies of the Software, and to permit persons to whom the Software is\nfurnished to do so, subject to the following conditions:\n\nThe above copyright notice and this permission notice shall be included in all\ncopies or substantial portions of the Software.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR\nIMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,\nFITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE\nAUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER\nLIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,\nOUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE\nSOFTWARE.\n\nArchive v0.1.0: 11 files, 23413 bytes\n\nFiles: LICENSE (1070b), README.md (3824b), references (0b), references/failure-taxonomy.md (10215b), references/post-mortem-template.md (5800b), scripts (0b), scripts/failure_forensics.py (19949b), scripts/sample_log.jsonl (1293b), skill-card.md (2174b), SKILL.md (10646b), _meta.json (136b)\n\nFile v0.1.0:SKILL.md\n\n---\nname: failure-forensics\ndescription: \"Use when an agent task fails or produces unexpected results. Performs structured post-mortem root cause analysis: categorizes the failure, traces the exact failure point through tool-call logs, reconstructs the decision chain, generates a post-mortem report, and saves lessons to prevent recurrence.\"\nversion: 1.0.0\nauthor: Denis Voronin\nlicense: MIT\nmetadata:\n  hermes:\n    tags: [debugging, post-mortem, forensics, root-cause-analysis, failure-analysis, agent-reliability]\n    related_skills: [systematic-debugging, debugging-hermes-tui-commands]\n---\n\n# Failure Forensics\n\n## Overview\n\nWhen an agent task fails, the default response is to retry — hoping for a different outcome. **Failure Forensics** rejects that reflex. Instead, the agent performs structured root cause analysis *before* retrying, treating every failure as evidence to be collected, categorized, and learned from.\n\nThe workflow has four phases:\n\n1. **Triage** — Categorize the failure using the taxonomy in [`references/failure-taxonomy.md`](references/failure-taxonomy.md).\n2. **Timeline Reconstruction** — Parse tool-call logs and agent decision points to build a chronological failure timeline. The script [`scripts/failure_forensics.py`](scripts/failure_forensics.py) automates this from JSON or JSONL log formats.\n3. **Causal Chain Analysis** — Trace the chain of decisions, assumptions, and actions that led from the task kickoff to the failure point. Identify the *root cause*, not just the proximate symptom.\n4. **Post-Mortem Report** — Generate a structured report from the template in [`references/post-mortem-template.md`](references/post-mortem-template.md) and persist it so future sessions can learn.\n\nThis skill turns a single failure into a permanent, reusable lesson.\n\n## When to Use\n\n- **An agent task failed** and retrying without understanding *why* is risky.\n- **A failure recurs** across attempts — you suspect a systemic cause, not bad luck.\n- **You need an artifact** documenting what went wrong for a team review or audit.\n- **A complex multi-step task** partially completed then broke — you need to understand which step is safe to resume from.\n- **You want to improve agent reliability** by building a corpus of past failure patterns.\n\n### Don't use for:\n\n- **Trivial failures with obvious fixes** (typo in a command, missing flag). Fix and move on.\n- **Live debugging** of an actively failing process — use `systematic-debugging` for that. Run forensics *after* the process is dead or the task is abandoned.\n- **Human performance reviews.** This skill analyzes agent + tool behavior, not people.\n\n## The Forensics Workflow\n\n### Phase 1: Triage — Categorize the Failure\n\nRead the full taxonomy in [`references/failure-taxonomy.md`](references/failure-taxonomy.md). At a high level, every failure falls into one of six categories:\n\n| Category | Signature | First Question |\n|---|---|---|\n| **Network** | Connection refused, timeout, DNS, TLS, 5xx HTTP | \"Is the endpoint reachable *right now*?\" |\n| **Permissions** | 401/403, EACCES, \"permission denied\", \"unauthorized\" | \"Does the credential/token have the needed scope?\" |\n| **Logic** | Code runs but output is wrong; assertions fail; data is corrupt | \"What assumption did the code make that was false?\" |\n| **Environment** | Missing binary, wrong version, missing env var, wrong OS | \"What does `env`/`which`/`uname` say vs. what was expected?\" |\n| **Dependency** | ImportError, version conflict, package not found, ABI mismatch | \"What changed in the dependency graph?\" |\n| **Resource** | OOM, disk full, too many open files, rate limit, quota exhausted | \"What was the ceiling, and what hit it?\" |\n\nRecord the category — it determines the questions you ask next.\n\n### Phase 2: Timeline Reconstruction\n\nCollect the evidence:\n\n1. **Tool-call logs.** If the agent session logged tool calls (JSON or JSONL with timestamps, tool name, args, result/error), feed them to the analyzer:\n\n   ```bash\n   python3 scripts/failure_forensics.py analyze \\\n     --log session.jsonl \\\n     --output timeline.md\n   ```\n\n   The script produces a chronological timeline with:\n   - Each tool call, its timestamp, duration, and outcome (success/failure)\n   - The **first failure point** flagged\n   - Error messages and exit codes extracted\n   - A summary of the failure category (inferred from error signatures)\n\n2. **Manual reconstruction.** If no structured logs exist, reconstruct the timeline from memory of the session. List each decision and action in order. Be honest about uncertainty — mark gaps explicitly.\n\n3. **What to capture for each step:**\n   - Timestamp (or relative ordering)\n   - The action or decision taken\n   - The *intent* behind it (what the agent was trying to achieve)\n   - The actual outcome\n   - Any assumption the agent made\n\n### Phase 3: Causal Chain Analysis\n\nThis is the core of forensics. You're looking for the **causal chain** — the sequence where each link made the next failure more likely.\n\nAsk these questions in order:\n\n1. **What was the immediate (proximate) cause of failure?**\n   - The error message, the crash point, the wrong output. This is the *symptom*.\n\n2. **What was the agent doing when it failed?**\n   - The specific tool call or action. What was the goal of that action?\n\n3. **Why did the agent take that action at that point?**\n   - Trace back one decision. What information did the agent have? What did it assume?\n\n4. **Was that assumption valid?**\n   - Check against logs, file contents, environment state. A failed assumption here is a link in the causal chain.\n\n5. **Continue backward** until you reach either:\n   - A **root cause**: a decision or state that, if different, would have prevented the entire failure cascade. Stop here.\n   - The **task kickoff**: if no single root cause emerges, the failure is *systemic* (multiple contributing factors).\n\n6. **Look for contributing factors** that didn't *cause* the failure but made it worse or harder to recover from:\n   - Missing retry logic\n   - Poor error messages that obscured the real problem\n   - Timeouts set too aggressively or too loosely\n   - Insufficient logging that made diagnosis harder\n\n**Anti-pattern: the \"five whys\" that stops at one.** The first \"why\" almost always produces the symptom, not the cause. Keep going. The root cause is usually 3-5 links back.\n\n### Phase 4: Post-Mortem Report\n\nFill out the template in [`references/post-mortem-template.md`](references/post-mortem-template.md). Key sections:\n\n- **Summary** — one paragraph, plain language. A reader who wasn't there should understand it.\n- **Timeline** — the reconstructed failure timeline from Phase 2.\n- **Root Cause** — the terminal link of the causal chain from Phase 3, stated plainly.\n- **Contributing Factors** — the rest of the chain.\n- **Action Items** — concrete, assigned, verifiable. \"Add retry logic\" is bad; \"Add exponential backoff retry (max 3 attempts) to the `web_extract` call in `session.py:142`\" is good.\n- **Lessons Learned** — generalizable insights. These are the durable output.\n\n**Save the report.** Write it to a persistent location (e.g., a `post-mortems/` directory, an issue tracker, or a knowledge base). A post-mortem that isn't saved didn't happen.\n\n## Using the Forensics Script\n\nThe script `scripts/failure_forensics.py` has three subcommands:\n\n### `analyze` — Build a failure timeline from logs\n\n```bash\n# JSONL log (one JSON object per line)\npython3 scripts/failure_forensics.py analyze --log session.jsonl --format jsonl\n\n# JSON array\npython3 scripts/failure_forensics.py analyze --log session.json --format json\n\n# Write report to file\npython3 scripts/failure_forensics.py analyze --log session.jsonl --output report.md\n```\n\n### `categorize` — Classify an error message\n\n```bash\npython3 scripts/failure_forensics.py categorize --error \"ConnectionRefusedError: [Errno 111] Connection refused\"\n# Output: network\n\npython3 scripts/failure_forensics.py categorize --error \"PermissionError: [Errno 13] Permission denied\"\n# Output: permissions\n```\n\n### `report` — Generate a post-mortem template pre-filled with timeline data\n\n```bash\npython3 scripts/failure_forensics.py report --log session.jsonl --title \"Deploy failure 2024-01-15\" --author \"agent\"\n```\n\n### Expected Log Format\n\nThe analyzer accepts JSON/JSONL where each entry is a tool call record:\n\n```json\n{\n  \"timestamp\": \"2024-01-15T10:23:45Z\",\n  \"tool\": \"terminal\",\n  \"args\": {\"command\": \"npm install\"},\n  \"result\": {\"success\": false, \"error\": \"EACCES: permission denied, open '/usr/lib/node_modules'\"},\n  \"duration_ms\": 1200\n}\n```\n\nRequired fields: `timestamp` (ISO 8601), `tool`. The script tolerates missing optional fields (`result`, `duration_ms`, `args`).\n\n## Common Pitfalls\n\n1. **Confusing symptom with cause.** \"The deploy failed because the build failed\" is a symptom. \"The build failed because `package-lock.json` was regenerated with a different Node version than CI\" is a cause. Keep digging past the symptom.\n\n2. **Stopping at human error.** \"I made a mistake\" is never a root cause. Ask *why* the mistake was possible — missing validation? Misleading documentation? Fatigue from context switching? Fix the system, not the human.\n\n3. **Writing action items that aren't actionable.** \"Be more careful\" is worthless. Every action item must specify *what* to change, *where*, and *how to verify* it works.\n\n4. **Skipping the timeline.** Without a chronological timeline, causal chain analysis becomes guessing. Build the timeline first, even if it's rough.\n\n5. **Not saving the post-mortem.** A post-mortem kept in chat history is lost when the session ends. Write it to a durable artifact (file, issue, doc).\n\n6. **Blame-oriented language.** Post-mortems are blameless by design. Describe what *happened*, not who *messed up*. This is especially important when the \"who\" is an agent — focus on the decision and the information available at the time.\n\n7. **Retrying before forensics.** The whole point is to analyze before retrying. If you retry first, you lose the original failure state and may introduce changes that mask the real cause.\n\n## Verification Checklist\n\n- [ ] Failure categorized into one of the six taxonomy categories\n- [ ] Timeline reconstructed (either via script or manually) with timestamps/ordering\n- [ ] Causal chain traced backward to a root cause or identified as systemic\n- [ ] Post-mortem report filled from the template\n- [ ] Report saved to a durable location\n- [ ] At least one concrete, verifiable action item identified\n- [ ] Lessons learned phrased as generalizable insights, not task-specific notes\n\nFile v0.1.0:README.md\n\n# Failure Forensics\n\n> A structured post-mortem analysis skill for AI agents. When a task fails, don't just retry — investigate.\n\n[![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE)\n\n## What It Does\n\n**Failure Forensics** is a skill for AI agent frameworks (Hermes Agent, OpenClaw, and compatible). When an agent task fails, instead of blindly retrying, the agent performs a four-phase structured root cause analysis:\n\n1. **Triage** — Categorize the failure (network, permissions, logic, environment, dependency, resource)\n2. **Timeline Reconstruction** — Parse tool-call logs to build a chronological failure timeline\n3. **Causal Chain Analysis** — Trace the decision chain backward to the root cause\n4. **Post-Mortem Report** — Generate a structured report and save lessons learned\n\n## Why\n\nRetrying a failed task without understanding why it failed is a gamble. You might:\n- Hit the same failure again (wasted effort)\n- Mask the real cause with incidental changes (harder to debug later)\n- Miss a systemic issue that will recur in different forms\n\nFailure Forensics turns each failure into a reusable lesson, building institutional memory that makes the agent more reliable over time.\n\n## Repository Structure\n\n```\nfailure-forensics/\n├── SKILL.md                          # Main skill definition (YAML frontmatter + workflow)\n├── README.md                         # This file\n├── LICENSE                           # MIT\n├── references/\n│   ├── failure-taxonomy.md           # Six-category failure taxonomy with signatures\n│   └── post-mortem-template.md       # Fill-in-the-blanks report template\n└── scripts/\n    └── failure_forensics.py          # Log parser, categorizer, report generator\n```\n\n## Quick Start\n\n### As a Hermes Agent Skill\n\nCopy or symlink this directory to your skills folder:\n\n```bash\ncp -r failure-forensics/ ~/.hermes/skills/\n```\n\nThe skill auto-loads. When a task fails, the agent will follow the forensics workflow described in `SKILL.md`.\n\n### Standalone (Script Only)\n\nThe Python script works independently — no agent required:\n\n```bash\n# Analyze a JSONL log of tool calls\npython3 scripts/failure_forensics.py analyze --log session.jsonl --output timeline.md\n\n# Categorize an error message\npython3 scripts/failure_forensics.py categorize --error \"ConnectionRefusedError: Connection refused\"\n\n# Generate a pre-filled post-mortem report\npython3 scripts/failure_forensics.py report --log session.jsonl --title \"Deploy failure\"\n```\n\n### Log Format\n\nThe analyzer reads JSON or JSONL files where each entry represents a tool call:\n\n```json\n{\n  \"timestamp\": \"2024-01-15T10:23:45Z\",\n  \"tool\": \"terminal\",\n  \"args\": {\"command\": \"npm install\"},\n  \"result\": {\"success\": false, \"error\": \"EACCES: permission denied\"},\n  \"duration_ms\": 1200\n}\n```\n\nSee `scripts/sample_log.jsonl` for a working example.\n\n## Failure Taxonomy (Summary)\n\n| Category | Signature | Example |\n|---|---|---|\n| **Network** | Connection refused, timeout, DNS, TLS | `curl: (7) Failed to connect` |\n| **Permissions** | 401/403, EACCES, unauthorized | `PermissionError: [Errno 13]` |\n| **Logic** | Wrong output, assertion failure | `AssertionError: expected 200, got 404` |\n| **Environment** | Missing binary, wrong version, missing env | `command not found: docker` |\n| **Dependency** | ImportError, version conflict | `ModuleNotFoundError: No module named 'foo'` |\n| **Resource** | OOM, disk full, rate limit | `OSError: [Errno 28] No space left` |\n\nFull details in [`references/failure-taxonomy.md`](references/failure-taxonomy.md).\n\n## Requirements\n\n- Python 3.8+\n- No external dependencies (stdlib only)\n\n## License\n\nMIT — see [LICENSE](LICENSE).\n\n## Author\n\n**Denis Voronin** — [voronindenis5@gmail.com](mailto:voronindenis5@gmail.com)\n\nFile v0.1.0:_meta.json\n\n{\n  \"ownerId\": \"kn75wwn4x6djaf28jbykeamazd81gtdp\",\n  \"slug\": \"failure-forensics\",\n  \"version\": \"0.1.0\",\n  \"publishedAt\": 1785959381304\n}\n\nFile v0.1.0:references/failure-taxonomy.md\n\n# Failure Taxonomy\n\nA reference for categorizing failures during Phase 1 (Triage) of the forensics workflow.\n\n## How to Use This Taxonomy\n\n1. Read the error message / failure signature.\n2. Match it against the patterns below.\n3. If multiple categories match, pick the **most specific** one. A `ModuleNotFoundError` is a *dependency* failure, not an *environment* failure, even though both relate to the system.\n4. If no category fits cleanly, record it as **uncategorized** and note the novel pattern. The taxonomy grows by accretion.\n\n---\n\n## 1. Network Failures\n\n**Core question:** Is the endpoint reachable *right now*, and from this environment?\n\n### Signatures\n\n| Pattern | Meaning |\n|---|---|\n| `Connection refused`, `ConnectionRefusedError` | Port open but nothing listening / rejected |\n| `Connection timed out`, `ETIMEDOUT` | Packet dropped, firewall, or host unreachable |\n| `Name or service not known`, `NXDOMAIN` | DNS resolution failure |\n| `SSL: CERTIFICATE_VERIFY_FAILED` | TLS cert expired, self-signed, or MITM |\n| `HTTP 502 Bad Gateway`, `503 Service Unavailable`, `504 Gateway Timeout` | Server-side failure |\n| `curl: (7) Failed to connect`, `curl: (28) Connection timed out` | CLI-level network failure |\n| `ECONNRESET`, `Connection reset by peer` | Remote end dropped the connection |\n\n### Diagnostic Questions\n\n- Does `curl -v <url>` or `nc -zv <host> <port>` work from the same environment?\n- Is this an internal vs. external endpoint? (internal may need VPN/peering)\n- Is there a proxy or corporate firewall in play?\n- Did this work before? What changed? (network config, DNS, certs)\n\n### Common Root Causes\n\n- Service is down or not started\n- Wrong port number (e.g., `:443` vs `:80`)\n- DNS misconfiguration or stale cache\n- Expired TLS certificate\n- Firewall / security group blocking the port\n- IPv6 vs IPv4 resolution mismatch\n\n---\n\n## 2. Permissions Failures\n\n**Core question:** Does the credential/token/user have the needed scope for this action?\n\n### Signatures\n\n| Pattern | Meaning |\n|---|---|\n| `401 Unauthorized`, `HTTP 401` | No credentials, or credentials rejected |\n| `403 Forbidden`, `HTTP 403` | Credentials valid, but lack permission |\n| `Permission denied`, `EACCES`, `PermissionError` | Filesystem permission denied |\n| `Access denied`, `UnauthorizedAccess` | Cloud API / IAM denial |\n| `insufficient privileges`, `requires elevated permissions` | OS-level privilege denial |\n| `invalid token`, `token expired`, `invalid_grant` | Auth token problem |\n\n### Diagnostic Questions\n\n- What user/service account is the agent running as?\n- What scopes/roles does the token have? (check the token's claims, not assumptions)\n- Is this a filesystem permission issue (check `ls -la`, `id`, `getfacl`)?\n- Is this an API/IAM issue (check the service's permission model)?\n- Did the token expire? Check issuance and expiry timestamps.\n\n### Common Root Causes\n\n- Token expired and wasn't refreshed\n- Token has correct identity but wrong scope/role\n- File owned by a different user; agent running as non-root\n- sudo / privilege escalation required but not available\n- IAM policy missing a specific action (e.g., `s3:GetObject` but not `s3:PutObject`)\n\n---\n\n## 3. Logic Failures\n\n**Core question:** What assumption did the code make that turned out to be false?\n\n### Signatures\n\n| Pattern | Meaning |\n|---|---|\n| `AssertionError` | Explicit assertion violated |\n| `TypeError`, `ValueError`, `KeyError` | Unexpected data shape or type |\n| Output is wrong (no exception) | Silent logic error |\n| `IndexError`, `AttributeError` | Assumed structure that doesn't exist |\n| Unexpected None / null | Missing value not handled |\n| Off-by-one, race condition | Classic algorithmic bugs |\n\n### Diagnostic Questions\n\n- What did the code *expect* the input/output to be? What was it actually?\n- Is this a data-dependent failure? (works on test data, fails on real data)\n- Is there an implicit assumption about ordering, types, or state?\n- Did the logic work before? What input changed?\n\n### Common Root Causes\n\n- Assumed data shape that the real data doesn't match (e.g., missing optional field)\n- Assumed a side effect completed (e.g., file written) when it hadn't\n- Assumed an operation was atomic when it wasn't (race condition)\n- Hardcoded value that was correct in one environment but not another\n- Logic that handles the happy path but not edge cases (empty list, None, concurrent access)\n\n---\n\n## 4. Environment Failures\n\n**Core question:** What does the runtime environment look like vs. what the task assumed?\n\n### Signatures\n\n| Pattern | Meaning |\n|---|---|\n| `command not found`, `No such file or directory` | Binary/tool not installed or not on PATH |\n| `env: ‘FOO’: No such file or directory` | Missing required environment variable |\n| Wrong version output | Binary present but wrong version |\n| `uname` mismatch | Wrong OS or architecture |\n| `too large section header offset` | Binary compiled for different arch |\n| `/usr/bin/python3: No module named pip` | Toolchain component missing |\n\n### Diagnostic Questions\n\n- `which <tool>` / `command -v <tool>` — is the binary on PATH?\n- `<tool> --version` — is it the expected version?\n- `echo $VAR` — are required env vars set?\n- `uname -a` — correct OS and architecture?\n- Is this running in a container, VM, or bare metal? What's the base image?\n\n### Common Root Causes\n\n- Tool installed in a different environment (e.g., dev machine but not CI)\n- PATH doesn't include the install directory\n- Wrong Docker base image\n- Missing environment variable that was set in `.bashrc` but not in the agent's shell\n- Architecture mismatch (e.g., arm64 binary on x86_64)\n\n---\n\n## 5. Dependency Failures\n\n**Core question:** What changed in the dependency graph?\n\n### Signatures\n\n| Pattern | Meaning |\n|---|---|\n| `ModuleNotFoundError`, `ImportError` | Python package not installed |\n| `Cannot find module`, `MODULE_NOT_FOUND` | Node.js package not installed |\n| `NoClassDefFoundError`, `ClassNotFoundException` | Java class not on classpath |\n| `pkg: unresolved dependency`, version conflict | Conflicting version requirements |\n| `ABI mismatch`, `undefined symbol` | Compiled extension built for wrong version |\n| `Package 'foo' not found` | Package not available in the repo |\n\n### Diagnostic Questions\n\n- `pip list` / `npm ls` / `gem list` — is the package installed?\n- Is there a `requirements.txt` / `package-lock.json` / `Gemfile.lock` that pins versions?\n- Was a dependency recently upgraded? Check `pip list --outdated` or git diff on lock files.\n- Are there conflicting version requirements from different packages?\n\n### Common Root Causes\n\n- Package not installed in the active virtualenv / node_modules\n- Version conflict: package A needs `lib>=2.0`, package B needs `lib<2.0`\n- Transitive dependency changed without updating lock file\n- Package installed for Python 3.11 but running on 3.9\n- Compiled C-extension built against a different version of the shared library\n\n---\n\n## 6. Resource Failures\n\n**Core question:** What was the ceiling, and what hit it?\n\n### Signatures\n\n| Pattern | Meaning |\n|---|---|\n| `Out of memory`, `OOMKilled`, `MemoryError` | RAM exhausted |\n| `No space left on device`, `ENOSPC` | Disk full |\n| `Too many open files`, `EMFILE`, `ENFILE` | File descriptor limit |\n| `429 Too Many Requests`, rate limit headers | API rate limit |\n| `Quota exceeded` | Cloud quota / billing limit |\n| `Cannot allocate memory` | Fork/thread allocation failure |\n| `SIGKILL` (exit code 137) | Process killed (often OOM killer) |\n\n### Diagnostic Questions\n\n- `free -h` / `df -h` — current memory and disk state\n- `ulimit -n` / `cat /proc/<pid>/limits` — file descriptor limit\n- For API limits: check the `X-RateLimit-Remaining` and `X-RateLimit-Reset` headers\n- Is this a spike or a steady-state exhaustion?\n- Was a resource leaked? (file handles not closed, memory not freed)\n\n### Common Root Causes\n\n- Memory leak (growing RSS over time)\n- Unbounded queue / cache without eviction\n- Processing a file or dataset larger than expected\n- File descriptor leak (opened connections not closed)\n- Too many concurrent operations hitting an API rate limit\n- Log file filling the disk\n- Cloud account hit a soft quota (e.g., max instances)\n\n---\n\n## Edge Cases and Overlaps\n\n### Network vs. Permissions\n\nAn HTTP 401/403 from an API *looks* like a network call but is a **permissions** failure. The network worked fine — the server responded. The issue is authorization.\n\n**Rule:** If the server responded at all (even with an error code), the network layer succeeded. Classify by the *meaning* of the response, not the fact that a request was made.\n\n### Environment vs. Dependency\n\n`command not found: python3` is an **environment** failure (Python itself isn't installed). `ModuleNotFoundError: No module named 'requests'` is a **dependency** failure (Python is there, but a package isn't).\n\n**Rule:** If the *runtime* (interpreter, package manager) is missing, it's environment. If the runtime exists but a *package* within it is missing, it's dependency.\n\n### Logic vs. Environment\n\nCode produces wrong output. Is the code buggy (logic) or is it running in an environment where an assumption doesn't hold (environment)?\n\n**Rule:** If the same code produces correct output in another environment, it's likely environment. If it's wrong everywhere, it's logic. When unsure, test in a clean environment first.\n\n### Resource vs. Network\n\nA request times out. Is it network latency or the server being overloaded?\n\n**Rule:** Check server-side metrics if available. A timeout under normal network conditions often indicates server-side resource exhaustion (a resource failure on the *server*, even though it manifests as a network failure on the *client*).\n\n---\n\n## Adding to the Taxonomy\n\nWhen you encounter a failure that doesn't fit any category:\n\n1. Document it with its signature and diagnostic questions.\n2. Check if it's genuinely new or an unmapped edge case of an existing category.\n3. If new, add it as a sub-category or propose a new top-level category.\n4. Update this file and note the addition in the post-mortem.\n\nThe taxonomy is not exhaustive by design — it's a living document that grows with experience.\n\nFile v0.1.0:references/post-mortem-template.md\n\n# Post-Mortem Report Template\n\nFill in every section. If a section doesn't apply, write \"N/A — [reason]\" rather than deleting it. A blank section is information; a missing section is ambiguity.\n\n---\n\n# Post-Mortem: [TITLE]\n\n**Date:** YYYY-MM-DD\n**Author:** [agent name / human name]\n**Task:** [one-line description of what the agent was trying to do]\n**Status:** [Failed / Partially completed / Recovered after intervention]\n\n## Summary\n\n[One paragraph, plain language. Describe what happened, not just that it failed. A reader who wasn't present should understand the failure and its impact from this paragraph alone. Aim for 3-5 sentences.]\n\n**Failure category:** [network / permissions / logic / environment / dependency / resource / uncategorized]\n\n## Timeline\n\nReconstruct the sequence of events leading to the failure. Use timestamps from logs where available. Mark the failure point explicitly with **[FAILURE]**.\n\n| Time (UTC) | Event | Outcome |\n|---|---|---|\n| 10:23:01 | Agent received task: \"Deploy service to staging\" | Task started |\n| 10:23:15 | Agent ran `git pull origin main` | Success |\n| 10:23:45 | Agent ran `npm install` | **[FAILURE]** — EACCES: permission denied |\n| 10:24:02 | Agent retried `npm install` with `sudo` | Different error: EACCES on different path |\n| 10:24:30 | Agent abandoned task | Task failed |\n\nIf using the forensics script, paste the generated timeline here.\n\n## Impact\n\n- **What was affected:** [services, data, users, downstream tasks]\n- **Severity:** [low / medium / high / critical]\n- **Duration of impact:** [how long the system was in a bad state, if applicable]\n- **Data loss:** [yes/no — if yes, what and how much]\n- **Recovery actions taken:** [what was done to restore service, if anything]\n\n## Root Cause\n\n[The terminal link of the causal chain. State this plainly and specifically. This should be a single, clear sentence that explains *why* the failure happened at the deepest level you could trace.]\n\n**Example (bad):** \"npm install failed.\"\n**Example (good):** \"The agent ran as a non-root user in a container where the global npm directory (`/usr/lib/node_modules`) was owned by root with no write permission for others, and `npm install` without `--prefix` defaults to global installation.\"\n\n## Causal Chain\n\nTrace backward from the failure point. Each entry should answer \"why did the previous step happen/ matter?\"\n\n1. **[FAILURE]** `npm install` returned EACCES on `/usr/lib/node_modules`\n2. **Because:** npm attempted a global install (no `--prefix` or local `package.json`)\n3. **Because:** The agent assumed the install target was local, but the working directory had no `package.json`\n4. **Because:** The agent didn't verify the working directory contents before running the install\n5. **Because:** The task description referenced a project at a path the agent assumed existed without checking ← **ROOT CAUSE**\n\n**Root cause:** The agent operated on an unverified assumption about the filesystem state (project path) and cascaded into a permissions failure that looked like an environment problem.\n\n## Contributing Factors\n\nFactors that didn't *cause* the failure but made it worse or harder to diagnose:\n\n- **Retry without analysis:** The agent retried with `sudo` before understanding the error, introducing a new failure mode and obscuring the original cause.\n- **Poor error context:** npm's error message mentioned the path but not that it was a global vs. local install distinction.\n- **No pre-flight check:** No step verified the working directory contained the expected project files.\n\n## What Went Well\n\n[Post-mortems that only list problems create a blame culture. Note what worked — fast detection, good logging, graceful degradation, etc.]\n\n- The agent correctly abandoned the task after two failures rather than continuing to cascade.\n- Tool-call logs captured the full error messages, enabling this analysis.\n\n## Action Items\n\nEach item must be **specific, assigned, and verifiable**.\n\n| # | Action | Owner | Verification | Priority |\n|---|---|---|---|---|\n| 1 | Add a pre-flight check: verify `package.json` exists in the working directory before running `npm install` | agent framework team | Unit test: `test_npm_install_requires_package_json` passes | High |\n| 2 | Add `--prefix` flag to npm install commands by default, or detect global vs. local context | agent framework team | Manual: run in a dir without package.json, confirm error is actionable | Medium |\n| 3 | Add retry-with-analysis rule to agent: after first failure, perform Phase 1 triage before retrying | this skill | Verify `failure-forensics` skill is loaded and triggered on retry | High |\n\n## Lessons Learned\n\nGeneralizable insights. These are the durable output of the post-mortem — they should apply beyond this specific incident.\n\n1. **Verify assumptions about filesystem state before acting.** \"The project is at this path\" is an assumption, not a fact. `ls` or `test -f` is cheap; cascading failures are expensive.\n2. **Retry is not a debugging strategy.** Retrying without analysis can introduce new failure modes and destroy evidence of the original cause.\n3. **Permission errors often mask environment/context errors.** A permissions failure at the symptom level may have a logic or assumption failure at the root.\n4. **Error messages rarely point at the root cause directly.** They point at the *symptom*. Always read them as clues, not diagnoses.\n\n## Appendix\n\n### Full Error Output\n\n```\n[Paste the complete error message / stack trace / log output here]\n```\n\n### Environment\n\n- **OS:** [e.g., Ubuntu 22.04 x86_64]\n- **Agent runtime:** [e.g., Hermes Agent v2.3]\n- **Key dependencies:** [versions of relevant tools]\n- **Working directory:** [path]\n\n### References\n\n- [Links to related issues, PRs, prior post-mortems, docs]\n\nFile v0.1.0:skill-card.md\n\n## Description:\n\nUse when an agent task fails or produces unexpected results. Performs structured post-mortem root cause analysis: categorizes the failure, traces the exact failure point through tool-call logs, reconstructs the decision chain, generates a post-mortem report, and saves lessons to prevent recurrence.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[voronindenis5](https://clawhub.ai/user/voronindenis5)\n\n### License/Terms of Use:\n\nMIT\n\n## Use Case:\n\nDevelopers and agent operators use this skill after a failed or unexpected agent task to categorize the failure, reconstruct the timeline, trace root causes, and produce a durable post-mortem report with lessons learned.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Failure logs and saved post-mortem reports can contain tokens, credentials, personal data, internal paths, prompts, or sensitive command output.\n\nMitigation: Review and redact sensitive content before saving or sharing reports, and store generated artifacts only in locations with appropriate access controls.\n\n## Reference(s):\n\n- [Failure Taxonomy](references/failure-taxonomy.md)\n- [Post-Mortem Report Template](references/post-mortem-template.md)\n- [Source Repository](https://github.com/voronindenis5/failure-forensics)\n- [ClawHub Skill Page](https://clawhub.ai/voronindenis5/skills/failure-forensics)\n- [Publisher Profile](https://clawhub.ai/user/voronindenis5)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, Shell commands, Guidance, Files]\n\n**Output Format:** [Markdown reports, timeline summaries, classifications, and command-line guidance]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [The bundled Python script accepts JSON or JSONL tool-call logs and can write timeline or post-mortem Markdown files.]\n\n## Skill Version(s):\n\n0.1.0 (source: ClawHub release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v0.1.0:LICENSE\n\nMIT License\n\nCopyright (c) 2026 Denis Voronin\n\nPermission is hereby granted, free of charge, to any person obtaining a copy\nof this software and associated documentation files (the \"Software\"), to deal\nin the Software without restriction, including without limitation the rights\nto use, copy, modify, merge, publish, distribute, sublicense, and/or sell\ncopies of the Software, and to permit persons to whom the Software is\nfurnished to do so, subject to the following conditions:\n\nThe above copyright notice and this permission notice shall be included in all\ncopies or substantial portions of the Software.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR\nIMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,\nFITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE\nAUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER\nLIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,\nOUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE\nSOFTWARE.","readmeExcerpt":"Skill: failure-forensics Owner: voronindenis5 Summary: Use when an agent task fails or produces unexpected results. Performs structured post-mortem root cause analysis: categorizes the failure, traces the exact failure point through tool-call logs, reconstructs the decision chain, generates a post-mortem report, and saves lessons to prevent recurrence. Tags: latest:0.1.1 Version history: v0.1.1 | 2026-08-11T11:58:54.","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"python3 scripts/failure_forensics.py analyze \\\n     --log session.jsonl \\\n     --output timeline.md"},{"language":"bash","snippet":"# JSONL log (one JSON object per line)\npython3 scripts/failure_forensics.py analyze --log session.jsonl --format jsonl\n\n# JSON array\npython3 scripts/failure_forensics.py analyze --log session.json --format json\n\n# Write report to file\npython3 scripts/failure_forensics.py analyze --log session.jsonl --output report.md"},{"language":"bash","snippet":"python3 scripts/failure_forensics.py categorize --error \"ConnectionRefusedError: [Errno 111] Connection refused\"\n# Output: network\n\npython3 scripts/failure_forensics.py categorize --error \"PermissionError: [Errno 13] Permission denied\"\n# Output: permissions"},{"language":"bash","snippet":"python3 scripts/failure_forensics.py report --log session.jsonl --title \"Deploy failure 2024-01-15\" --author \"agent\""},{"language":"json","snippet":"{\n  \"timestamp\": \"2024-01-15T10:23:45Z\",\n  \"tool\": \"terminal\",\n  \"args\": {\"command\": \"npm install\"},\n  \"result\": {\"success\": false, \"error\": \"EACCES: permission denied, open '/usr/lib/node_modules'\"},\n  \"duration_ms\": 1200\n}"},{"language":"text","snippet":"failure-forensics/\n├── SKILL.md                          # Main skill definition (YAML frontmatter + workflow)\n├── README.md                         # This file\n├── LICENSE                           # MIT\n├── references/\n│   ├── failure-taxonomy.md           # Six-category failure taxonomy with signatures\n│   └── post-mortem-template.md       # Fill-in-the-blanks report template\n└── scripts/\n    └── failure_forensics.py          # Log parser, categorizer, report generator"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: failure-forensics\ndescription: \"Use when an agent task fails or produces unexpected results. Performs structured post-mortem root cause analysis: categorizes the failure, traces the exact failure point through tool-call logs, reconstructs the decision chain, generates a post-mortem report, and saves lessons to prevent recurrence.\"\nversion: 1.0.0\nauthor: Denis Voronin\nlicense: MIT\nmetadata:\n  hermes:\n    tags: [debugging, post-mortem, forensics, root-cause-analysis, failure-analysis, agent-reliability]\n    related_skills: [systematic-debugging, debugging-hermes-tui-commands]\n---\n\n# Failure Forensics\n\n## Overview\n\nWhen an agent task fails, the default response is to retry — hoping for a different outcome. **Failure Forensics** rejects that reflex. Instead, the agent performs structured root cause analysis *before* retrying, treating every failure as evidence to be collected, categorized, and learned from.\n\nThe workflow has four phases:\n\n1. **Triage** — Categorize the failure using the taxonomy in [`references/failure-taxonomy.md`](references/failure-taxonomy.md).\n2. **Timeline Reconstruction** — Parse tool-call logs and agent decision points to build a chronological failure timeline. The script [`scripts/failure_forensics.py`](scripts/failure_forensics.py) automates this from JSON or JSONL log formats.\n3. **Causal Chain Analysis** — Trace the chain of decisions, assumptions, and actions that led from the task kickoff to the failure point. Identify the *root cause*, not just the proximate symptom.\n4. **Post-Mortem Report** — Generate a structured report from the template in [`references/post-mortem-template.md`](references/post-mortem-template.md) and persist it so future sessions can learn.\n\nThis skill turns a single failure into a permanent, reusable lesson.\n\n## When to Use\n\n- **An agent task failed** and retrying without understanding *why* is risky.\n- **A failure recurs** across attempts — you suspect a systemic cause, not bad luck.\n- **You need an artifact** documenting what went wrong for a team review or audit.\n- **A complex multi-step task** partially completed then broke — you need to understand which step is safe to resume from.\n- **You want to improve agent reliability** by building a corpus of past failure patterns.\n\n### Don't use for:\n\n- **Trivial failures with obvious fixes** (typo in a command, missing flag). Fix and move on.\n- **Live debugging** of an actively failing process — use `systematic-debugging` for that. Run forensics *after* the process is dead or the task is abandoned.\n- **Human performance reviews.** This skill analyzes agent + tool behavior, not people.\n\n## The Forensics Workflow\n\n### Phase 1: Triage — Categorize the Failure\n\nRead the full taxonomy in [`references/failure-taxonomy.md`](references/failure-taxonomy.md). At a high level, every failure falls into one of six categories:\n\n| Category | Signature | First Question |\n|---|---|---|\n| **Network** | Connection refused, timeout, DNS, TLS, 5xx HTTP | \"Is the"},{"path":"README.md","content":"# Failure Forensics\n\n> A structured post-mortem analysis skill for AI agents. When a task fails, don't just retry — investigate.\n\n[![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE)\n\n## What It Does\n\n**Failure Forensics** is a skill for AI agent frameworks (Hermes Agent, OpenClaw, and compatible). When an agent task fails, instead of blindly retrying, the agent performs a four-phase structured root cause analysis:\n\n1. **Triage** — Categorize the failure (network, permissions, logic, environment, dependency, resource)\n2. **Timeline Reconstruction** — Parse tool-call logs to build a chronological failure timeline\n3. **Causal Chain Analysis** — Trace the decision chain backward to the root cause\n4. **Post-Mortem Report** — Generate a structured report and save lessons learned\n\n## Why\n\nRetrying a failed task without understanding why it failed is a gamble. You might:\n- Hit the same failure again (wasted effort)\n- Mask the real cause with incidental changes (harder to debug later)\n- Miss a systemic issue that will recur in different forms\n\nFailure Forensics turns each failure into a reusable lesson, building institutional memory that makes the agent more reliable over time.\n\n## Repository Structure\n\n```\nfailure-forensics/\n├── SKILL.md                          # Main skill definition (YAML frontmatter + workflow)\n├── README.md                         # This file\n├── LICENSE                           # MIT\n├── references/\n│   ├── failure-taxonomy.md           # Six-category failure taxonomy with signatures\n│   └── post-mortem-template.md       # Fill-in-the-blanks report template\n└── scripts/\n    └── failure_forensics.py          # Log parser, categorizer, report generator\n```\n\n## Quick Start\n\n### As a Hermes Agent Skill\n\nCopy or symlink this directory to your skills folder:\n\n```bash\ncp -r failure-forensics/ ~/.hermes/skills/\n```\n\nThe skill auto-loads. When a task fails, the agent will follow the forensics workflow described in `SKILL.md`.\n\n### Standalone (Script Only)\n\nThe Python script works independently — no agent required:\n\n```bash\n# Analyze a JSONL log of tool calls\npython3 scripts/failure_forensics.py analyze --log session.jsonl --output timeline.md\n\n# Categorize an error message\npython3 scripts/failure_forensics.py categorize --error \"ConnectionRefusedError: Connection refused\"\n\n# Generate a pre-filled post-mortem report\npython3 scripts/failure_forensics.py report --log session.jsonl --title \"Deploy failure\"\n```\n\n### Log Format\n\nThe analyzer reads JSON or JSONL files where each entry represents a tool call:\n\n```json\n{\n  \"timestamp\": \"2024-01-15T10:23:45Z\",\n  \"tool\": \"terminal\",\n  \"args\": {\"command\": \"npm install\"},\n  \"result\": {\"success\": false, \"error\": \"EACCES: permission denied\"},\n  \"duration_ms\": 1200\n}\n```\n\nSee `scripts/sample_log.jsonl` for a working example.\n\n## Failure Taxonomy (Summary)\n\n| Category | Signature | Example |\n|---|---|---|\n| **Network** | Connection refused, timeout, DNS, TLS | `curl: (7) Failed "},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn75wwn4x6djaf28jbykeamazd81gtdp\",\n  \"slug\": \"failure-forensics\",\n  \"version\": \"0.1.1\",\n  \"publishedAt\": 1786449534013\n}"},{"path":"references/failure-taxonomy.md","content":"# Failure Taxonomy\n\nA reference for categorizing failures during Phase 1 (Triage) of the forensics workflow.\n\n## How to Use This Taxonomy\n\n1. Read the error message / failure signature.\n2. Match it against the patterns below.\n3. If multiple categories match, pick the **most specific** one. A `ModuleNotFoundError` is a *dependency* failure, not an *environment* failure, even though both relate to the system.\n4. If no category fits cleanly, record it as **uncategorized** and note the novel pattern. The taxonomy grows by accretion.\n\n---\n\n## 1. Network Failures\n\n**Core question:** Is the endpoint reachable *right now*, and from this environment?\n\n### Signatures\n\n| Pattern | Meaning |\n|---|---|\n| `Connection refused`, `ConnectionRefusedError` | Port open but nothing listening / rejected |\n| `Connection timed out`, `ETIMEDOUT` | Packet dropped, firewall, or host unreachable |\n| `Name or service not known`, `NXDOMAIN` | DNS resolution failure |\n| `SSL: CERTIFICATE_VERIFY_FAILED` | TLS cert expired, self-signed, or MITM |\n| `HTTP 502 Bad Gateway`, `503 Service Unavailable`, `504 Gateway Timeout` | Server-side failure |\n| `curl: (7) Failed to connect`, `curl: (28) Connection timed out` | CLI-level network failure |\n| `ECONNRESET`, `Connection reset by peer` | Remote end dropped the connection |\n\n### Diagnostic Questions\n\n- Does `curl -v <url>` or `nc -zv <host> <port>` work from the same environment?\n- Is this an internal vs. external endpoint? (internal may need VPN/peering)\n- Is there a proxy or corporate firewall in play?\n- Did this work before? What changed? (network config, DNS, certs)\n\n### Common Root Causes\n\n- Service is down or not started\n- Wrong port number (e.g., `:443` vs `:80`)\n- DNS misconfiguration or stale cache\n- Expired TLS certificate\n- Firewall / security group blocking the port\n- IPv6 vs IPv4 resolution mismatch\n\n---\n\n## 2. Permissions Failures\n\n**Core question:** Does the credential/token/user have the needed scope for this action?\n\n### Signatures\n\n| Pattern | Meaning |\n|---|---|\n| `401 Unauthorized`, `HTTP 401` | No credentials, or credentials rejected |\n| `403 Forbidden`, `HTTP 403` | Credentials valid, but lack permission |\n| `Permission denied`, `EACCES`, `PermissionError` | Filesystem permission denied |\n| `Access denied`, `UnauthorizedAccess` | Cloud API / IAM denial |\n| `insufficient privileges`, `requires elevated permissions` | OS-level privilege denial |\n| `invalid token`, `token expired`, `invalid_grant` | Auth token problem |\n\n### Diagnostic Questions\n\n- What user/service account is the agent running as?\n- What scopes/roles does the token have? (check the token's claims, not assumptions)\n- Is this a filesystem permission issue (check `ls -la`, `id`, `getfacl`)?\n- Is this an API/IAM issue (check the service's permission model)?\n- Did the token expire? Check issuance and expiry timestamps.\n\n### Common Root Causes\n\n- Token expired and wasn't refreshed\n- Token has correct identity but wrong scope/role\n- File owned by a differ"},{"path":"references/post-mortem-template.md","content":"# Post-Mortem Report Template\n\nFill in every section. If a section doesn't apply, write \"N/A — [reason]\" rather than deleting it. A blank section is information; a missing section is ambiguity.\n\n---\n\n# Post-Mortem: [TITLE]\n\n**Date:** YYYY-MM-DD\n**Author:** [agent name / human name]\n**Task:** [one-line description of what the agent was trying to do]\n**Status:** [Failed / Partially completed / Recovered after intervention]\n\n## Summary\n\n[One paragraph, plain language. Describe what happened, not just that it failed. A reader who wasn't present should understand the failure and its impact from this paragraph alone. Aim for 3-5 sentences.]\n\n**Failure category:** [network / permissions / logic / environment / dependency / resource / uncategorized]\n\n## Timeline\n\nReconstruct the sequence of events leading to the failure. Use timestamps from logs where available. Mark the failure point explicitly with **[FAILURE]**.\n\n| Time (UTC) | Event | Outcome |\n|---|---|---|\n| 10:23:01 | Agent received task: \"Deploy service to staging\" | Task started |\n| 10:23:15 | Agent ran `git pull origin main` | Success |\n| 10:23:45 | Agent ran `npm install` | **[FAILURE]** — EACCES: permission denied |\n| 10:24:02 | Agent retried `npm install` with `sudo` | Different error: EACCES on different path |\n| 10:24:30 | Agent abandoned task | Task failed |\n\nIf using the forensics script, paste the generated timeline here.\n\n## Impact\n\n- **What was affected:** [services, data, users, downstream tasks]\n- **Severity:** [low / medium / high / critical]\n- **Duration of impact:** [how long the system was in a bad state, if applicable]\n- **Data loss:** [yes/no — if yes, what and how much]\n- **Recovery actions taken:** [what was done to restore service, if anything]\n\n## Root Cause\n\n[The terminal link of the causal chain. State this plainly and specifically. This should be a single, clear sentence that explains *why* the failure happened at the deepest level you could trace.]\n\n**Example (bad):** \"npm install failed.\"\n**Example (good):** \"The agent ran as a non-root user in a container where the global npm directory (`/usr/lib/node_modules`) was owned by root with no write permission for others, and `npm install` without `--prefix` defaults to global installation.\"\n\n## Causal Chain\n\nTrace backward from the failure point. Each entry should answer \"why did the previous step happen/ matter?\"\n\n1. **[FAILURE]** `npm install` returned EACCES on `/usr/lib/node_modules`\n2. **Because:** npm attempted a global install (no `--prefix` or local `package.json`)\n3. **Because:** The agent assumed the install target was local, but the working directory had no `package.json`\n4. **Because:** The agent didn't verify the working directory contents before running the install\n5. **Because:** The task description referenced a project at a path the agent assumed existed without checking ← **ROOT CAUSE**\n\n**Root cause:** The agent operated on an unverified assumption about the filesystem state (project path) and cascaded i"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Use when an agent task fails or produces unexpected results. Performs structured post-mortem root cause analysis: categorizes the failure, traces the exact failure point through tool-call logs, reconstructs the decision chain, generates a post-mortem report, and saves lessons to prevent recurrence. Skill: failure-forensics Owner: voronindenis5 Summary: Use when an agent task fails or produces unexpected results. Performs structured post-mortem root cause analysis: categorizes the failure, traces the exact failure point through tool-call logs, reconstructs the decision chain, generates a post-mortem report, and saves lessons to prevent recurrence. Tags: latest:0.1.1 Version history: v0.1.1 | 2026-08-11T11:58:54.","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1939,"uniquenessScore":47,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T14:15:58.668Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T14:15:58.668Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T05:57:52.035Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}