{"id":"0f645bea-ec2a-44f2-898f-bec41127471b","entityType":"agent","slug":"clawhub-skills-1kalin-afrexai-sre-platform","name":"afrexai-sre-platform","canonicalUrl":"https://www.xpersona.co/agent/clawhub-skills-1kalin-afrexai-sre-platform","canonicalPath":"/agent/clawhub-skills-1kalin-afrexai-sre-platform","generatedAt":"2026-10-09T19:11:54.646Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":null},"description":"SRE & Incident Management Platform SRE & Incident Management Platform Complete Site Reliability Engineering system — from SLO definition through incident response, chaos engineering, and operational excellence. Zero dependencies. --- Phase 1: Reliability Assessment Before building anything, assess where you are. Service Catalog Entry Maturity Assessment (Score 1-5 per dimension) | Dimension | 1 (Ad-hoc) | 3 (Defined) | 5 (Optimized) | Score | |-------","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. Last updated 4/15/2026.","installCommand":"clawhub skill install skills:1kalin:afrexai-sre-platform","sourceUrl":"https://github.com/openclaw/skills/tree/main/skills/1kalin/afrexai-sre-platform","homepage":null,"primaryLinks":[{"label":"View on ClawHub","url":"https://github.com/openclaw/skills/tree/main/skills/1kalin/afrexai-sre-platform","kind":"source"}],"safetyScore":84,"overallRank":62,"popularityScore":50,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"SRE & Incident Management Platform SRE & Incident Management Platform Complete Site Reliability Engineering system — from SLO definition through incident respon"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[{"label":"always","status":"self-declared"},{"label":"escalate","status":"self-declared"}],"verifiedCount":0,"selfDeclaredCount":3,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"},{"key":"always","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"},{"key":"escalate","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile capability:always|supported|profile capability:escalate|supported|profile"}},"adoption":{"evidence":{"source":"no-adoption-signals","verified":false,"confidence":"low","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":"No source adoption metrics were available."},"stars":null,"forks":null,"downloads":null,"packageName":null,"latestVersion":null,"tractionLabel":null},"release":{"evidence":{"source":"agent-index","verified":false,"confidence":"medium","updatedAt":"2026-02-25T05:52:49.007Z","emptyReason":null},"lastUpdatedAt":"2026-04-15T00:45:39.800Z","lastCrawledAt":"2026-02-25T05:52:49.007Z","lastIndexedAt":null,"nextCrawlAt":"2026-02-26T05:52:49.007Z","lastVerifiedAt":null,"highlights":[]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install skills:1kalin:afrexai-sre-platform","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-sre-platform/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-sre-platform/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-sre-platform/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-sre-platform/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-sre-platform/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-sre-platform/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-09T19:11:54.645Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-sre-platform/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-sre-platform/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-sre-platform/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-sre-platform/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":null},"readme":"# SRE & Incident Management Platform\n\nComplete Site Reliability Engineering system — from SLO definition through incident response, chaos engineering, and operational excellence. Zero dependencies.\n\n---\n\n## Phase 1: Reliability Assessment\n\nBefore building anything, assess where you are.\n\n### Service Catalog Entry\n\n```yaml\nservice:\n  name: \"\"\n  tier: \"\"  # critical | important | standard | experimental\n  owner_team: \"\"\n  oncall_rotation: \"\"\n  dependencies:\n    upstream: []    # services we call\n    downstream: []  # services that call us\n  data_classification: \"\"  # public | internal | confidential | restricted\n  deployment_frequency: \"\"  # daily | weekly | biweekly | monthly\n  architecture: \"\"  # monolith | microservice | serverless | hybrid\n  language: \"\"\n  infra: \"\"  # k8s | ECS | Lambda | VM | bare-metal\n  traffic_pattern: \"\"  # steady | diurnal | spiky | seasonal\n  peak_rps: 0\n  storage_gb: 0\n  monthly_cost_usd: 0\n```\n\n### Maturity Assessment (Score 1-5 per dimension)\n\n| Dimension | 1 (Ad-hoc) | 3 (Defined) | 5 (Optimized) | Score |\n|-----------|-----------|-------------|---------------|-------|\n| SLOs | No SLOs defined | SLOs exist, reviewed quarterly | Data-driven SLOs, auto error budgets | |\n| Monitoring | Basic health checks | Golden signals + dashboards | Full observability, anomaly detection | |\n| Incident Response | No runbooks, hero culture | Documented process, postmortems | Automated detection, structured ICS | |\n| Automation | Manual deployments | CI/CD pipeline, some automation | Self-healing, auto-scaling, GitOps | |\n| Chaos Engineering | No testing | Basic failure injection | Continuous chaos in production | |\n| Capacity Planning | Reactive scaling | Quarterly forecasting | Predictive auto-scaling | |\n| Toil Management | >50% toil | Toil tracked, reduction plans | <25% toil, systematic elimination | |\n| On-Call Health | Burnout, 24/7 individuals | Rotation exists, escalation paths | Balanced load, <2 pages/shift | |\n\n**Score interpretation:**\n- 8-16: Firefighting mode — start with SLOs + incident process\n- 17-24: Foundation built — add chaos engineering + toil reduction\n- 25-32: Maturing — optimize error budgets + capacity planning\n- 33-40: Advanced — focus on predictive reliability + culture\n\n---\n\n## Phase 2: SLI/SLO Framework\n\n### SLI Selection by Service Type\n\n| Service Type | Primary SLI | Secondary SLIs |\n|-------------|-------------|----------------|\n| API/Backend | Request success rate | Latency p50/p95/p99, throughput |\n| Frontend/Web | Page load (LCP) | FID/INP, CLS, error rate |\n| Data Pipeline | Freshness | Correctness, completeness, throughput |\n| Storage | Durability | Availability, latency |\n| Streaming | Processing latency | Throughput, ordering, data loss rate |\n| Batch Job | Success rate | Duration, SLA compliance |\n| ML Model | Prediction latency | Accuracy drift, feature freshness |\n\n### SLI Specification Template\n\n```yaml\nsli:\n  name: \"request_success_rate\"\n  description: \"Proportion of valid requests served successfully\"\n  type: \"availability\"  # availability | latency | quality | freshness\n  measurement:\n    good_events: \"HTTP responses with status < 500\"\n    total_events: \"All HTTP requests excluding health checks\"\n    source: \"load balancer access logs\"\n    aggregation: \"sum(good) / sum(total) over rolling 28-day window\"\n  exclusions:\n    - \"Health check endpoints (/healthz, /readyz)\"\n    - \"Synthetic monitoring traffic\"\n    - \"Requests from blocked IPs\"\n    - \"4xx responses (client errors)\"\n```\n\n### SLO Target Selection Guide\n\n| Nines | Uptime % | Downtime/month | Appropriate for |\n|-------|----------|----------------|-----------------|\n| 2 nines | 99% | 7h 18m | Internal tools, dev environments |\n| 2.5 | 99.5% | 3h 39m | Non-critical services, backoffice |\n| 3 nines | 99.9% | 43m 50s | Standard production services |\n| 3.5 | 99.95% | 21m 55s | Important customer-facing services |\n| 4 nines | 99.99% | 4m 23s | Critical services, payments, auth |\n| 5 nines | 99.999% | 26s | Life-safety, financial clearing |\n\n**Rules for setting targets:**\n1. Start lower than you think — you can always tighten\n2. SLO < SLA (always have buffer — typically 0.1-0.5% margin)\n3. Internal SLO < External SLO (catch problems before customers do)\n4. Each nine costs ~10x more to achieve\n5. If you can't measure it, you can't SLO it\n\n### SLO Document Template\n\n```yaml\nslo:\n  service: \"\"\n  sli: \"\"\n  target: 99.9  # percentage\n  window: \"28d\"  # rolling window\n  error_budget: 0.1  # 100% - target\n  error_budget_minutes: 40  # per 28-day window\n  \n  burn_rate_alerts:\n    - name: \"fast_burn\"\n      burn_rate: 14.4  # exhausts budget in 2 hours\n      short_window: \"5m\"\n      long_window: \"1h\"\n      severity: \"page\"\n    - name: \"medium_burn\"\n      burn_rate: 6.0   # exhausts budget in ~5 hours\n      short_window: \"30m\"\n      long_window: \"6h\"\n      severity: \"page\"\n    - name: \"slow_burn\"\n      burn_rate: 1.0   # exhausts budget in 28 days\n      short_window: \"6h\"\n      long_window: \"3d\"\n      severity: \"ticket\"\n  \n  review_cadence: \"monthly\"\n  owner: \"\"\n  stakeholders: []\n  \n  escalation_when_budget_exhausted:\n    - \"Halt non-critical deployments\"\n    - \"Redirect engineering to reliability work\"\n    - \"Escalate to VP Engineering if no improvement in 48h\"\n```\n\n---\n\n## Phase 3: Error Budget Management\n\n### Error Budget Policy\n\n```yaml\nerror_budget_policy:\n  service: \"\"\n  \n  budget_states:\n    healthy:\n      condition: \"remaining_budget > 50%\"\n      actions:\n        - \"Normal development velocity\"\n        - \"Feature work prioritized\"\n        - \"Chaos experiments allowed\"\n    \n    warning:\n      condition: \"remaining_budget 25-50%\"\n      actions:\n        - \"Increase monitoring scrutiny\"\n        - \"Review recent changes for risk\"\n        - \"Limit risky deployments to business hours\"\n        - \"No chaos experiments\"\n    \n    critical:\n      condition: \"remaining_budget 0-25%\"\n      actions:\n        - \"Feature freeze — reliability work only\"\n        - \"All deployments require SRE approval\"\n        - \"Mandatory rollback plan for every change\"\n        - \"Daily error budget review\"\n    \n    exhausted:\n      condition: \"remaining_budget <= 0\"\n      actions:\n        - \"Complete deployment freeze\"\n        - \"All engineering redirected to reliability\"\n        - \"VP Engineering notified\"\n        - \"Postmortem required for budget exhaustion\"\n        - \"Freeze maintained until budget recovers to 10%\"\n  \n  exceptions:\n    - \"Security patches always allowed\"\n    - \"Regulatory compliance changes always allowed\"\n    - \"Data loss prevention always allowed\"\n  \n  reset: \"Rolling 28-day window (no manual resets)\"\n```\n\n### Burn Rate Calculation\n\n```\nBurn rate = (error rate observed) / (error rate allowed by SLO)\n\nExample:\n- SLO: 99.9% (error budget = 0.1%)\n- Current error rate: 0.5%\n- Burn rate = 0.5% / 0.1% = 5x\n\nAt 5x burn rate → budget exhausted in 28d / 5 = 5.6 days\n```\n\n### Error Budget Dashboard\n\nTrack weekly:\n\n| Metric | Current | Trend | Status |\n|--------|---------|-------|--------|\n| Budget remaining (%) | | ↑↓→ | 🟢🟡🔴 |\n| Budget consumed this week | | | |\n| Burn rate (1h / 6h / 24h) | | | |\n| Incidents consuming budget | | | |\n| Top error contributor | | | |\n| Projected exhaustion date | | | |\n\n---\n\n## Phase 4: Monitoring & Alerting Architecture\n\n### Four Golden Signals\n\n| Signal | What to Measure | Alert When |\n|--------|----------------|------------|\n| **Latency** | p50, p95, p99 response time | p99 > 2x baseline for 5 min |\n| **Traffic** | Requests/sec, concurrent users | >30% drop (indicates upstream issue) OR >50% spike |\n| **Errors** | 5xx rate, timeout rate, exception rate | Error rate > SLO burn rate threshold |\n| **Saturation** | CPU, memory, disk, connections, queue depth | >80% sustained for 10 min |\n\n### USE Method (Infrastructure)\n\nFor every resource, track:\n- **Utilization**: % of capacity used (0-100%)\n- **Saturation**: queue depth / wait time (0 = no waiting)\n- **Errors**: error count / error rate\n\n### RED Method (Services)\n\nFor every service, track:\n- **Rate**: requests per second\n- **Errors**: failed requests per second\n- **Duration**: latency distribution\n\n### Alert Design Rules\n\n1. **Every alert must have a runbook link** — no exceptions\n2. **Every alert must be actionable** — if you can't act on it, delete it\n3. **Symptoms over causes** — alert on \"users can't check out\" not \"database CPU high\"\n4. **Multi-window, multi-burn-rate** — avoid single-threshold alerts\n5. **Page only for customer impact** — everything else is a ticket\n6. **Alert fatigue = death** — review alert volume monthly; target <5 pages/week per service\n\n### Alert Severity Guide\n\n| Severity | Response Time | Notification | Examples |\n|----------|--------------|-------------|----------|\n| P0/Page | <5 min | PagerDuty + phone | SLO burn rate critical, data loss, security breach |\n| P1/Urgent | <30 min | Slack + PagerDuty | Degraded service, elevated errors, capacity warning |\n| P2/Ticket | Next business day | Ticket auto-created | Slow burn, non-critical component down |\n| P3/Log | Weekly review | Dashboard only | Informational, trend detection |\n\n### Structured Log Standard\n\n```json\n{\n  \"timestamp\": \"2026-02-17T11:24:00.000Z\",\n  \"level\": \"error\",\n  \"service\": \"payment-api\",\n  \"trace_id\": \"abc123\",\n  \"span_id\": \"def456\",\n  \"message\": \"Payment processing failed\",\n  \"error_type\": \"TimeoutException\",\n  \"error_message\": \"Gateway timeout after 30s\",\n  \"http_method\": \"POST\",\n  \"http_path\": \"/api/v1/payments\",\n  \"http_status\": 504,\n  \"duration_ms\": 30012,\n  \"customer_id\": \"cust_xxx\",\n  \"payment_id\": \"pay_yyy\",\n  \"amount_cents\": 4999,\n  \"retry_count\": 2,\n  \"environment\": \"production\",\n  \"host\": \"payment-api-7b4d9-xk2p1\",\n  \"region\": \"us-east-1\"\n}\n```\n\n---\n\n## Phase 5: Incident Response Framework\n\n### Severity Classification Matrix\n\n| | Impact: 1 User | Impact: <25% Users | Impact: >25% Users | Impact: All Users |\n|-|----------------|--------------------|--------------------|-------------------|\n| **Core function down** | SEV3 | SEV2 | SEV1 | SEV1 |\n| **Degraded performance** | SEV4 | SEV3 | SEV2 | SEV1 |\n| **Non-core feature down** | SEV4 | SEV3 | SEV3 | SEV2 |\n| **Cosmetic/minor** | SEV4 | SEV4 | SEV3 | SEV3 |\n\n**Auto-escalation triggers:**\n- Any data loss → SEV1 minimum\n- Security breach with PII → SEV1\n- Revenue-impacting → SEV1 or SEV2\n- SLA breach imminent → auto-escalate one level\n\n### Incident Command System (ICS)\n\n| Role | Responsibility | Assigned |\n|------|---------------|----------|\n| **Incident Commander (IC)** | Owns resolution, makes decisions, manages timeline | |\n| **Communications Lead** | Status updates, stakeholder comms, customer-facing | |\n| **Operations Lead** | Hands-on-keyboard, executing fixes | |\n| **Subject Matter Expert** | Deep knowledge of affected system | |\n| **Scribe** | Documenting timeline, actions, decisions | |\n\n**IC Rules:**\n1. IC does NOT debug — IC coordinates\n2. IC makes final decisions when team disagrees\n3. IC can escalate severity at any time\n4. IC owns handoff if rotation changes\n5. IC calls end-of-incident\n\n### Incident Response Workflow\n\n```\nDETECT → TRIAGE → RESPOND → MITIGATE → RESOLVE → REVIEW\n\nStep 1: DETECT (0-5 min)\n├── Alert fires OR user report received\n├── On-call acknowledges within SLA\n└── Quick assessment: is this real? What severity?\n\nStep 2: TRIAGE (5-15 min)\n├── Classify severity using matrix above\n├── Assign IC and roles\n├── Open incident channel (#inc-YYYY-MM-DD-title)\n├── Post initial status update\n└── Start timeline document\n\nStep 3: RESPOND (15 min - ongoing)\n├── IC briefs team: \"Here's what we know, here's what we don't\"\n├── Operations Lead begins investigation\n├── Check: recent deployments? Config changes? Dependency issues?\n├── Parallel investigation tracks if needed\n└── 15-minute check-ins for SEV1, 30-min for SEV2\n\nStep 4: MITIGATE (ASAP)\n├── Priority: STOP THE BLEEDING\n├── Options (fastest first):\n│   ├── Rollback last deployment\n│   ├── Feature flag disable\n│   ├── Traffic shift / failover\n│   ├── Scale up / circuit breaker\n│   └── Manual data fix\n├── Mitigated ≠ Resolved — temporary fix is OK\n└── Update status: \"Impact mitigated, root cause investigation ongoing\"\n\nStep 5: RESOLVE\n├── Root cause identified and fixed\n├── Verification: SLIs back to normal for 30+ minutes\n├── All-clear communicated\n└── IC declares incident resolved\n\nStep 6: REVIEW (within 5 business days)\n├── Blameless postmortem written\n├── Action items assigned with owners and deadlines\n├── Postmortem review meeting\n└── Action items tracked to completion\n```\n\n### Communication Templates\n\n**Initial notification (internal):**\n```\n🔴 INCIDENT: [Title]\nSeverity: SEV[X]\nImpact: [Who/what is affected]\nStatus: Investigating\nIC: [Name]\nChannel: #inc-[date]-[slug]\nNext update: [time]\n```\n\n**Customer-facing status:**\n```\n[Service] - Investigating increased error rates\n\nWe are currently investigating reports of [symptom]. \nSome users may experience [user-visible impact].\nOur team is actively working on a resolution.\nWe will provide an update within [time].\n```\n\n**Resolution notification:**\n```\n✅ RESOLVED: [Title]\nDuration: [X hours Y minutes]\nImpact: [Summary]\nRoot cause: [One sentence]\nPostmortem: [Link] (within 5 business days)\n```\n\n---\n\n## Phase 6: Postmortem Framework\n\n### Blameless Postmortem Template\n\n```yaml\npostmortem:\n  title: \"\"\n  date: \"\"\n  severity: \"\"  # SEV1-4\n  duration: \"\"  # total incident duration\n  authors: []\n  reviewers: []\n  status: \"draft\"  # draft | in-review | final\n  \n  summary: |\n    One paragraph: what happened, what was the impact, how was it resolved.\n  \n  impact:\n    users_affected: 0\n    duration_minutes: 0\n    revenue_impact_usd: 0\n    slo_budget_consumed_pct: 0\n    data_loss: false\n    customer_tickets: 0\n  \n  timeline:\n    - time: \"\"\n      event: \"\"\n      # Chronological, every significant event\n      # Include detection time, escalation, mitigation attempts\n  \n  root_cause: |\n    Technical explanation of WHY it happened.\n    Go deep — surface causes are not root causes.\n  \n  contributing_factors:\n    - \"\"  # What made it worse or delayed resolution?\n  \n  detection:\n    how_detected: \"\"  # alert | user report | manual check\n    time_to_detect_minutes: 0\n    could_have_detected_sooner: \"\"\n  \n  resolution:\n    how_resolved: \"\"\n    time_to_mitigate_minutes: 0\n    time_to_resolve_minutes: 0\n  \n  what_went_well:\n    - \"\"  # Explicitly call out what worked\n  \n  what_went_wrong:\n    - \"\"\n  \n  where_we_got_lucky:\n    - \"\"  # Things that could have made it worse\n  \n  action_items:\n    - id: \"AI-001\"\n      type: \"\"  # prevent | detect | mitigate | process\n      description: \"\"\n      owner: \"\"\n      priority: \"\"  # P0 | P1 | P2\n      deadline: \"\"\n      status: \"open\"  # open | in-progress | done\n      ticket: \"\"\n```\n\n### Root Cause Analysis Methods\n\n**Five Whys (simple incidents):**\n1. Why did users see errors? → API returned 500s\n2. Why did API return 500s? → Database connection pool exhausted\n3. Why was pool exhausted? → Long-running query held connections\n4. Why was query long-running? → Missing index on new column\n5. Why was index missing? → Migration didn't include index; no query performance review in CI\n\n→ **Root cause:** No automated query performance check in deployment pipeline\n→ **Action:** Add query plan analysis to CI for migration PRs\n\n**Fishbone / Ishikawa (complex incidents):**\n\n```\nCategories to investigate:\n├── People: Training? Fatigue? Communication?\n├── Process: Runbook? Escalation? Change management?\n├── Technology: Bug? Config? Capacity? Dependency?\n├── Environment: Network? Cloud provider? Third party?\n├── Monitoring: Detection gap? Alert fatigue? Dashboard gap?\n└── Testing: Test coverage? Load testing? Chaos testing?\n```\n\n**Contributing Factor Categories:**\n| Category | Questions |\n|----------|-----------|\n| Trigger | What change or event started it? |\n| Propagation | Why did it spread? Why wasn't it contained? |\n| Detection | Why wasn't it caught earlier? |\n| Resolution | What slowed the fix? |\n| Process | What process gaps contributed? |\n\n### Postmortem Review Meeting (60 min)\n\n```\n1. Timeline walk-through (15 min)\n   - Author presents chronology\n   - Attendees add context (\"I remember seeing X at this point\")\n\n2. Root cause deep-dive (15 min)  \n   - Do we agree on root cause?\n   - Are there additional contributing factors?\n\n3. Action item review (20 min)\n   - Are these the RIGHT actions?\n   - Are they prioritized correctly?\n   - Do owners agree on deadlines?\n\n4. Process improvements (10 min)\n   - Could we have detected this sooner?\n   - Could we have resolved this faster?\n   - What would have prevented this entirely?\n```\n\n---\n\n## Phase 7: Chaos Engineering\n\n### Chaos Maturity Model\n\n| Level | Name | Activities |\n|-------|------|-----------|\n| 0 | None | No chaos testing |\n| 1 | Exploratory | Manual fault injection in staging |\n| 2 | Systematic | Scheduled chaos experiments in staging |\n| 3 | Production | Controlled chaos in production (Game Days) |\n| 4 | Continuous | Automated chaos in production with safety controls |\n\n### Chaos Experiment Template\n\n```yaml\nexperiment:\n  name: \"\"\n  hypothesis: \"When [fault], the system will [expected behavior]\"\n  \n  steady_state:\n    metrics:\n      - name: \"\"\n        baseline: \"\"\n        acceptable_range: \"\"\n  \n  method:\n    fault_type: \"\"  # network | compute | storage | dependency | data\n    target: \"\"      # which service/component\n    blast_radius: \"\"  # single pod | single AZ | percentage of traffic\n    duration: \"\"\n    \n  safety:\n    abort_conditions:\n      - \"SLO burn rate exceeds 10x\"\n      - \"Customer-visible errors detected\"\n      - \"Alert fires that we didn't expect\"\n    rollback_plan: \"\"\n    required_approvals: []\n    \n  results:\n    outcome: \"\"  # confirmed | disproved | inconclusive\n    observations: []\n    action_items: []\n```\n\n### Chaos Experiment Library\n\n| Category | Experiment | Validates |\n|----------|-----------|-----------|\n| **Network** | Add 200ms latency to DB calls | Timeout handling, circuit breakers |\n| **Network** | Drop 5% of packets to downstream | Retry logic, error handling |\n| **Network** | DNS resolution failure | Caching, fallback, error messages |\n| **Compute** | Kill random pod every 10 min | Auto-restart, load balancing |\n| **Compute** | CPU stress to 95% on 1 node | Auto-scaling, graceful degradation |\n| **Compute** | Fill disk to 95% | Disk monitoring, log rotation, alerts |\n| **Storage** | Increase DB latency 5x | Connection pool handling, timeouts |\n| **Storage** | Simulate cache failure (Redis down) | Cache-aside pattern, DB fallback |\n| **Dependency** | Block external API (payment provider) | Circuit breaker, queuing, retry |\n| **Dependency** | Return 429s from auth service | Rate limit handling, backoff |\n| **Data** | Clock skew on subset of nodes | Timestamp handling, ordering |\n| **Scale** | 10x traffic spike over 5 minutes | Auto-scaling speed, queue depth |\n\n### Game Day Runbook\n\n```\nPRE-GAME (1 week before):\n□ Experiment designed and reviewed\n□ Steady-state metrics identified\n□ Abort conditions defined\n□ All participants briefed\n□ Runbacks tested in staging\n□ Stakeholders notified\n\nGAME DAY:\n□ Verify steady state (15 min baseline)\n□ Announce in #engineering: \"Chaos Game Day starting\"\n□ Inject fault\n□ Observe and document\n□ If abort condition hit → rollback immediately\n□ Run for planned duration\n□ Remove fault\n□ Verify recovery to steady state\n\nPOST-GAME (same day):\n□ Results documented\n□ Surprises noted\n□ Action items created\n□ Share findings in team meeting\n```\n\n---\n\n## Phase 8: Toil Management\n\n### Toil Identification\n\n**Definition:** Work that is manual, repetitive, automatable, tactical, without enduring value, and scales linearly with service growth.\n\n### Toil Inventory Template\n\n```yaml\ntoil_item:\n  name: \"\"\n  category: \"\"  # deployment | scaling | config | data | access | monitoring | recovery\n  frequency: \"\"  # daily | weekly | monthly | per-incident\n  time_per_occurrence_min: 0\n  occurrences_per_month: 0\n  total_hours_per_month: 0\n  teams_affected: []\n  automation_difficulty: \"\"  # low | medium | high\n  automation_value: 0  # hours saved per month\n  priority_score: 0  # value / difficulty\n```\n\n### Toil Reduction Priority Matrix\n\n| | Low Effort | Medium Effort | High Effort |\n|-|-----------|--------------|-------------|\n| **High Value** (>10 hrs/mo) | DO FIRST | DO SECOND | PLAN |\n| **Med Value** (2-10 hrs/mo) | DO SECOND | PLAN | EVALUATE |\n| **Low Value** (<2 hrs/mo) | QUICK WIN | SKIP | SKIP |\n\n### Common Toil Targets (Ranked by Impact)\n\n1. **Manual deployments** → CI/CD pipeline + GitOps\n2. **Access provisioning** → Self-service + auto-approval for low-risk\n3. **Certificate renewals** → Auto-renewal (cert-manager, Let's Encrypt)\n4. **Scaling decisions** → HPA + predictive auto-scaling\n5. **Log investigation** → Structured logging + correlation + dashboards\n6. **Data fixes** → Self-service admin tools + validation at ingestion\n7. **Config changes** → Config-as-code + automated rollout\n8. **Incident response** → Automated runbooks for known issues\n9. **Capacity reporting** → Automated dashboards + forecasting\n10. **On-call triage** → Noise reduction + auto-remediation for known patterns\n\n### Toil Budget Rule\n**Target: <25% of SRE time spent on toil.** Track monthly. If above 25%, prioritize automation over all feature work.\n\n---\n\n## Phase 9: Capacity Planning\n\n### Capacity Model Template\n\n```yaml\ncapacity_model:\n  service: \"\"\n  bottleneck_resource: \"\"  # CPU | memory | storage | connections | bandwidth\n  \n  current_state:\n    peak_utilization_pct: 0\n    headroom_pct: 0\n    cost_per_month_usd: 0\n    \n  growth_forecast:\n    metric: \"\"  # MAU | requests/sec | storage_gb\n    current: 0\n    monthly_growth_pct: 0\n    projected_6mo: 0\n    projected_12mo: 0\n    \n  scaling_strategy:\n    type: \"\"  # horizontal | vertical | hybrid\n    auto_scaling: true\n    min_instances: 0\n    max_instances: 0\n    scale_up_threshold: 80  # % utilization\n    scale_down_threshold: 30\n    cooldown_seconds: 300\n    \n  cost_projection:\n    current_monthly: 0\n    projected_6mo_monthly: 0\n    projected_12mo_monthly: 0\n```\n\n### Capacity Planning Cadence\n\n| Frequency | Action |\n|-----------|--------|\n| Daily | Review auto-scaling events, check for anomalies |\n| Weekly | Review utilization trends, spot-check headroom |\n| Monthly | Update growth model, review cost projections |\n| Quarterly | Full capacity review, budget planning, architecture check |\n| Pre-launch | Load test to 2x expected peak, verify scaling |\n\n### Load Testing Benchmarks\n\n| Scenario | Method | Duration | Target |\n|----------|--------|----------|--------|\n| Baseline | Steady load at current peak | 30 min | Establish metrics |\n| Growth | 2x current peak | 15 min | Verify scaling works |\n| Spike | 10x normal in 60 seconds | 5 min | Circuit breakers hold |\n| Soak | 1.5x normal load | 4 hours | No memory leaks, degradation |\n| Stress | Ramp until failure | Until break | Find actual limits |\n\n---\n\n## Phase 10: On-Call Excellence\n\n### On-Call Health Metrics\n\n| Metric | Healthy | Warning | Critical |\n|--------|---------|---------|----------|\n| Pages per shift | <2 | 2-5 | >5 |\n| Off-hours pages | <1/week | 1-3/week | >3/week |\n| Time to acknowledge | <5 min | 5-15 min | >15 min |\n| Time to mitigate | <30 min | 30-60 min | >60 min |\n| False positive rate | <10% | 10-30% | >30% |\n| Escalation rate | <20% | 20-40% | >40% |\n| On-call satisfaction | >4/5 | 3-4/5 | <3/5 |\n\n### On-Call Rotation Best Practices\n\n1. **Minimum rotation size: 5 people** (one week on, four weeks off)\n2. **No back-to-back weeks** unless team is too small (fix the team size)\n3. **Follow-the-sun** for global teams (no one pages at 3 AM if avoidable)\n4. **Primary + secondary** on-call always\n5. **Handoff document** at rotation change — open issues, recent deploys, known risks\n6. **Compensation** — on-call pay, time off in lieu, or equivalent\n\n### On-Call Handoff Template\n\n```\n## On-Call Handoff: [Date]\n\n### Open Issues\n- [Issue]: [Status, next steps]\n\n### Recent Changes (last 7 days)\n- [Deployment/config change]: [Risk level, rollback plan]\n\n### Known Risks\n- [Event/condition]: [What to watch for]\n\n### Scheduled Maintenance\n- [When]: [What, duration, rollback plan]\n\n### Runbook Updates\n- [Any new/updated runbooks since last rotation]\n```\n\n### Runbook Template\n\n```yaml\nrunbook:\n  title: \"\"\n  alert_name: \"\"  # exact alert that triggers this\n  last_updated: \"\"\n  owner: \"\"\n  \n  overview: |\n    What this alert means in plain English.\n    \n  impact: |\n    What users/systems are affected and how.\n    \n  diagnosis:\n    - step: \"Check service health\"\n      command: \"\"\n      expected: \"\"\n      if_unexpected: \"\"\n    - step: \"Check recent deployments\"\n      command: \"\"\n      expected: \"\"\n      if_unexpected: \"Rollback: [command]\"\n    - step: \"Check dependencies\"\n      command: \"\"\n      expected: \"\"\n      if_unexpected: \"\"\n      \n  mitigation:\n    - option: \"Rollback\"\n      when: \"Recent deployment suspected\"\n      steps: []\n    - option: \"Scale up\"\n      when: \"Traffic spike\"\n      steps: []\n    - option: \"Failover\"\n      when: \"Single component failure\"\n      steps: []\n      \n  escalation:\n    after_minutes: 30\n    contact: \"\"\n    context_to_provide: \"\"\n```\n\n---\n\n## Phase 11: Reliability Review & Governance\n\n### Weekly SRE Review (30 min)\n\n```\n1. SLO Status (5 min)\n   - Budget remaining per service\n   - Any burn rate alerts this week?\n\n2. Incident Review (10 min)\n   - Incidents this week: count, severity, duration\n   - Open postmortem action items: status check\n\n3. On-Call Health (5 min)\n   - Pages this week (total, off-hours, false positives)\n   - Any on-call feedback?\n\n4. Reliability Work (10 min)\n   - Automation shipped this week\n   - Toil reduced (hours saved)\n   - Chaos experiments run\n   - Capacity concerns\n```\n\n### Monthly Reliability Report\n\n```yaml\nmonthly_report:\n  period: \"\"\n  \n  slo_summary:\n    services_meeting_slo: 0\n    services_breaching_slo: 0\n    worst_performing: \"\"\n    \n  incidents:\n    total: 0\n    by_severity: { SEV1: 0, SEV2: 0, SEV3: 0, SEV4: 0 }\n    mttr_minutes: 0\n    mttd_minutes: 0\n    repeat_incidents: 0\n    \n  error_budget:\n    services_in_healthy: 0\n    services_in_warning: 0\n    services_in_critical: 0\n    services_exhausted: 0\n    \n  toil:\n    hours_spent: 0\n    hours_automated_away: 0\n    pct_of_sre_time: 0\n    \n  on_call:\n    total_pages: 0\n    off_hours_pages: 0\n    false_positive_pct: 0\n    avg_ack_time_min: 0\n    \n  action_items:\n    open: 0\n    completed_this_month: 0\n    overdue: 0\n    \n  highlights: []\n  concerns: []\n  next_month_priorities: []\n```\n\n### Production Readiness Review Checklist\n\nBefore any new service goes to production:\n\n| Category | Check | Status |\n|----------|-------|--------|\n| **SLOs** | SLIs defined and measured | |\n| **SLOs** | SLO targets set with stakeholder agreement | |\n| **SLOs** | Error budget policy documented | |\n| **Monitoring** | Golden signals dashboarded | |\n| **Monitoring** | Alerting configured with runbooks | |\n| **Monitoring** | Structured logging implemented | |\n| **Monitoring** | Distributed tracing enabled | |\n| **Incidents** | On-call rotation established | |\n| **Incidents** | Escalation paths documented | |\n| **Incidents** | Runbooks for top 5 failure modes | |\n| **Capacity** | Load tested to 2x expected peak | |\n| **Capacity** | Auto-scaling configured and tested | |\n| **Capacity** | Resource limits set (CPU, memory) | |\n| **Resilience** | Graceful degradation implemented | |\n| **Resilience** | Circuit breakers for dependencies | |\n| **Resilience** | Retry with exponential backoff | |\n| **Resilience** | Timeout configured for all external calls | |\n| **Deploy** | Rollback tested and documented | |\n| **Deploy** | Canary/blue-green deployment ready | |\n| **Deploy** | Feature flags for risky features | |\n| **Security** | Authentication and authorization | |\n| **Security** | Secrets in vault (not env vars) | |\n| **Security** | Dependencies scanned | |\n| **Data** | Backup and restore tested | |\n| **Data** | Data retention policy defined | |\n| **Docs** | Architecture diagram current | |\n| **Docs** | API documentation published | |\n| **Docs** | Operational runbook complete | |\n\n---\n\n## Phase 12: Advanced Patterns\n\n### Self-Healing Automation\n\n```yaml\nauto_remediation:\n  - trigger: \"pod_crash_loop\"\n    condition: \"restart_count > 3 in 10 min\"\n    action: \"Delete pod, let scheduler reschedule\"\n    escalate_if: \"Still crashing after 3 auto-remediations\"\n    \n  - trigger: \"disk_usage_high\"\n    condition: \"disk_usage > 85%\"\n    action: \"Run log cleanup script, archive old data\"\n    escalate_if: \"Still above 85% after cleanup\"\n    \n  - trigger: \"connection_pool_exhausted\"\n    condition: \"available_connections = 0\"\n    action: \"Kill idle connections, increase pool temporarily\"\n    escalate_if: \"Pool exhausted again within 1 hour\"\n    \n  - trigger: \"certificate_expiring\"\n    condition: \"days_until_expiry < 14\"\n    action: \"Trigger cert renewal\"\n    escalate_if: \"Renewal fails\"\n```\n\n### Multi-Region Reliability\n\n| Strategy | Complexity | RTO | Cost |\n|----------|-----------|-----|------|\n| Active-passive | Low | Minutes | 1.5x |\n| Active-active read | Medium | Seconds | 1.8x |\n| Active-active full | High | Near-zero | 2-3x |\n| Cell-based | Very high | Per-cell | 2-4x |\n\n**Decision guide:**\n- SLO < 99.9% → Single region with good backups\n- SLO 99.9-99.95% → Active-passive with automated failover\n- SLO > 99.95% → Active-active (read or full)\n- SLO > 99.99% → Cell-based architecture\n\n### Reliability Culture Indicators\n\n**Healthy signals:**\n- Postmortems are blameless and well-attended\n- Error budgets are respected (feature freeze actually happens)\n- On-call is shared fairly and compensated\n- Toil is tracked and reducing quarter-over-quarter\n- Chaos experiments happen regularly\n- Teams own their reliability (not just SRE)\n\n**Warning signs:**\n- \"Hero culture\" — same person always saves the day\n- Postmortems are blame-focused or skipped\n- Error budget exhaustion doesn't change behavior\n- On-call is dreaded, same 2 people always paged\n- \"We'll fix reliability after this feature ships\" (always)\n- SRE team is just an ops team with a new name\n\n---\n\n## Quality Scoring Rubric (0-100)\n\n| Dimension | Weight | 0-2 | 3-4 | 5 |\n|-----------|--------|-----|-----|---|\n| SLO Coverage | 20% | No SLOs | SLOs for critical services | All services with SLOs, error budgets, reviews |\n| Monitoring | 15% | Basic health checks | Golden signals + dashboards | Full observability stack + anomaly detection |\n| Incident Response | 15% | Ad-hoc, no process | ICS roles, runbooks, postmortems | Structured ICS, blameless culture, action tracking |\n| Automation | 15% | Manual everything | CI/CD + some automation | Self-healing, GitOps, <25% toil |\n| Chaos Engineering | 10% | None | Staging experiments | Continuous production chaos with safety |\n| Capacity Planning | 10% | Reactive | Quarterly forecasting | Predictive, auto-scaling, cost-optimized |\n| On-Call Health | 10% | Burnout, hero culture | Fair rotation, <5 pages/shift | Balanced, compensated, <2 pages/shift |\n| Documentation | 5% | Nothing written | Runbooks exist | Complete, current, tested runbooks |\n\n---\n\n## Natural Language Commands\n\n- \"Assess reliability for [service]\" → Run maturity assessment\n- \"Define SLOs for [service]\" → Walk through SLI selection + SLO setting\n- \"Check error budget for [service]\" → Calculate current budget status\n- \"Start incident for [description]\" → Create incident channel, assign IC, begin workflow\n- \"Write postmortem for [incident]\" → Generate structured postmortem\n- \"Plan chaos experiment for [service]\" → Design experiment with hypothesis\n- \"Audit toil for [team]\" → Inventory and prioritize toil\n- \"Review on-call health\" → Analyze page volume, satisfaction, fairness\n- \"Production readiness review for [service]\" → Run full checklist\n- \"Monthly reliability report\" → Generate comprehensive report\n- \"Design runbook for [alert]\" → Create structured runbook\n- \"Plan capacity for [service] growing at [X%]\" → Build capacity model\n","readmeExcerpt":"SRE & Incident Management Platform Complete Site Reliability Engineering system — from SLO definition through incident response, chaos engineering, and operational excellence. Zero dependencies. --- Phase 1: Reliability Assessment Before building anything, assess where you are. Service Catalog Entry Maturity Assessment (Score 1-5 per dimension) | Dimension | 1 (Ad-hoc) | 3 (Defined) | 5 (Optimized) | Score | |-------","codeSnippets":[],"executableExamples":[{"language":"yaml","snippet":"service:\n  name: \"\"\n  tier: \"\"  # critical | important | standard | experimental\n  owner_team: \"\"\n  oncall_rotation: \"\"\n  dependencies:\n    upstream: []    # services we call\n    downstream: []  # services that call us\n  data_classification: \"\"  # public | internal | confidential | restricted\n  deployment_frequency: \"\"  # daily | weekly | biweekly | monthly\n  architecture: \"\"  # monolith | microservice | serverless | hybrid\n  language: \"\"\n  infra: \"\"  # k8s | ECS | Lambda | VM | bare-metal\n  traffic_pattern: \"\"  # steady | diurnal | spiky | seasonal\n  peak_rps: 0\n  storage_gb: 0\n  monthly_cost_usd: 0"},{"language":"yaml","snippet":"sli:\n  name: \"request_success_rate\"\n  description: \"Proportion of valid requests served successfully\"\n  type: \"availability\"  # availability | latency | quality | freshness\n  measurement:\n    good_events: \"HTTP responses with status < 500\"\n    total_events: \"All HTTP requests excluding health checks\"\n    source: \"load balancer access logs\"\n    aggregation: \"sum(good) / sum(total) over rolling 28-day window\"\n  exclusions:\n    - \"Health check endpoints (/healthz, /readyz)\"\n    - \"Synthetic monitoring traffic\"\n    - \"Requests from blocked IPs\"\n    - \"4xx responses (client errors)\""},{"language":"yaml","snippet":"slo:\n  service: \"\"\n  sli: \"\"\n  target: 99.9  # percentage\n  window: \"28d\"  # rolling window\n  error_budget: 0.1  # 100% - target\n  error_budget_minutes: 40  # per 28-day window\n  \n  burn_rate_alerts:\n    - name: \"fast_burn\"\n      burn_rate: 14.4  # exhausts budget in 2 hours\n      short_window: \"5m\"\n      long_window: \"1h\"\n      severity: \"page\"\n    - name: \"medium_burn\"\n      burn_rate: 6.0   # exhausts budget in ~5 hours\n      short_window: \"30m\"\n      long_window: \"6h\"\n      severity: \"page\"\n    - name: \"slow_burn\"\n      burn_rate: 1.0   # exhausts budget in 28 days\n      short_window: \"6h\"\n      long_window: \"3d\"\n      severity: \"ticket\"\n  \n  review_cadence: \"monthly\"\n  owner: \"\"\n  stakeholders: []\n  \n  escalation_when_budget_exhausted:\n    - \"Halt non-critical deployments\"\n    - \"Redirect engineering to reliability work\"\n    - \"Escalate to VP Engineering if no improvement in 48h\""},{"language":"yaml","snippet":"error_budget_policy:\n  service: \"\"\n  \n  budget_states:\n    healthy:\n      condition: \"remaining_budget > 50%\"\n      actions:\n        - \"Normal development velocity\"\n        - \"Feature work prioritized\"\n        - \"Chaos experiments allowed\"\n    \n    warning:\n      condition: \"remaining_budget 25-50%\"\n      actions:\n        - \"Increase monitoring scrutiny\"\n        - \"Review recent changes for risk\"\n        - \"Limit risky deployments to business hours\"\n        - \"No chaos experiments\"\n    \n    critical:\n      condition: \"remaining_budget 0-25%\"\n      actions:\n        - \"Feature freeze — reliability work only\"\n        - \"All deployments require SRE approval\"\n        - \"Mandatory rollback plan for every change\"\n        - \"Daily error budget review\"\n    \n    exhausted:\n      condition: \"remaining_budget <= 0\"\n      actions:\n        - \"Complete deployment freeze\"\n        - \"All engineering redirected to reliability\"\n        - \"VP Engineering notified\"\n        - \"Postmortem required for budget exhaustion\"\n        - \"Freeze maintained until budget recovers to 10%\"\n  \n  exceptions:\n    - \"Security patches always allowed\"\n    - \"Regulatory compliance changes always allowed\"\n    - \"Data loss prevention always allowed\"\n  \n  reset: \"Rolling 28-day window (no manual resets)\""},{"language":"text","snippet":"Burn rate = (error rate observed) / (error rate allowed by SLO)\n\nExample:\n- SLO: 99.9% (error budget = 0.1%)\n- Current error rate: 0.5%\n- Burn rate = 0.5% / 0.1% = 5x\n\nAt 5x burn rate → budget exhausted in 28d / 5 = 5.6 days"},{"language":"json","snippet":"{\n  \"timestamp\": \"2026-02-17T11:24:00.000Z\",\n  \"level\": \"error\",\n  \"service\": \"payment-api\",\n  \"trace_id\": \"abc123\",\n  \"span_id\": \"def456\",\n  \"message\": \"Payment processing failed\",\n  \"error_type\": \"TimeoutException\",\n  \"error_message\": \"Gateway timeout after 30s\",\n  \"http_method\": \"POST\",\n  \"http_path\": \"/api/v1/payments\",\n  \"http_status\": 504,\n  \"duration_ms\": 30012,\n  \"customer_id\": \"cust_xxx\",\n  \"payment_id\": \"pay_yyy\",\n  \"amount_cents\": 4999,\n  \"retry_count\": 2,\n  \"environment\": \"production\",\n  \"host\": \"payment-api-7b4d9-xk2p1\",\n  \"region\": \"us-east-1\"\n}"}],"parameters":{},"dependencies":[],"permissions":[],"extractedFiles":[],"languages":["typescript"],"docsSourceLabel":"CLAWHUB","editorialOverview":"SRE & Incident Management Platform SRE & Incident Management Platform Complete Site Reliability Engineering system — from SLO definition through incident response, chaos engineering, and operational excellence. Zero dependencies. --- Phase 1: Reliability Assessment Before building anything, assess where you are. Service Catalog Entry Maturity Assessment (Score 1-5 per dimension) | Dimension | 1 (Ad-hoc) | 3 (Defined) | 5 (Optimized) | Score | |-------","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":362,"uniquenessScore":68,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T19:11:54.646Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}