{"id":"5e7fc6ca-0100-4dee-b50b-96260a8844f7","entityType":"agent","slug":"clawhub-skills-1kalin-afrexai-observability-engine","name":"afrexai-observability-engine","canonicalUrl":"https://www.xpersona.co/agent/clawhub-skills-1kalin-afrexai-observability-engine","canonicalPath":"/agent/clawhub-skills-1kalin-afrexai-observability-engine","generatedAt":"2026-10-10T06:20:52.265Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":null},"description":"Complete observability & reliability engineering system. Use when designing monitoring, implementing structured logging, setting up distributed tracing, building alerting systems, creating SLO/SLI frameworks, running incident response, conducting post-mortems, or auditing system reliability. Covers all three pillars (logs/metrics/traces), alert design, dashboard architecture, on-call operations, chaos engineering, and cost optimization. --- name: afrexai-observability-engine model: standard description: Complete observability & reliability engineering system. Use when designing monitoring, implementing structured logging, setting up distributed tracing, building alerting systems, creating SLO/SLI frameworks, running incident response, conducting post-mortems, or auditing system reliability. Covers all three pillars (logs/metrics/traces), alert desig","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. Last updated 4/15/2026.","installCommand":"clawhub skill install skills:1kalin:afrexai-observability-engine","sourceUrl":"https://github.com/openclaw/skills/tree/main/skills/1kalin/afrexai-observability-engine","homepage":null,"primaryLinks":[{"label":"View on ClawHub","url":"https://github.com/openclaw/skills/tree/main/skills/1kalin/afrexai-observability-engine","kind":"source"}],"safetyScore":84,"overallRank":62,"popularityScore":50,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Complete observability & reliability engineering system. Use when designing monitoring, implementing structured logging, setting up distributed tracing, buildin"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[{"label":"limits","status":"self-declared"},{"label":"be","status":"self-declared"},{"label":"b3","status":"self-declared"},{"label":"team","status":"self-declared"},{"label":"escalates","status":"self-declared"},{"label":"ticket","status":"self-declared"}],"verifiedCount":0,"selfDeclaredCount":7,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"},{"key":"limits","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"},{"key":"be","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"},{"key":"b3","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"},{"key":"team","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"},{"key":"escalates","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"},{"key":"ticket","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile capability:limits|supported|profile capability:be|supported|profile capability:b3|supported|profile capability:team|supported|profile capability:escalates|supported|profile capability:ticket|supported|profile"}},"adoption":{"evidence":{"source":"no-adoption-signals","verified":false,"confidence":"low","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":"No source adoption metrics were available."},"stars":null,"forks":null,"downloads":null,"packageName":null,"latestVersion":null,"tractionLabel":null},"release":{"evidence":{"source":"agent-index","verified":false,"confidence":"medium","updatedAt":"2026-02-25T06:17:46.150Z","emptyReason":null},"lastUpdatedAt":"2026-04-15T00:45:39.800Z","lastCrawledAt":"2026-02-25T06:17:46.150Z","lastIndexedAt":null,"nextCrawlAt":"2026-02-26T06:17:46.150Z","lastVerifiedAt":null,"highlights":[]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install skills:1kalin:afrexai-observability-engine","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-observability-engine/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-observability-engine/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-observability-engine/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-observability-engine/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-observability-engine/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-observability-engine/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T06:20:52.264Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-observability-engine/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-observability-engine/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-observability-engine/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-observability-engine/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":null},"readme":"---\nname: afrexai-observability-engine\nmodel: standard\ndescription: Complete observability & reliability engineering system. Use when designing monitoring, implementing structured logging, setting up distributed tracing, building alerting systems, creating SLO/SLI frameworks, running incident response, conducting post-mortems, or auditing system reliability. Covers all three pillars (logs/metrics/traces), alert design, dashboard architecture, on-call operations, chaos engineering, and cost optimization.\nversion: 1.0.0\ntags: observability, monitoring, logging, tracing, alerting, SRE, incident-response, SLO, metrics, devops, reliability, on-call, post-mortem, dashboards\n---\n\n# Observability & Reliability Engineering\n\nComplete system for building observable, reliable services — from structured logging to incident response to SLO-driven development.\n\n---\n\n## Quick Health Check (/16)\n\nScore your current observability posture:\n\n| Signal | Healthy (2) | Weak (1) | Missing (0) |\n|--------|-------------|----------|-------------|\n| Structured logging | JSON logs with trace_id correlation | Logs exist but unstructured | Console.log / print statements |\n| Metrics collection | RED/USE metrics with dashboards | Some metrics, no dashboards | No metrics |\n| Distributed tracing | Full request path with sampling | Partial traces, key services only | No tracing |\n| Alerting | SLO-based alerts with runbooks | Threshold alerts, some runbooks | No alerts or all-noise |\n| Incident response | Defined process with roles + post-mortems | Ad-hoc response, some docs | \"Whoever notices fixes it\" |\n| SLOs defined | SLOs with error budgets tracked weekly | Informal availability targets | No reliability targets |\n| On-call rotation | Structured rotation with escalation | Informal \"call someone\" | No on-call |\n| Cost management | Observability budget tracked monthly | Some awareness of costs | No idea what you spend |\n\n**12-16:** Production-grade. Focus on optimization.\n**8-11:** Foundation exists. Fill the gaps systematically.\n**4-7:** Significant risk. Prioritize alerting + incident response.\n**0-3:** Flying blind. Start with Phase 1 immediately.\n\n---\n\n## Phase 1: Structured Logging\n\n### Log Architecture\n\n```\nApplication → Structured JSON → Log Router → Storage → Query Engine\n                                    ↓\n                              Alert Pipeline\n```\n\n### Required Fields (Every Log Line)\n\n| Field | Type | Purpose | Example |\n|-------|------|---------|---------|\n| `timestamp` | ISO-8601 UTC | When | `2026-02-22T18:30:00.123Z` |\n| `level` | enum | Severity | `info`, `warn`, `error`, `fatal` |\n| `service` | string | Which service | `payment-api` |\n| `version` | string | Which deploy | `v2.3.1` |\n| `environment` | string | Which env | `production` |\n| `message` | string | What happened | `Payment processed successfully` |\n| `trace_id` | string | Request correlation | `abc123def456` |\n| `span_id` | string | Operation within trace | `span_789` |\n| `duration_ms` | number | How long | `142` |\n\n### Contextual Fields (Add Per Domain)\n\n```yaml\n# HTTP request context\nhttp:\n  method: POST\n  path: /api/v1/orders\n  status: 201\n  client_ip: 203.0.113.42  # Anonymize in logs if needed\n  user_agent: \"Mozilla/5.0...\"\n  request_id: \"req_abc123\"\n\n# Business context\nbusiness:\n  user_id: \"usr_456\"\n  tenant_id: \"tenant_789\"\n  order_id: \"ord_012\"\n  action: \"checkout\"\n  amount_cents: 4999\n  currency: \"USD\"\n\n# Error context\nerror:\n  type: \"PaymentDeclinedError\"\n  message: \"Card declined: insufficient funds\"\n  code: \"CARD_DECLINED\"\n  stack: \"...\" # Only in non-production or DEBUG level\n  retry_count: 2\n  retryable: true\n```\n\n### Log Level Decision Tree\n\n```\nIs the process about to crash?\n  → FATAL (exit after logging)\n\nDid an operation fail that needs human attention?\n  → ERROR (page someone or create ticket)\n\nDid something unexpected happen but we recovered?\n  → WARN (review in daily triage)\n\nIs this a normal business event worth recording?\n  → INFO (audit trail, business metrics)\n\nIs this useful for debugging but noisy in production?\n  → DEBUG (off in prod, on in staging)\n\nIs this only useful when stepping through code?\n  → TRACE (never in production)\n```\n\n### Log Level Rules\n\n1. **ERROR means action required** — if no one needs to act on it, it's WARN\n2. **INFO is for business events** — not internal implementation details\n3. **No logging inside tight loops** — aggregate and log summary\n4. **Log at boundaries** — API entry/exit, queue consume/publish, DB calls\n5. **Never log secrets** — API keys, tokens, passwords, PII (see scrubbing below)\n\n### PII & Secret Scrubbing\n\n```yaml\nscrub_patterns:\n  # Always redact\n  - field_patterns: [\"password\", \"secret\", \"token\", \"api_key\", \"authorization\"]\n    action: replace_with_redacted\n  \n  # Hash for correlation without exposure\n  - field_patterns: [\"email\", \"phone\", \"ssn\", \"national_id\"]\n    action: sha256_hash\n  \n  # Mask partially\n  - field_patterns: [\"credit_card\", \"card_number\"]\n    action: mask_last_4  # \"****-****-****-1234\"\n  \n  # IP anonymization\n  - field_patterns: [\"client_ip\", \"ip_address\"]\n    action: zero_last_octet  # 203.0.113.0\n```\n\n### Logger Setup (By Language)\n\n**Node.js (Pino):**\n```typescript\nimport pino from 'pino';\nimport { AsyncLocalStorage } from 'node:async_hooks';\n\nconst als = new AsyncLocalStorage<Record<string, string>>();\n\nconst logger = pino({\n  level: process.env.LOG_LEVEL || 'info',\n  formatters: {\n    level: (label) => ({ level: label }),\n  },\n  mixin: () => als.getStore() ?? {},\n  redact: ['req.headers.authorization', '*.password', '*.token'],\n  timestamp: pino.stdTimeFunctions.isoTime,\n});\n\n// Middleware: inject context\napp.use((req, res, next) => {\n  const ctx = {\n    trace_id: req.headers['x-trace-id'] || crypto.randomUUID(),\n    request_id: crypto.randomUUID(),\n    service: 'payment-api',\n    version: process.env.APP_VERSION,\n  };\n  als.run(ctx, () => next());\n});\n```\n\n**Python (structlog):**\n```python\nimport structlog\nstructlog.configure(\n    processors=[\n        structlog.contextvars.merge_contextvars,\n        structlog.processors.add_log_level,\n        structlog.processors.TimeStamper(fmt=\"iso\", utc=True),\n        structlog.processors.JSONRenderer(),\n    ],\n)\nlog = structlog.get_logger()\n# Bind context per-request:\nstructlog.contextvars.bind_contextvars(trace_id=trace_id, user_id=user_id)\n```\n\n**Go (zerolog):**\n```go\nlog := zerolog.New(os.Stdout).With().\n    Timestamp().\n    Str(\"service\", \"payment-api\").\n    Str(\"version\", version).\n    Logger()\n// Per-request:\nreqLog := log.With().Str(\"trace_id\", traceID).Logger()\n```\n\n### Log Storage Decision\n\n| Volume | Solution | Retention | Cost |\n|--------|----------|-----------|------|\n| <10 GB/day | Loki + Grafana | 30 days hot, 90 days cold | Low |\n| 10-100 GB/day | Elasticsearch / OpenSearch | 14 days hot, 90 days S3 | Medium |\n| 100+ GB/day | ClickHouse or Datadog | 7 days hot, 30 days archive | High |\n| Budget-constrained | Loki + S3 backend | 90 days all cold | Very low |\n\n### 10 Logging Anti-Patterns\n\n| # | Anti-Pattern | Fix |\n|---|-------------|-----|\n| 1 | `log.error(err)` with no context | Always include: what operation, what input, what state |\n| 2 | Logging request/response bodies | Log only in DEBUG; redact sensitive fields |\n| 3 | String concatenation in log messages | Use structured fields: `log.info(\"processed\", { order_id, amount })` |\n| 4 | Catch-and-log-and-rethrow | Log at the boundary where you handle it, not every layer |\n| 5 | Different log formats per service | Standardize schema across all services |\n| 6 | No log rotation / retention policy | Set max size + TTL; archive to cold storage |\n| 7 | Logging inside hot paths | Aggregate: log summary every N items or every interval |\n| 8 | Missing correlation IDs | Propagate trace_id from first entry point through all services |\n| 9 | Boolean log levels (`verbose: true`) | Use standard levels with configurable minimum |\n| 10 | Logging PII in plain text | Implement scrubbing at the logger level |\n\n---\n\n## Phase 2: Metrics Collection\n\n### The RED Method (Request-Driven Services)\n\nFor every service endpoint, track:\n\n| Metric | What | Prometheus Example |\n|--------|------|--------------------|\n| **R**ate | Requests per second | `http_requests_total{method, path, status}` |\n| **E**rrors | Failed requests per second | `http_requests_total{status=~\"5..\"}` / total |\n| **D**uration | Latency distribution | `http_request_duration_seconds{method, path}` (histogram) |\n\n### The USE Method (Infrastructure Resources)\n\nFor every resource (CPU, memory, disk, network):\n\n| Metric | What | Example |\n|--------|------|---------|\n| **U**tilization | % resource busy | CPU usage 78% |\n| **S**aturation | Queue depth / backpressure | 12 requests queued |\n| **E**rrors | Resource errors | 3 disk I/O errors |\n\n### Golden Signals (Google SRE)\n\n| Signal | Meaning | Source |\n|--------|---------|--------|\n| Latency | Time to serve requests | RED Duration |\n| Traffic | Demand on the system | RED Rate |\n| Errors | Rate of failed requests | RED Errors |\n| Saturation | How \"full\" the service is | USE Saturation |\n\n### Metric Types & When to Use Each\n\n| Type | Use Case | Example |\n|------|----------|---------|\n| **Counter** | Things that only go up | Total requests, errors, bytes sent |\n| **Gauge** | Current value that goes up/down | Active connections, queue depth, temperature |\n| **Histogram** | Distribution of values | Request latency, response size |\n| **Summary** | Pre-calculated percentiles | Client-side latency (when you need exact percentiles) |\n\n**Rule:** Use histograms over summaries in most cases — they're aggregatable across instances.\n\n### Naming Conventions\n\n```\n# Pattern: <namespace>_<subsystem>_<name>_<unit>\nhttp_server_request_duration_seconds\nhttp_server_requests_total\ndb_pool_connections_active\nqueue_messages_pending\ncache_hit_ratio\n\n# Rules:\n# 1. Use snake_case\n# 2. Include unit suffix (_seconds, _bytes, _total)\n# 3. _total suffix for counters\n# 4. Don't include label names in metric name\n# 5. Use base units (seconds not milliseconds, bytes not kilobytes)\n```\n\n### Label Design Rules\n\n| Rule | Why | Example |\n|------|-----|---------|\n| Keep cardinality <100 per label | High cardinality kills performance | `status=\"200\"` not `status=\"200 OK\"` |\n| No user IDs as labels | Unbounded cardinality | Use log correlation instead |\n| No request paths with IDs | `/api/users/123` creates millions of series | Normalize: `/api/users/:id` |\n| Max 5-7 labels per metric | Each combo = a time series | `{method, path, status, service}` |\n\n### Instrumentation Checklist\n\n```yaml\napplication_metrics:\n  # HTTP layer\n  - http_request_duration_seconds: histogram {method, path, status}\n  - http_request_size_bytes: histogram {method, path}\n  - http_response_size_bytes: histogram {method, path}\n  - http_requests_in_flight: gauge\n  \n  # Business logic\n  - orders_processed_total: counter {status, payment_method}\n  - order_value_dollars: histogram {payment_method}\n  - user_signups_total: counter {source}\n  \n  # Dependencies\n  - db_query_duration_seconds: histogram {query_type, table}\n  - db_connections_active: gauge {pool}\n  - db_connections_idle: gauge {pool}\n  - cache_requests_total: counter {result: hit|miss}\n  - external_api_duration_seconds: histogram {service, endpoint}\n  - external_api_errors_total: counter {service, error_type}\n  \n  # Queue / async\n  - queue_messages_published_total: counter {queue}\n  - queue_messages_consumed_total: counter {queue, status}\n  - queue_processing_duration_seconds: histogram {queue}\n  - queue_depth: gauge {queue}\n  - queue_consumer_lag: gauge {queue, consumer_group}\n\ninfrastructure_metrics:\n  # Node exporter / cAdvisor provides these automatically\n  - cpu_usage_percent: gauge {instance}\n  - memory_usage_bytes: gauge {instance}\n  - disk_usage_bytes: gauge {instance, mount}\n  - disk_io_seconds: counter {instance, device}\n  - network_bytes: counter {instance, direction}\n  - container_cpu_usage: gauge {pod, container}\n  - container_memory_usage: gauge {pod, container}\n```\n\n### Stack Recommendations\n\n| Component | Options | Recommendation |\n|-----------|---------|----------------|\n| Collection | Prometheus, OTEL Collector, Datadog Agent | Prometheus (free) or OTEL Collector (vendor-neutral) |\n| Storage | Prometheus, Thanos, Mimir, VictoriaMetrics | VictoriaMetrics (best cost/perf) or Mimir (Grafana ecosystem) |\n| Visualization | Grafana, Datadog, New Relic | Grafana (free, extensible) |\n| Alerting | Alertmanager, Grafana Alerting, PagerDuty | Alertmanager + PagerDuty routing |\n\n---\n\n## Phase 3: Distributed Tracing\n\n### Trace Architecture\n\n```\nClient Request\n  → API Gateway (root span)\n    → Auth Service (child span)\n    → Order Service (child span)\n      → Database Query (child span)\n      → Payment Service (child span)\n        → Stripe API (child span)\n    → Notification Service (child span)\n      → Email Provider (child span)\n```\n\n### OpenTelemetry Setup\n\n**Auto-instrumentation (Node.js):**\n```typescript\n// tracing.ts — import BEFORE anything else\nimport { NodeSDK } from '@opentelemetry/sdk-node';\nimport { getNodeAutoInstrumentations } from '@opentelemetry/auto-instrumentations-node';\nimport { OTLPTraceExporter } from '@opentelemetry/exporter-trace-otlp-http';\n\nconst sdk = new NodeSDK({\n  traceExporter: new OTLPTraceExporter({\n    url: process.env.OTEL_EXPORTER_OTLP_ENDPOINT || 'http://localhost:4318/v1/traces',\n  }),\n  instrumentations: [getNodeAutoInstrumentations({\n    '@opentelemetry/instrumentation-http': { ignoreIncomingPaths: ['/health', '/ready'] },\n    '@opentelemetry/instrumentation-express': { enabled: true },\n  })],\n  serviceName: process.env.OTEL_SERVICE_NAME || 'payment-api',\n});\nsdk.start();\n```\n\n**Custom spans for business logic:**\n```typescript\nimport { trace, SpanStatusCode } from '@opentelemetry/api';\n\nconst tracer = trace.getTracer('payment-service');\n\nasync function processPayment(order: Order) {\n  return tracer.startActiveSpan('process-payment', async (span) => {\n    span.setAttributes({\n      'order.id': order.id,\n      'order.amount_cents': order.amountCents,\n      'payment.method': order.paymentMethod,\n    });\n    try {\n      const result = await chargeCard(order);\n      span.setAttributes({ 'payment.status': result.status });\n      return result;\n    } catch (err) {\n      span.setStatus({ code: SpanStatusCode.ERROR, message: err.message });\n      span.recordException(err);\n      throw err;\n    } finally {\n      span.end();\n    }\n  });\n}\n```\n\n### Sampling Strategies\n\n| Strategy | When | Config |\n|----------|------|--------|\n| **Always On** | Dev/staging, low traffic (<100 rps) | `ratio: 1.0` |\n| **Probabilistic** | Moderate traffic (100-1000 rps) | `ratio: 0.1` (10%) |\n| **Rate-limited** | High traffic (>1000 rps) | `max_traces_per_second: 100` |\n| **Tail-based** | Want all errors + slow requests | Collector-side: keep if error OR duration > p99 |\n| **Parent-based** | Respect upstream decisions | If parent sampled, child sampled |\n\n**Recommendation:** Start with parent-based + probabilistic (10%). Add tail-based at the collector to capture all errors.\n\n### Context Propagation\n\n| Header | Standard | Format |\n|--------|----------|--------|\n| `traceparent` | W3C Trace Context | `00-{trace_id}-{span_id}-{flags}` |\n| `tracestate` | W3C Trace Context | Vendor-specific key-value pairs |\n| `b3` | Zipkin B3 | `{trace_id}-{span_id}-{sampled}` |\n\n**Rule:** Use W3C Trace Context (`traceparent`) as primary. Support B3 for legacy Zipkin systems.\n\n### Trace Storage\n\n| Volume | Solution | Retention |\n|--------|----------|-----------|\n| <50 GB/day | Jaeger + Elasticsearch | 7 days |\n| 50-500 GB/day | Tempo + S3 | 14 days |\n| 500+ GB/day | Tempo + S3 with aggressive sampling | 7 days |\n| Budget-constrained | Jaeger + Badger (local disk) | 3 days |\n\n---\n\n## Phase 4: SLOs, SLIs & Error Budgets\n\n### SLI Selection by Service Type\n\n| Service Type | Primary SLI | Secondary SLI | Measurement |\n|--------------|-------------|---------------|-------------|\n| API / Web | Availability + Latency | Error rate | Server-side + synthetic |\n| Data pipeline | Freshness + Correctness | Throughput | Pipeline timestamps + checksums |\n| Storage | Durability + Availability | Latency | Checksums + uptime monitoring |\n| Streaming | Throughput + Latency | Message loss rate | Consumer lag + e2e latency |\n| Batch jobs | Success rate + Freshness | Duration | Job scheduler metrics |\n\n### SLO Definition Template\n\n```yaml\nslo:\n  name: \"Payment API Availability\"\n  service: payment-api\n  owner: payments-team\n  \n  sli:\n    type: availability\n    definition: \"Proportion of non-5xx responses\"\n    measurement: |\n      sum(rate(http_requests_total{service=\"payment-api\",status!~\"5..\"}[5m]))\n      /\n      sum(rate(http_requests_total{service=\"payment-api\"}[5m]))\n    \n  target: 99.95%  # 21.9 min downtime/month\n  window: rolling_30d\n  \n  error_budget:\n    total_minutes: 21.9  # per 30 days\n    burn_rate_alerts:\n      - severity: critical\n        burn_rate: 14.4x  # Budget consumed in 2 hours\n        short_window: 5m\n        long_window: 1h\n      - severity: warning\n        burn_rate: 6x    # Budget consumed in 5 days\n        short_window: 30m\n        long_window: 6h\n      - severity: ticket\n        burn_rate: 1x    # Budget consumed in 30 days\n        short_window: 6h\n        long_window: 3d\n  \n  consequences:\n    budget_remaining_above_50pct: \"Normal development velocity\"\n    budget_remaining_20_to_50pct: \"Prioritize reliability work\"\n    budget_remaining_below_20pct: \"Feature freeze; reliability only\"\n    budget_exhausted: \"All hands on reliability until budget recovers\"\n```\n\n### Common SLO Targets\n\n| Service Tier | Availability | p50 Latency | p99 Latency | Monthly Downtime |\n|--------------|-------------|-------------|-------------|------------------|\n| Tier 0 (payments, auth) | 99.99% | <100ms | <500ms | 4.3 min |\n| Tier 1 (core API) | 99.95% | <200ms | <1s | 21.9 min |\n| Tier 2 (non-critical) | 99.9% | <500ms | <2s | 43.8 min |\n| Tier 3 (internal tools) | 99.5% | <1s | <5s | 3.6 hours |\n| Batch / pipeline | 99% (success rate) | N/A | N/A | N/A |\n\n### Error Budget Tracking\n\n```yaml\n# Weekly error budget review template\nerror_budget_review:\n  week: \"2026-W08\"\n  service: payment-api\n  slo_target: 99.95%\n  \n  budget:\n    total_minutes_this_period: 21.9\n    consumed_minutes: 8.2\n    remaining_minutes: 13.7\n    remaining_percent: 62.6%\n    \n  incidents_consuming_budget:\n    - date: \"2026-02-18\"\n      duration_minutes: 5.1\n      cause: \"Database connection pool exhaustion\"\n      preventable: true\n      action: \"Increase pool size + add saturation alert\"\n    - date: \"2026-02-20\"\n      duration_minutes: 3.1\n      cause: \"Upstream payment provider timeout\"\n      preventable: false\n      action: \"Add circuit breaker with fallback\"\n  \n  velocity_decision: \"Normal — 62.6% budget remaining\"\n  reliability_work_this_week:\n    - \"Add connection pool saturation alert\"\n    - \"Implement circuit breaker for payment provider\"\n```\n\n---\n\n## Phase 5: Alert Design\n\n### Alert Quality Principles\n\n1. **Every alert must be actionable** — if no one needs to act, it's not an alert\n2. **Every alert needs a runbook** — linked directly in the alert annotation\n3. **Symptom-based over cause-based** — alert on \"users can't checkout\" not \"CPU high\"\n4. **Multi-window burn rate** — not static thresholds (see SLO alerts above)\n5. **Alert on absence, not just presence** — \"no orders in 15 min\" catches silent failures\n\n### Alert Severity Levels\n\n| Severity | Response Time | Channel | Who | Example |\n|----------|--------------|---------|-----|---------|\n| **P0 — Critical** | <5 min | Page (PagerDuty/Opsgenie) | On-call engineer | Payment system down |\n| **P1 — High** | <30 min | Page during business hours, Slack 24/7 | On-call | Error rate >5% for 10 min |\n| **P2 — Medium** | <4 hours | Slack channel | Team | p99 latency degraded 2x |\n| **P3 — Low** | Next business day | Ticket auto-created | Team backlog | Disk usage >80% |\n| **Info** | N/A | Dashboard only | No one | Deploy completed |\n\n### Alerting Anti-Patterns\n\n| Anti-Pattern | Problem | Fix |\n|-------------|---------|-----|\n| Static CPU/memory thresholds | Noisy, not user-impacting | Use SLO-based burn rate alerts |\n| Alert per instance | 50 instances = 50 alerts for same issue | Aggregate: alert on service-level error rate |\n| No deduplication | Same alert fires 100 times | Group by service + alert name; set repeat interval |\n| Missing runbook | Engineer gets paged, doesn't know what to do | Every alert links to a runbook |\n| Threshold too sensitive | Fires on brief spikes | Use `for: 5m` to require sustained condition |\n| Too many P0s | Alert fatigue → ignoring real incidents | Audit monthly; demote or remove noisy alerts |\n\n### Alert Template (Prometheus Alertmanager)\n\n```yaml\ngroups:\n  - name: payment-api-slo\n    rules:\n      - alert: PaymentAPIHighErrorRate\n        expr: |\n          (\n            sum(rate(http_requests_total{service=\"payment-api\",status=~\"5..\"}[5m]))\n            /\n            sum(rate(http_requests_total{service=\"payment-api\"}[5m]))\n          ) > 0.01\n        for: 5m\n        labels:\n          severity: critical\n          service: payment-api\n          team: payments\n        annotations:\n          summary: \"Payment API error rate {{ $value | humanizePercentage }} (>1%)\"\n          description: \"5xx error rate has exceeded 1% for 5 minutes\"\n          runbook: \"https://wiki.internal/runbooks/payment-api-errors\"\n          dashboard: \"https://grafana.internal/d/payment-api\"\n          \n      - alert: PaymentAPINoTraffic\n        expr: |\n          sum(rate(http_requests_total{service=\"payment-api\"}[15m])) == 0\n        for: 5m\n        labels:\n          severity: critical\n          service: payment-api\n        annotations:\n          summary: \"Payment API receiving zero traffic for 5 minutes\"\n          runbook: \"https://wiki.internal/runbooks/payment-api-no-traffic\"\n\n      - alert: PaymentAPILatencyHigh\n        expr: |\n          histogram_quantile(0.99, \n            sum(rate(http_request_duration_seconds_bucket{service=\"payment-api\"}[5m])) by (le)\n          ) > 2\n        for: 10m\n        labels:\n          severity: warning\n        annotations:\n          summary: \"Payment API p99 latency {{ $value }}s (>2s for 10min)\"\n          runbook: \"https://wiki.internal/runbooks/payment-api-latency\"\n```\n\n### Runbook Template\n\n```markdown\n# Runbook: PaymentAPIHighErrorRate\n\n## What This Alert Means\nThe payment API is returning >1% 5xx errors over a 5-minute window.\nUsers are likely failing to complete checkouts.\n\n## Impact\n- Users cannot process payments\n- Revenue loss: ~$X per minute (based on average traffic)\n- SLO: Payment API availability (target: 99.95%)\n\n## Immediate Actions\n1. Check the error dashboard: [link]\n2. Check recent deploys: `kubectl rollout history deployment/payment-api`\n3. Check upstream dependencies:\n   - Database: [dashboard link]\n   - Stripe API: [status page]\n   - Redis cache: [dashboard link]\n4. Check application logs:\n   ```\n   kubectl logs -l app=payment-api --since=10m | jq 'select(.level==\"error\")'\n   ```\n\n## Common Causes & Fixes\n| Cause | Diagnosis | Fix |\n|-------|-----------|-----|\n| Bad deploy | Errors started at deploy time | `kubectl rollout undo deployment/payment-api` |\n| DB connection exhaustion | `db_connections_active` at max | Restart pods (rolling) + increase pool size |\n| Stripe outage | Stripe status page red | Enable fallback payment processor |\n| Memory leak | Memory climbing, OOMKilled events | Rolling restart + investigate |\n\n## Escalation\n- If unresolved after 15 min: page payment team lead\n- If revenue impact >$10K: page VP Engineering\n- If Stripe outage: communicate to support team for customer messaging\n\n## Resolution\n- Confirm error rate <0.1% for 10 min\n- Post in #incidents: root cause + duration + impact\n- Schedule post-mortem if downtime >5 min\n```\n\n---\n\n## Phase 6: Dashboard Architecture\n\n### Dashboard Hierarchy\n\n```\nL1: Executive / Business Dashboard (non-technical stakeholders)\n  ↓\nL2: Service Overview Dashboard (on-call, quick triage)\n  ↓\nL3: Service Deep-Dive Dashboard (debugging specific service)\n  ↓\nL4: Infrastructure Dashboard (resource-level details)\n```\n\n### L1: Business Dashboard\n\n```yaml\npanels:\n  - title: \"Revenue per Minute\"\n    type: stat\n    query: \"sum(rate(orders_total{status='completed'}[5m])) * avg(order_value_dollars)\"\n  - title: \"Active Users (5min)\"\n    type: stat\n    query: \"count(count by (user_id) (http_requests_total{...}[5m]))\"\n  - title: \"Checkout Success Rate\"\n    type: gauge\n    query: \"sum(rate(checkout_total{status='success'}[1h])) / sum(rate(checkout_total[1h]))\"\n    thresholds: [95, 98, 99.5]\n  - title: \"Error Budget Remaining\"\n    type: gauge\n    query: \"1 - (error_budget_consumed / error_budget_total)\"\n```\n\n### L2: Service Overview Dashboard\n\nEvery service gets one of these with identical layout:\n\n```yaml\nrow_1_traffic:\n  - \"Request Rate (rps)\" — timeseries, by status code\n  - \"Error Rate (%)\" — timeseries, threshold line at SLO\n  - \"Active Requests\" — gauge\n\nrow_2_latency:\n  - \"Latency Distribution\" — heatmap\n  - \"p50 / p95 / p99\" — timeseries, threshold lines\n  - \"Latency by Endpoint\" — table, sorted by p99\n\nrow_3_dependencies:\n  - \"Downstream Latency\" — timeseries per dependency\n  - \"Downstream Error Rate\" — timeseries per dependency\n  - \"Database Query Duration\" — timeseries by query type\n\nrow_4_resources:\n  - \"CPU Usage\" — timeseries per pod\n  - \"Memory Usage\" — timeseries per pod\n  - \"Pod Restarts\" — stat\n\nrow_5_business:\n  - \"Business Metric 1\" — service-specific\n  - \"Business Metric 2\" — service-specific\n```\n\n### Dashboard Rules\n\n1. **Time range default: last 1 hour** — most debugging happens in recent time\n2. **Variable selectors at top**: environment, service, instance\n3. **Consistent color coding**: green=good, yellow=degraded, red=bad across all dashboards\n4. **Link alerts to dashboards** — every alert annotation includes dashboard URL\n5. **No more than 15 panels per dashboard** — split into L3 if needed\n6. **Include \"as of\" timestamp** — so screenshots in incidents are unambiguous\n7. **Dashboard as code** — store Grafana JSON in git, provision via API\n\n---\n\n## Phase 7: Incident Response\n\n### Incident Severity Classification\n\n| Severity | Criteria | Response | Communication |\n|----------|----------|----------|---------------|\n| **SEV-1** | Service down, data loss risk, security breach | All hands, war room | Status page update every 15 min |\n| **SEV-2** | Degraded service, SLO at risk, partial outage | On-call + backup | Status page update every 30 min |\n| **SEV-3** | Minor degradation, workaround exists | On-call during hours | Internal Slack update |\n| **SEV-4** | Cosmetic, low impact | Next sprint | None |\n\n### Incident Roles\n\n| Role | Responsibility | Who |\n|------|---------------|-----|\n| **Incident Commander (IC)** | Owns the incident. Coordinates. Makes decisions. | On-call lead |\n| **Technical Lead** | Diagnoses and fixes. Communicates technical status to IC. | Senior engineer |\n| **Communications Lead** | Updates status page, Slack, stakeholders. | Product/support |\n| **Scribe** | Documents timeline, actions, decisions in real-time. | Anyone available |\n\n### Incident Response Workflow\n\n```\n1. DETECT\n   - Alert fires → on-call paged\n   - Customer report → support escalates\n   - Internal discovery → engineer reports\n   \n2. TRIAGE (first 5 minutes)\n   - Confirm the issue is real (not false alert)\n   - Classify severity (SEV-1 through SEV-4)\n   - Open incident channel: #inc-YYYY-MM-DD-short-description\n   - Assign roles (IC, Tech Lead, Comms)\n   \n3. MITIGATE (next 5-30 minutes)\n   - Goal: STOP THE BLEEDING, not find root cause\n   - Options (try in order):\n     a. Rollback last deploy\n     b. Scale up / restart pods\n     c. Toggle feature flag off\n     d. Redirect traffic / enable fallback\n     e. Manual data fix\n   - Document every action with timestamp\n   \n4. STABILIZE\n   - Confirm mitigation is working (metrics back to normal)\n   - Monitor for 15-30 min for recurrence\n   - Update status page: \"Monitoring fix\"\n   \n5. RESOLVE\n   - Confirm all metrics healthy for 30+ min\n   - Update status page: \"Resolved\"\n   - Schedule post-mortem (within 48 hours for SEV-1/2)\n   - Send internal summary to stakeholders\n```\n\n### Incident Channel Template\n\n```\n📋 Incident: Payment API 5xx Errors\n🔴 Severity: SEV-2\n🕐 Started: 2026-02-22 14:23 UTC\n👤 IC: @alice\n🔧 Tech Lead: @bob\n📢 Comms: @charlie\n\nStatus: MITIGATING\nImpact: ~5% of checkout requests failing\nCustomer-facing: Yes\n\nTimeline:\n14:23 — Alert fired: PaymentAPIHighErrorRate\n14:25 — IC assigned: @alice, confirmed real via dashboard\n14:28 — Tech Lead: error logs show connection pool exhaustion post-deploy\n14:31 — Rolled back deployment v2.3.1 → v2.3.0\n14:35 — Error rate dropping, monitoring\n14:50 — Error rate <0.1%, marking resolved\n```\n\n---\n\n## Phase 8: Post-Mortem Framework\n\n### Blameless Post-Mortem Template\n\n```yaml\npost_mortem:\n  title: \"Payment API Connection Pool Exhaustion\"\n  date: \"2026-02-22\"\n  severity: SEV-2\n  duration: 27 minutes (14:23 — 14:50 UTC)\n  authors: [\"@alice\", \"@bob\"]\n  reviewers: [\"@engineering-leads\"]\n  status: action_items_in_progress\n  \n  summary: |\n    A deployment at 14:15 introduced a connection leak in the payment API.\n    Connection pool was exhausted by 14:23, causing 5xx errors for ~5% of\n    checkout requests. Rolled back at 14:31; recovered by 14:50.\n  \n  impact:\n    user_impact: \"~340 users saw checkout failures over 27 minutes\"\n    revenue_impact: \"$2,100 estimated (based on average order value × failed checkouts)\"\n    slo_impact: \"Consumed 5.1 min of 21.9 min monthly error budget (23%)\"\n    data_impact: \"No data loss. 12 orders failed; users could retry successfully.\"\n  \n  timeline:\n    - time: \"14:15\"\n      event: \"Deploy v2.3.1 rolled out (3/3 pods updated)\"\n    - time: \"14:23\"\n      event: \"PaymentAPIHighErrorRate alert fired\"\n    - time: \"14:25\"\n      event: \"IC assigned, confirmed via dashboard\"\n    - time: \"14:28\"\n      event: \"Root cause identified: new ORM query not releasing connections\"\n    - time: \"14:31\"\n      event: \"Rollback initiated: v2.3.1 → v2.3.0\"\n    - time: \"14:35\"\n      event: \"Error rate declining\"\n    - time: \"14:50\"\n      event: \"Resolved: error rate <0.1% sustained\"\n  \n  root_cause: |\n    The v2.3.1 deploy introduced a new database query in the order validation\n    path. The query used a raw connection instead of the pool's managed client,\n    so connections were acquired but never released. Under load, the pool\n    exhausted within 8 minutes.\n  \n  contributing_factors:\n    - \"No integration test for connection pool behavior under load\"\n    - \"Connection pool saturation metric existed but had no alert\"\n    - \"Code review didn't catch raw connection usage\"\n  \n  what_went_well:\n    - \"Alert fired within 8 minutes of deploy\"\n    - \"IC assigned in 2 minutes\"\n    - \"Root cause identified in 3 minutes (clear in logs)\"\n    - \"Rollback executed cleanly\"\n  \n  what_went_wrong:\n    - \"8-minute detection gap after deploy\"\n    - \"No canary deployment to catch before full rollout\"\n    - \"Connection pool saturation had no alert\"\n  \n  action_items:\n    - action: \"Add connection pool saturation alert (>80% for 2 min)\"\n      owner: \"@bob\"\n      priority: P1\n      due: \"2026-02-25\"\n      status: in_progress\n      ticket: \"ENG-1234\"\n    - action: \"Enable canary deployments for payment-api\"\n      owner: \"@alice\"\n      priority: P1\n      due: \"2026-03-01\"\n      ticket: \"ENG-1235\"\n    - action: \"Add linting rule: no raw DB connections in application code\"\n      owner: \"@charlie\"\n      priority: P2\n      due: \"2026-03-07\"\n      ticket: \"ENG-1236\"\n    - action: \"Load test payment-api connection pool in staging\"\n      owner: \"@bob\"\n      priority: P2\n      due: \"2026-03-07\"\n      ticket: \"ENG-1237\"\n  \n  lessons_learned:\n    - \"Resource saturation metrics need alerts, not just dashboards\"\n    - \"Canary deployments are mandatory for Tier 0 services\"\n    - \"ORM abstractions don't guarantee connection safety — review raw queries\"\n```\n\n### Post-Mortem Meeting Agenda (60 minutes)\n\n```\n1. (5 min) Context setting — IC reads the summary\n2. (15 min) Timeline walkthrough — what happened, when, by whom\n3. (15 min) Root cause deep-dive — 5 Whys exercise\n4. (5 min) What went well — celebrate good response\n5. (15 min) Action items — assign owners, priorities, due dates\n6. (5 min) Wrap-up — review date for action item check-in\n```\n\n### 5 Whys Exercise\n\n```\nProblem: 5xx errors in payment API\n\nWhy 1: Database connections were exhausted\nWhy 2: A new query acquired connections without releasing them\nWhy 3: The query used a raw connection instead of the pool manager\nWhy 4: The ORM's raw query API doesn't auto-release (by design)\nWhy 5: We don't have a linting rule or code review checklist item for this\n\nRoot cause: Missing guard against raw connection usage in application code\nSystemic fix: Linting rule + connection pool saturation alerting\n```\n\n---\n\n## Phase 9: On-Call Operations\n\n### On-Call Structure\n\n```yaml\non_call:\n  rotation: weekly\n  handoff_day: Monday 10:00 UTC\n  \n  primary:\n    response_time: 5 minutes (SEV-1/2), 30 minutes (SEV-3)\n    escalation_after: 15 minutes no-ack\n    \n  secondary:\n    response_time: 15 minutes (SEV-1), 1 hour (SEV-2/3)\n    escalation_after: 30 minutes no-ack\n    \n  manager_escalation:\n    trigger: SEV-1 unresolved after 30 minutes\n    \n  handoff_checklist:\n    - Review open incidents and active alerts\n    - Check error budget status for all services\n    - Read post-mortems from previous week\n    - Verify PagerDuty schedule and contact info\n    - Test alert routing (send test page)\n```\n\n### On-Call Health Metrics\n\n| Metric | Healthy | Needs Attention | Unhealthy |\n|--------|---------|-----------------|-----------|\n| Pages per week | <5 | 5-15 | >15 |\n| After-hours pages per week | <2 | 2-5 | >5 |\n| False positive rate | <10% | 10-30% | >30% |\n| Mean time to acknowledge | <5 min | 5-15 min | >15 min |\n| Mean time to resolve | <30 min | 30-120 min | >120 min |\n| Toil ratio (manual vs automated) | <30% | 30-60% | >60% |\n\n### Weekly On-Call Review Template\n\n```yaml\non_call_review:\n  week: \"2026-W08\"\n  engineer: \"@bob\"\n  \n  incidents:\n    total: 7\n    sev_1: 0\n    sev_2: 1\n    sev_3: 4\n    false_positives: 2\n    after_hours: 3\n    \n  time_spent:\n    incident_response: \"4.5 hours\"\n    toil_automation: \"2 hours\"\n    runbook_updates: \"1 hour\"\n    \n  improvements_made:\n    - \"Silenced noisy disk alert on dev servers\"\n    - \"Added auto-remediation for pod restart threshold\"\n    \n  improvements_needed:\n    - \"Cache expiry alert fires every Tuesday at 03:00 — needs investigation\"\n    - \"Payment retry logic needs circuit breaker (caused 3 alerts)\"\n    \n  handoff_notes: |\n    Watch payment-api p99 latency — it's been creeping up since Wednesday.\n    Stripe changed their sandbox endpoints; staging may throw errors.\n```\n\n---\n\n## Phase 10: Chaos Engineering & Reliability Testing\n\n### Chaos Principles\n\n1. Start with a hypothesis: \"If X fails, the system should Y\"\n2. Run in production (start small — one instance, one AZ)\n3. Minimize blast radius with automatic rollback\n4. Build confidence incrementally: staging → canary → production\n\n### Chaos Experiment Template\n\n```yaml\nchaos_experiment:\n  name: \"Payment DB failover\"\n  hypothesis: \"If the primary database becomes unavailable, traffic should\n    failover to the replica within 30 seconds with <1% error rate spike\"\n  \n  steady_state:\n    - metric: \"checkout_success_rate\"\n      expected: \">99.5%\"\n    - metric: \"db_query_duration_p99\"\n      expected: \"<200ms\"\n  \n  injection:\n    type: \"network_partition\"\n    target: \"payment-db-primary\"\n    duration: \"5 minutes\"\n    blast_radius: \"single AZ\"\n  \n  abort_conditions:\n    - \"checkout_success_rate < 95% for > 60 seconds\"\n    - \"revenue_per_minute drops > 50%\"\n    - \"any SEV-1 incident declared\"\n  \n  results:\n    failover_time: \"22 seconds\"\n    error_spike: \"0.3% for 25 seconds\"\n    hypothesis_confirmed: true\n    \n  follow_up_actions:\n    - \"Document failover behavior in runbook\"\n    - \"Add failover time as SLI (target: <30s)\"\n```\n\n### Chaos Engineering Maturity Levels\n\n| Level | What You Test | Tools |\n|-------|--------------|-------|\n| 1: Manual | Kill a pod, see what happens | `kubectl delete pod` |\n| 2: Automated | Scheduled pod kills, network delays | Chaos Monkey, Litmus |\n| 3: Game Days | Multi-failure scenarios with team exercise | Custom scripts + coordination |\n| 4: Continuous | Automated chaos in production with auto-rollback | Gremlin, Chaos Mesh |\n\n---\n\n## Phase 11: Observability Cost Optimization\n\n### Cost Drivers (Ranked)\n\n| # | Driver | Typical % of Bill | Optimization |\n|---|--------|-------------------|-------------|\n| 1 | Log volume | 40-60% | Reduce verbosity, drop DEBUG, sample repetitive |\n| 2 | Metric cardinality | 15-25% | Drop unused metrics, limit labels |\n| 3 | Trace volume | 10-20% | Sampling, tail-based sampling |\n| 4 | Retention | 10-15% | Tiered storage (hot → warm → cold) |\n| 5 | Query cost | 5-10% | Optimize dashboard queries, set max scan limits |\n\n### Cost Reduction Checklist\n\n```yaml\ncost_optimization:\n  logs:\n    - action: \"Drop DEBUG/TRACE in production\"\n      savings: \"30-50% of log volume\"\n    - action: \"Sample health check logs (1:100)\"\n      savings: \"5-15% of log volume\"\n    - action: \"Deduplicate identical error bursts\"\n      savings: \"10-20% during incidents\"\n    - action: \"Move logs older than 7 days to S3/cold storage\"\n      savings: \"60-80% of storage cost\"\n    - action: \"Drop request/response body logging\"\n      savings: \"20-40% of log volume\"\n  \n  metrics:\n    - action: \"Audit unused metrics (no dashboard, no alert)\"\n      savings: \"10-30% of series\"\n    - action: \"Reduce histogram bucket count (default 11 → 8)\"\n      savings: \"~27% of histogram series\"\n    - action: \"Remove high-cardinality labels\"\n      savings: \"Variable — can be massive\"\n    - action: \"Increase scrape interval for non-critical metrics (15s → 60s)\"\n      savings: \"75% of data points for those metrics\"\n  \n  traces:\n    - action: \"Implement tail-based sampling\"\n      savings: \"80-95% of trace volume\"\n    - action: \"Drop internal health check traces\"\n      savings: \"5-20% of trace volume\"\n    - action: \"Reduce span attribute size (truncate long strings)\"\n      savings: \"10-30% of trace storage\"\n  \n  general:\n    - action: \"Review and right-size retention policies quarterly\"\n    - action: \"Set query timeouts and result limits on dashboards\"\n    - action: \"Use recording rules for expensive queries\"\n```\n\n### Monthly Cost Review Template\n\n```yaml\nobservability_cost_review:\n  month: \"February 2026\"\n  total_cost: \"$X,XXX\"\n  \n  breakdown:\n    logs: { volume: \"X TB\", cost: \"$X\", pct: \"X%\" }\n    metrics: { series: \"X million\", cost: \"$X\", pct: \"X%\" }\n    traces: { volume: \"X TB\", cost: \"$X\", pct: \"X%\" }\n    infrastructure: { instances: X, cost: \"$X\", pct: \"X%\" }\n  \n  cost_per:\n    request: \"$0.000X\"\n    service: \"$X average\"\n    engineer: \"$X per engineer\"\n  \n  optimizations_applied: []\n  optimizations_planned: []\n  budget_status: \"on_track | over_budget | under_budget\"\n```\n\n---\n\n## Phase 12: Advanced Patterns\n\n### Correlation: Connecting the Three Pillars\n\n```\nEvery log line includes: trace_id, span_id\nEvery trace span includes: service, operation\nEvery metric includes: service label\n\nCorrelation paths:\n  Alert fires (metric) → Click → Dashboard (metric) → Filter by time window\n    → Trace search (same service + time) → Find failing trace\n    → Logs (filter by trace_id) → See exact error\n    \n  Support ticket (user report) → Find request_id in logs\n    → Extract trace_id → View full trace → Identify slow span\n    → Check span's service metrics → Confirm pattern\n```\n\n### Synthetic Monitoring\n\n```yaml\nsynthetic_checks:\n  - name: \"Checkout flow\"\n    type: browser\n    frequency: 5m\n    locations: [us-east, eu-west, ap-southeast]\n    steps:\n      - navigate: \"https://app.example.com/products\"\n      - click: \"Add to Cart\"\n      - click: \"Checkout\"\n      - assert: \"Order confirmation page loads in <3s\"\n    alert_on: \"2 consecutive failures from same location\"\n    \n  - name: \"API health\"\n    type: api\n    frequency: 1m\n    endpoints:\n      - url: \"https://api.example.com/health\"\n        expected_status: 200\n        max_latency_ms: 500\n      - url: \"https://api.example.com/v1/products?limit=1\"\n        expected_status: 200\n        max_latency_ms: 1000\n```\n\n### Feature Flag Observability\n\n```yaml\n# Correlate feature flags with metrics\nfeature_flag_monitoring:\n  - flag: \"new_checkout_flow\"\n    metrics_to_compare:\n      - \"checkout_conversion_rate\" # by flag variant\n      - \"checkout_error_rate\"\n      - \"checkout_latency_p99\"\n    alerts:\n      - \"If error rate for new variant > 2x control, auto-disable flag\"\n```\n\n### Observability Maturity Model\n\n| Dimension | Level 1 | Level 2 | Level 3 | Level 4 |\n|-----------|---------|---------|---------|---------|\n| Logging | Unstructured logs | Structured JSON, centralized | Correlated with traces | Automated log analysis |\n| Metrics | Basic infra metrics | RED/USE for services | SLO-based with error budgets | Predictive (anomaly detection) |\n| Tracing | No tracing | Key services instrumented | Full distributed tracing | Trace-driven testing |\n| Alerting | Static thresholds | Multi-signal alerts | Burn-rate based on SLOs | Auto-remediation |\n| Incident Response | Ad hoc | Defined process + roles | Post-mortems with action tracking | Chaos engineering in prod |\n| Culture | \"Ops team handles it\" | Shared ownership (you build it, you run it) | SLO-driven development velocity | Reliability as a feature |\n\n---\n\n## Quality Scoring Rubric (0-100)\n\n| Dimension | Weight | 0 | 5 | 10 |\n|-----------|--------|---|---|-----|\n| Logging quality | 15% | Unstructured, no correlation | Structured JSON, missing fields | Full schema, trace correlation, PII scrubbing |\n| Metrics coverage | 15% | No metrics | RED or USE, not both | RED + USE + business metrics + custom |\n| Tracing completeness | 10% | No tracing | Key services | Full path, sampling strategy, tail-based |\n| SLO maturity | 15% | No reliability targets | Informal targets | SLOs with error budgets, burn-rate alerts, weekly review |\n| Alert quality | 15% | Noisy/missing | Actionable, some runbooks | SLO-based, full runbooks, low false positive |\n| Incident response | 10% | Ad hoc | Defined process | Full process, roles, post-mortems, chaos engineering |\n| Dashboard design | 10% | No dashboards | Basic panels | Hierarchical L1-L4, consistent, linked to alerts |\n| Cost efficiency | 10% | Unknown cost | Tracked | Optimized, reviewed monthly, within budget |\n\n**90-100:** World-class. Teach others. **70-89:** Production-ready. Fill specific gaps. **50-69:** Functional but fragile. **<50:** Significant reliability risk.\n\n---\n\n## 10 Observability Commandments\n\n1. **Structured or it didn't happen** — unstructured logs are technical debt\n2. **Correlate everything** — trace_id connects logs, traces, and metrics\n3. **Alert on symptoms, not causes** — users don't care about CPU, they care about latency\n4. **Every alert gets a runbook** — no runbook = no alert\n5. **SLOs drive velocity** — error budgets decide when to ship vs stabilize\n6. **Dashboards have hierarchy** — executives don't need pod CPU graphs\n7. **Blameless post-mortems always** — blame prevents learning\n8. **Cost is a feature** — observability that bankrupts you isn't observability\n9. **You build it, you run it** — the team that ships code owns its observability\n10. **Practice failure** — chaos engineering builds confidence\n\n---\n\n## 12 Natural Language Commands\n\n| Command | What It Does |\n|---------|-------------|\n| \"Audit our observability\" | Run the /16 health check, score each dimension, prioritize gaps |\n| \"Design logging for [service]\" | Generate structured log schema with context fields for the service |\n| \"Set up metrics for [service]\" | Create RED + USE + business metric instrumentation plan |\n| \"Create SLOs for [service]\" | Define SLIs, targets, error budgets, and burn-rate alert rules |\n| \"Design alerts for [service]\" | Create alert rules with severity, thresholds, and runbook templates |\n| \"Build dashboard for [service]\" | Design L2 service overview dashboard with panel specifications |\n| \"Write a runbook for [alert]\" | Generate structured runbook with diagnosis steps and fixes |\n| \"Run post-mortem for [incident]\" | Generate blameless post-mortem document with timeline and action items |\n| \"Set up on-call for [team]\" | Design rotation, escalation policy, handoff checklist |\n| \"Plan chaos experiment for [scenario]\" | Design experiment with hypothesis, injection, abort conditions |\n| \"Optimize observability costs\" | Audit current spend, identify top savings, create reduction plan |\n| \"Design tracing for [system]\" | Create OpenTelemetry instrumentation plan with sampling strategy |\n\n---\n\n## ⚡ Level Up Your Observability\n\nThis skill gives you the methodology. For industry-specific implementation patterns:\n\n- **SaaS companies:** [AfrexAI SaaS Context Pack ($47)](https://afrexai-cto.github.io/context-packs/) — includes SaaS-specific SLOs, multi-tenant monitoring, and usage-based billing observability\n- **Fintech:** [AfrexAI Fintech Context Pack ($47)](https://afrexai-cto.github.io/context-packs/) — compliance audit logging, transaction monitoring, fraud detection signals\n- **Healthcare:** [AfrexAI Healthcare Context Pack ($47)](https://afrexai-cto.github.io/context-packs/) — HIPAA audit trails, PHI access logging, uptime requirements\n\n### 🔗 More Free Skills by AfrexAI\n\n- `afrexai-devops-engine` — CI/CD, infrastructure, deployment strategies\n- `afrexai-api-architect` — API design, security, versioning\n- `afrexai-database-engineering` — Schema design, query optimization, migrations\n- `afrexai-code-reviewer` — Code review methodology with SPEAR framework\n- `afrexai-prompt-engineering` — System prompt design, testing, optimization\n\n**Browse all AfrexAI skills:** [clawhub.com](https://clawhub.com) | [Full storefront](https://afrexai-cto.github.io/context-packs/)\n","readmeExcerpt":"--- name: afrexai-observability-engine model: standard description: Complete observability & reliability engineering system. Use when designing monitoring, implementing structured logging, setting up distributed tracing, building alerting systems, creating SLO/SLI frameworks, running incident response, conducting post-mortems, or auditing system reliability. Covers all three pillars (logs/metrics/traces), alert desig","codeSnippets":[],"executableExamples":[{"language":"text","snippet":"Application → Structured JSON → Log Router → Storage → Query Engine\n                                    ↓\n                              Alert Pipeline"},{"language":"yaml","snippet":"# HTTP request context\nhttp:\n  method: POST\n  path: /api/v1/orders\n  status: 201\n  client_ip: 203.0.113.42  # Anonymize in logs if needed\n  user_agent: \"Mozilla/5.0...\"\n  request_id: \"req_abc123\"\n\n# Business context\nbusiness:\n  user_id: \"usr_456\"\n  tenant_id: \"tenant_789\"\n  order_id: \"ord_012\"\n  action: \"checkout\"\n  amount_cents: 4999\n  currency: \"USD\"\n\n# Error context\nerror:\n  type: \"PaymentDeclinedError\"\n  message: \"Card declined: insufficient funds\"\n  code: \"CARD_DECLINED\"\n  stack: \"...\" # Only in non-production or DEBUG level\n  retry_count: 2\n  retryable: true"},{"language":"text","snippet":"Is the process about to crash?\n  → FATAL (exit after logging)\n\nDid an operation fail that needs human attention?\n  → ERROR (page someone or create ticket)\n\nDid something unexpected happen but we recovered?\n  → WARN (review in daily triage)\n\nIs this a normal business event worth recording?\n  → INFO (audit trail, business metrics)\n\nIs this useful for debugging but noisy in production?\n  → DEBUG (off in prod, on in staging)\n\nIs this only useful when stepping through code?\n  → TRACE (never in production)"},{"language":"yaml","snippet":"scrub_patterns:\n  # Always redact\n  - field_patterns: [\"password\", \"secret\", \"token\", \"api_key\", \"authorization\"]\n    action: replace_with_redacted\n  \n  # Hash for correlation without exposure\n  - field_patterns: [\"email\", \"phone\", \"ssn\", \"national_id\"]\n    action: sha256_hash\n  \n  # Mask partially\n  - field_patterns: [\"credit_card\", \"card_number\"]\n    action: mask_last_4  # \"****-****-****-1234\"\n  \n  # IP anonymization\n  - field_patterns: [\"client_ip\", \"ip_address\"]\n    action: zero_last_octet  # 203.0.113.0"},{"language":"typescript","snippet":"import pino from 'pino';\nimport { AsyncLocalStorage } from 'node:async_hooks';\n\nconst als = new AsyncLocalStorage<Record<string, string>>();\n\nconst logger = pino({\n  level: process.env.LOG_LEVEL || 'info',\n  formatters: {\n    level: (label) => ({ level: label }),\n  },\n  mixin: () => als.getStore() ?? {},\n  redact: ['req.headers.authorization', '*.password', '*.token'],\n  timestamp: pino.stdTimeFunctions.isoTime,\n});\n\n// Middleware: inject context\napp.use((req, res, next) => {\n  const ctx = {\n    trace_id: req.headers['x-trace-id'] || crypto.randomUUID(),\n    request_id: crypto.randomUUID(),\n    service: 'payment-api',\n    version: process.env.APP_VERSION,\n  };\n  als.run(ctx, () => next());\n});"},{"language":"python","snippet":"import structlog\nstructlog.configure(\n    processors=[\n        structlog.contextvars.merge_contextvars,\n        structlog.processors.add_log_level,\n        structlog.processors.TimeStamper(fmt=\"iso\", utc=True),\n        structlog.processors.JSONRenderer(),\n    ],\n)\nlog = structlog.get_logger()\n# Bind context per-request:\nstructlog.contextvars.bind_contextvars(trace_id=trace_id, user_id=user_id)"}],"parameters":{},"dependencies":[],"permissions":[],"extractedFiles":[],"languages":["typescript"],"docsSourceLabel":"CLAWHUB","editorialOverview":"Complete observability & reliability engineering system. Use when designing monitoring, implementing structured logging, setting up distributed tracing, building alerting systems, creating SLO/SLI frameworks, running incident response, conducting post-mortems, or auditing system reliability. Covers all three pillars (logs/metrics/traces), alert design, dashboard architecture, on-call operations, chaos engineering, and cost optimization. --- name: afrexai-observability-engine model: standard description: Complete observability & reliability engineering system. Use when designing monitoring, implementing structured logging, setting up distributed tracing, building alerting systems, creating SLO/SLI frameworks, running incident response, conducting post-mortems, or auditing system reliability. Covers all three pillars (logs/metrics/traces), alert desig","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":415,"uniquenessScore":63,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T06:20:52.265Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}