{"id":"befb55b3-6857-4cad-b638-4b4da31d00f6","entityType":"agent","slug":"crewai-shivakrishna44-devops-microservices-crewai","name":"devops-microservices-crewAi","canonicalUrl":"https://www.xpersona.co/agent/crewai-shivakrishna44-devops-microservices-crewai","canonicalPath":"/agent/crewai-shivakrishna44-devops-microservices-crewai","generatedAt":"2026-10-10T04:07:48.499Z","source":"GITHUB_REPOS","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-09T11:50:37.719Z","emptyReason":null},"description":"Production-ready microservices platform on AWS EKS using GitHub Actions for CI and ArgoCD for CD (GitOps). This project uses CrewAI with a hierarchical process (manager agent auto-delegates) to run a team of specialized DevOps agents that monitor infrastructure, detect issues, and optimize costs. devops-microservices-crewAi This repo contains **two related things** that share one EKS cluster and Terraform/AWS foundation: 1. **A microservices platform** — three Flask services (order, payment, user) built and deployed to EKS via GitHub Actions → ECR → Helm (GitOps). This is the day-to-day app CI/CD. See **$1** below. 2. **A human-approved EKS-upgrade agent** (CrewAI) — an AI multi-agent workflow that safely upg","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. Last updated 10/9/2026.","installCommand":null,"sourceUrl":"https://github.com/ShivaKrishna44/devops-microservices-crewAi","homepage":null,"primaryLinks":[{"label":"View Source","url":"https://github.com/ShivaKrishna44/devops-microservices-crewAi","kind":"source"}],"safetyScore":66,"overallRank":34.4,"popularityScore":0,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Production-ready microservices platform on AWS EKS using GitHub Actions for CI and ArgoCD for CD (GitOps). This project uses CrewAI with a hierarchical process "},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T11:50:37.719Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[{"label":"crewai","status":"self-declared"},{"label":"multi-agent","status":"self-declared"}],"verifiedCount":0,"selfDeclaredCount":3,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"},{"key":"crewai","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"},{"key":"multi-agent","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile capability:crewai|supported|profile capability:multi-agent|supported|profile"}},"adoption":{"evidence":{"source":"no-adoption-signals","verified":false,"confidence":"low","updatedAt":"2026-10-09T11:50:37.719Z","emptyReason":"No source adoption metrics were available."},"stars":0,"forks":0,"downloads":null,"packageName":null,"latestVersion":null,"tractionLabel":null},"release":{"evidence":{"source":"agent-index","verified":false,"confidence":"medium","updatedAt":"2026-10-09T11:50:37.713Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T11:50:37.719Z","lastCrawledAt":"2026-10-09T11:50:37.713Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-16T11:50:37.713Z","lastVerifiedAt":null,"highlights":[]},"execution":{"evidence":{"source":"GITHUB REPOS","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":null,"setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/crewai-shivakrishna44-devops-microservices-crewai/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/crewai-shivakrishna44-devops-microservices-crewai/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/crewai-shivakrishna44-devops-microservices-crewai/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/crewai-shivakrishna44-devops-microservices-crewai/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/crewai-shivakrishna44-devops-microservices-crewai/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/crewai-shivakrishna44-devops-microservices-crewai/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"GITHUB_REPOS","generatedAt":"2026-10-10T04:07:48.498Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/crewai-shivakrishna44-devops-microservices-crewai/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/crewai-shivakrishna44-devops-microservices-crewai/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/crewai-shivakrishna44-devops-microservices-crewai/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/crewai-shivakrishna44-devops-microservices-crewai/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"GITHUB REPOS","verified":false,"confidence":"high","updatedAt":"2026-10-09T11:50:37.719Z","emptyReason":null},"readme":"# devops-microservices-crewAi\n\nThis repo contains **two related things** that share one EKS cluster and\nTerraform/AWS foundation:\n\n1. **A microservices platform** — three Flask services (order, payment, user)\n   built and deployed to EKS via GitHub Actions → ECR → Helm (GitOps). This is\n   the day-to-day app CI/CD. See **[§ Microservices Platform](#microservices-platform)** below.\n2. **A human-approved EKS-upgrade agent** (CrewAI) — an AI multi-agent workflow\n   that safely upgrades the cluster's Kubernetes version behind heavy\n   guardrails and a human approval gate. See **[§ EKS Upgrade Agent](#eks-upgrade-by-agent-human-approved)** below.\n\nThey're independent: you can run the microservices CI/CD without ever touching\nthe agent, and vice versa. The agent is the more complex, safety-critical piece\nand is documented in full further down.\n\n---\n\n# Microservices Platform\n\nThree Flask microservices (`order-service`, `payment-service`, `user-service`)\nbuilt, containerized, and deployed to the EKS cluster via GitHub Actions + Helm,\nwith infrastructure managed by Terraform.\n\n## Layout (microservices side)\n\n```\napp/{order,payment,user}-service/   # Flask app + Dockerfile + requirements.txt\ncharts/microservice/                # shared Helm chart + per-service values\nTerraform/                          # VPC, EKS, IAM/IRSA, ECR, backend\n.github/workflows/ci-cd.yml         # build -> push to ECR -> GitOps commit\n.github/workflows/codeql.yml        # static analysis\n```\n\n## End-to-end: zero to live\n\n**1. Provision infra (Terraform)** — backend bucket/lock table must already exist:\n```bash\ncd Terraform\nterraform init -backend-config=tfvars/dev/backend.tfvars\nterraform plan  -var-file=tfvars/dev/dev.tfvars\nterraform apply -var-file=tfvars/dev/dev.tfvars\n```\n\n**2. Point kubectl at the cluster:**\n```bash\naws eks update-kubeconfig --name expense-dev --region us-east-1\nkubectl get nodes\n```\n\n**3. Install the AWS Load Balancer Controller** (Terraform creates the IRSA role;\nthe controller is installed via Helm). Fill the role ARN into\n`alb-controller-values.yaml` first (don't commit a real ARN):\n```bash\nhelm repo add eks https://aws.github.io/eks-charts && helm repo update\nhelm install aws-load-balancer-controller eks/aws-load-balancer-controller \\\n  -n kube-system -f alb-controller-values.yaml\n```\n\n**4. CI/CD** — push to `main` (or run the workflow manually). `ci-cd.yml` builds\neach service's image, pushes to ECR (auth via GitHub **OIDC**, no static keys),\nand commits the new image tag into `charts/microservice/values-<svc>.yaml`.\n\n**5. Deploy via Helm** (first release; ArgoCD can automate this afterward):\n```bash\nhelm upgrade --install order-service charts/microservice \\\n  -f charts/microservice/values.yaml -f charts/microservice/values-order.yaml -n default\n# repeat for payment-service / user-service\n```\n\n**6. Confirm:**\n```bash\nkubectl get pods\nkubectl get ingress   # ALB DNS name\n```\n\n> **OIDC trust-policy note (learned the hard way):** this repo was created after\n> GitHub's immutable-subject-claim change, so the IAM role's trust policy must\n> match the real `sub` claim format\n> `repo:<owner>@<ownerId>/<repo>@<repoId>:ref:refs/heads/main`, not the old\n> plain `repo:<owner>/<repo>:...` format — otherwise `AssumeRoleWithWebIdentity`\n> fails with a generic `AccessDenied`. Pull the exact value from a CloudTrail\n> `AssumeRoleWithWebIdentity` event if in doubt.\n\n---\n\n# EKS Upgrade by Agent (Human-Approved)\n# guarded-eks-upgrade-agent\n\nAn AI multi-agent workflow that upgrades an Amazon EKS cluster **one minor version at a time**, driven by a target version a human provides, and executed **only after explicit human approval**.\n\nThe point of this project is not \"let AI upgrade the cluster.\" It's the opposite: EKS upgrades are high-risk and irreversible (you can't downgrade the control plane), so the agents do the *tedious, error-prone* work — checking compatibility, drafting the plan, validating afterward — while a human stays firmly in control of the one dangerous action: applying the upgrade.\n\n**Docs:** [Setup & Usage](#setup) · [DEMO-GUIDE.md](DEMO-GUIDE.md) (walkthrough) · [docs/TESTING-GUIDE.md](docs/TESTING-GUIDE.md) (how to test) · [docs/DEPLOYMENT-GUIDE.md](docs/DEPLOYMENT-GUIDE.md) (deploy & operate) · [bin/README.md](bin/README.md) (helper scripts)\n\n---\n\n## The workflow\n\n```\n   Human provides target version (e.g. 1.34)\n                  │\n                  ▼\n   ┌──────────────────────────────┐\n   │ 1. PRE-CHECK AGENT            │  Reads current version, validates the jump,\n   │                              │  scans for deprecated APIs, checks addon\n   │                              │  compatibility and node readiness.\n   └──────────────┬───────────────┘\n                  ▼\n   ┌──────────────────────────────┐\n   │ 2. UPGRADE PLANNER AGENT      │  Runs `terraform plan`, summarizes exactly\n   │                              │  what will change, and produces a\n   │                              │  GO / NO-GO recommendation with evidence.\n   └──────────────┬───────────────┘\n                  ▼\n        ╔══════════════════════════╗\n        ║  APPROVAL GATE (HUMAN)    ║  Nothing is applied until a human types\n        ║  APPROVE / REJECT         ║  APPROVE. Decision is logged with actor,\n        ╚══════════════┬═══════════╝  timestamp, reason, and the evidence shown.\n                  │ (only if APPROVED)\n                  ▼\n   ┌──────────────────────────────┐\n   │ 3. EXECUTOR AGENT             │  `terraform apply` — control plane first,\n   │                              │  then managed node groups. Gated: refuses\n   │                              │  to run without a valid approval token.\n   └──────────────┬───────────────┘\n                  ▼\n   ┌──────────────────────────────┐\n   │ 4. POST-UPGRADE VALIDATOR     │  Confirms new version, nodes Ready,\n   │                              │  system pods healthy. Reports PASS/FAIL.\n   └──────────────────────────────┘\n```\n\n## The agents\n\n| Agent | Responsibility | Can it change anything? |\n|-------|----------------|-------------------------|\n| **Pre-Check Agent** | Validate the target version, scan deprecated APIs, addon/node readiness | No — read-only |\n| **Upgrade Planner** | `terraform plan`, summarize the diff, GO/NO-GO recommendation | No — plan only |\n| **Executor Agent** | `terraform apply` (control plane → node groups) | **Yes — but only with an APPROVED token** |\n| **Post-Upgrade Validator** | Verify version, node readiness, pod health | No — read-only |\n\n## Guardrails & approval gates (defense in depth)\n\nSafety is layered — an upgrade must clear **every** layer. If any one blocks, nothing is applied.\n\n### Layer 1 — Read-only by default\nThree of the four agents (Pre-Check, Planner, Validator) have **no tools that can change anything**. Only the Executor can modify infrastructure, and only when unlocked.\n\n### Layer 1.5 — Deterministic pre-flight (before anything else)\nRun directly (not via the LLM) at the start of `request_and_check`:\n- **Cluster exists and is ACTIVE** — if the cluster is `UPDATING` (an upgrade/change already in flight), or the name/account/region is wrong, the run **aborts before any plan**. (Also enforced as `check_cluster_upgradeable` in the pre-check agent.)\n\n### Layer 2 — Deterministic guardrails (non-LLM) — `guardrails.py`\nPattern-based checks that run before any apply. They cannot be \"talked out of\" a no, because no LLM is in the path:\n\n| Guardrail | Blocks when… | Severity |\n|-----------|--------------|----------|\n| `version_format` | target isn't a valid version (e.g. `1.34`) | CRITICAL |\n| `single_minor_step` | not exactly current+1 minor (downgrade, skip, or major change) | CRITICAL |\n| `cluster_name_confirmation` | the human didn't type the exact target cluster name | CRITICAL |\n| `region_allowlist` | region isn't in `ALLOWED_REGIONS` | CRITICAL |\n| `prod_two_person` | prod cluster + two-person required (raises the approver count) | WARN |\n| `precheck_verdict` | the read-only pre-check evidence contained UNSAFE / NO-GO | CRITICAL |\n\n**Availability pre-checks (zero-downtime readiness)** — run in the pre-check phase:\n\n| Check | Flags when… |\n|-------|-------------|\n| `check_pdb_coverage` | multi-replica app workloads have **no PodDisruptionBudget** — a node drain could evict all replicas at once |\n| `check_pdb_strength` | a critical PDB is **outside the safe band** — either **too loose** (`disruptionsAllowed/currentHealthy` > threshold → over-eviction downtime) or **too strict** (`disruptionsAllowed == 0`, e.g. `maxUnavailable: 0` / `minAvailable == replicas` → the drain stalls or force-evicts → downtime) (preventive) |\n| `check_capacity_headroom` | fewer than **2 Ready nodes** — a rolling node replacement would drain the only node, causing downtime |\n| `check_ec2_surge_quota` | the account lacks **EC2 vCPU quota** to launch the surge nodes — the rollover would **freeze mid-upgrade** (see below) |\n\n#### Prevention + detection, together\n\nAvailability during the node rollover is protected two ways:\n- **Prevention (`check_pdb_strength`, pre-flight):** verifies critical deployments have a PDB strict enough that Kubernetes itself will **refuse** a drain that would drop them below the threshold. A PDB that merely exists isn't enough — a `maxUnavailable: 50%` PDB still permits a 50% drop. This check reads each PDB's live `disruptionsAllowed / currentHealthy` and blocks the upgrade if it exceeds the threshold.\n- **Detection (live monitor, during rollover):** the concurrent monitor sounds the alarm the moment healthy pods actually drop below the floor.\n\nThe preventive PDB check is the real control (it stops the disruption from happening); the live monitor is the safety net that catches anything the PDBs didn't.\n\n#### The surge-capacity freeze (why `check_ec2_surge_quota` matters)\n\nTrue zero-downtime node upgrades bring up **new (surge) nodes before draining old ones**. Those surge instances count against the EC2 **\"Running On-Demand Standard instances\" vCPU quota** (`L-1216C47A`). If the account is near that limit:\n\n1. The managed node group tries to launch the surge node.\n2. AWS refuses (quota exceeded).\n3. The rollout **freezes** — old nodes aren't drained, new nodes can't launch, the node group sits in `UPDATING` until it times out or you intervene.\n\n`check_ec2_surge_quota` estimates the surge vCPUs the node groups will request (instance type × surge nodes per group) and compares against remaining quota headroom (`quota − in-use`). A shortfall reports `ALERT` / `UNSAFE`, which the `precheck_verdict` guardrail treats as **blocking** — so you fix the quota *before* approving, not mid-rollout.\n\n**Recovery if it does freeze:** request an increase to the `L-1216C47A` quota (Service Quotas console), wait for it to apply, then re-run `terraform apply` — the node group resumes the rollout from where it stalled. The control plane (already upgraded in phase 1) is unaffected.\n\n### Layer 3 — Human approval gate — `approval_gate.py`\n- **Typed cluster-name confirmation** — the operator must type the exact cluster name, preventing \"right command, wrong cluster\" mistakes.\n- **Explicit APPROVE** — a human types `APPROVE`; agents cannot self-approve.\n- **Two-person approval for production** — clusters matching `PROD_CLUSTER_MARKERS` require **two distinct approvers** (configurable). One person cannot approve twice.\n- **Evidence binding (hash-only)** — the approval is tied to a SHA-256 hash of the pre-check + plan evidence. The raw plan is **not stored** — only its hash. A separate approver re-runs the read-only pre-check; the gate verifies the freshly-generated evidence hashes to the same value. If the cluster/plan drifted since the request, the hash won't match and the approval is refused.\n- **Expiry (TTL)** — approvals go stale after `APPROVAL_TTL_MINUTES`, so a forgotten approval can't be used later against a now-different cluster.\n- **Version binding** — approving `1.34` never approves `1.35`.\n- **Account + region binding** — the approval records the AWS account ID and region; a same-named cluster in a different account/region can't reuse the approval (re-verified at apply).\n- **Two-person enforced at apply** — for a production cluster, the apply gate independently requires ≥ 2 **distinct** approvers on record, even if the stored `required_approvers` was somehow lower (belt-and-suspenders, not just advisory).\n\n### Layer 4 — Gated executor + deterministic apply path\nThe apply is driven **deterministically**, not from an LLM summary:\n- `do_apply` first **re-verifies the approval against freshly-regenerated evidence** (`approval_check`) — so plan **drift between approval and apply is caught** (evidence-hash mismatch) and an **expired approval (TTL)** is rejected.\n- It then calls the executor and validation tools directly and decides pass/fail from their **own status strings** (`SUCCESS` / `COMPLETED WITH ALARM` / `BLOCKED` / `FAILED`). A degraded or failed upgrade **cannot be narrated into a success** by the model.\n- **`COMPLETED WITH ALARM` (an availability breach during rollover) is treated as FAILURE** (non-zero exit), not success.\n- The executor tool itself also re-verifies the stored approval record (version, `APPROVED`, enough distinct approvers, not expired) — it validates the **record the human created**, never agent-supplied text. Any failure returns `BLOCKED` and touches nothing.\n\n### Layer 5 — Zero-downtime execution (sequencing + validation loop)\nThe apply is **phased**, not a single blind `terraform apply`:\n\n1. **Baseline first** — before touching anything, a health snapshot is captured (which nodes/pods/deployments are healthy now) so regressions can be detected later.\n2. **Phase 1 — control plane only** (`-target` the cluster). Node groups are not touched yet.\n3. **Sequence gate (dual, before touching nodes)** — the run **blocks and loops** until BOTH are true on the same iteration, stable for two consecutive checks:\n   - **AWS API:** `describe-cluster` reports the control plane is `ACTIVE` on the target version, and\n   - **Kubernetes API:** `kubectl get nodes` actually responds within `APISERVER_LATENCY_THRESHOLD_S` (default 10s).\n   A control plane can report `ACTIVE` in the AWS API while the API server is still slow right after an upgrade — so we require the real `kubectl` round-trip to be fast before proceeding. If the gate isn't satisfied in time, the run **halts before node groups**.\n4. **Phase 2 — node groups, with a LIVE availability monitor** — the managed rolling replacement (surge node up, old node drained) runs in a background thread while a monitor polls **every 15s**. The moment a **critical deployment drops below its availability floor** — `ceil(baseline_healthy × (1 − AVAILABILITY_DROP_THRESHOLD))`, e.g. losing more than **20%** of its baseline healthy pods — it:\n   - **sounds the alarm immediately** (error log + optional webhook), and\n   - if `HALT_ON_AVAILABILITY_BREACH=true` (default), **halts further node draining** so a human can debug — see below.\n   The result is flagged `COMPLETED WITH ALARM` (and, if halted, includes the cordoned nodes + resume steps) so the breach is never silently swallowed.\n\n#### How \"halt further draining\" works (and its honest constraint)\n\nA managed-node-group `terraform apply` **cannot be safely killed mid-instance** — interrupting it can leave the node group in a worse, inconsistent state. So the halt does **not** kill terraform. Instead it stops the drain *wave* where Kubernetes controls it: it **cordons every old (pre-flight) node still present**, marking them unschedulable so no further pods move onto them, and — combined with the strict PDBs verified pre-flight — Kubernetes **refuses further evictions**. The in-flight instance finishes, then progress stops.\n\n**Resume after debugging:** fix the workload (scale up, fix the failing pod, loosen nothing you shouldn't), then `kubectl uncordon <node>` the cordoned nodes and re-run `terraform apply` to finish the rollover. Set `HALT_ON_AVAILABILITY_BREACH=false` for alarm-only behavior (no cordon).\n5. **Validation loop (after)** — `wait_for_healthy` polls cluster health **every 30s for up to 20 minutes**, and passes only when **all** acceptance criteria hold together:\n   - every node is **Ready**,\n   - every **old (pre-flight) node is fully terminated** — a stalled rollover that leaves old nodes lingering must not pass,\n   - no pods in a bad state, and\n   - **deployment replica counts match the pre-flight baseline** (no workload silently lost/gained replicas).\n   On timeout it reports exactly which criteria are still unmet.\n6. **Regression check** — `compare_to_baseline` flags anything that was healthy **before** the upgrade but is broken **now**. Pre-existing problems don't count; new breakage does. The validator only reports PASS if version is correct, the cluster is healthy, and there are no regressions.\n\n### Layer 6 — Correct, auditable record\n- **Full audit trail** — every event (request, each approval, rejection, reset) is persisted with actor, timestamp, reason, and evidence hash, so the whole decision chain is reconstructable.\n\n### Honest limitations\n- Zero-downtime depends on your **workloads** being HA (multiple replicas + PDBs + anti-affinity). The agent *checks* for PDBs, node capacity, and EC2 surge quota and *warns/blocks*, but it can't make a single-replica app highly available.\n- The surge-quota check estimates vCPUs from instance type × surge nodes and the `L-1216C47A` quota. It's an estimate (unknown instance types use a conservative fallback; it doesn't account for RIs/Savings Plans or non-Standard families) — treat an ALERT as \"investigate,\" and it can't see real-time AWS capacity, only your quota.\n- The phased `-target` apply assumes the standard `terraform-aws-modules/eks/aws` resource address (`module.eks.aws_eks_cluster.this`). If your module structure differs, adjust the target in `upgrade_tools.py`.\n- The regression check is a coarse health compare (pod status, deployment readiness), not deep app-level SLO monitoring. For production, pair it with real synthetic checks / Prometheus alerts.\n\n### Configurable policy (`.env`)\n| Setting | Default | Purpose |\n|---------|---------|---------|\n| `APPROVAL_TTL_MINUTES` | `60` | how long an approval stays valid (0 = never) |\n| `ALLOWED_REGIONS` | `us-east-1` | regions the agent may operate in |\n| `PROD_CLUSTER_MARKERS` | `prod,production,live` | substrings that mark a cluster as production |\n| `REQUIRE_TWO_PERSON_FOR_PROD` | `true` | require two distinct approvers for prod |\n\n## Beyond upgrade — other gated cluster operations\n\nThe same safety model (deterministic guardrails → typed cluster-name confirmation → typed `APPROVE` → evidence-hash + TTL binding → apply-time re-verification → deterministic status prefixes) now covers four additional operations. **None of them bypass the human approval gate** — there is no auto-approve or skip-approval path anywhere.\n\nSelect the operation with `--operation`. Default is `upgrade` (unchanged behavior).\n\n| Operation | CLI | Approval identity is bound to | Operation-specific guardrails |\n|-----------|-----|-------------------------------|-------------------------------|\n| **upgrade** (default) | `--operation upgrade --target-version 1.34` | the target version | single-minor step, version format |\n| **launch** | `--operation launch --target-version 1.34` | the target version | cluster must **not already exist** (status `ABSENT`), version format |\n| **scale** | `--operation scale --nodes 4` | the desired node count | node count sane (**no scale-to-zero**, no absurd counts), cluster ACTIVE |\n| **addon** | `--operation addon --addon vpc-cni` | the addon name | addon must be named (no blanket change), cluster ACTIVE |\n| **teardown** | `--operation teardown` | the cluster name | **strictest** — see below |\n\nEvery operation still runs the operation-agnostic checks: typed cluster-name match, region allow-list, and the pre-check verdict scan.\n\n### Teardown is gated hardest\n\nDestroying a cluster is total and irreversible, so `teardown` adds controls on top of everything above:\n\n- **Always two-person** — teardown requires **two distinct approvers regardless of whether the cluster looks like production** (`gates.ALWAYS_TWO_PERSON_OPERATIONS`). The apply gate *and* the executor tool each independently enforce this, even if a stored `required_approvers` were somehow lower.\n- **Cluster name typed twice** — the operator types the exact cluster name at two separate prompts (`gr_teardown_double_confirm`). A single mistyped/auto-filled prompt can't trigger a destroy.\n- **Production is refused outright** — a cluster matching `PROD_CLUSTER_MARKERS` is blocked at the guardrail layer (`gr_teardown_not_prod`), a hard stop, not a warning.\n- **Cluster must be ACTIVE** — you can't tear down something mid-update.\n\n### How the generalized gate stays backward-compatible\n\nThe approval store identity was generalized from `target_version` to an `(operation, target)` pair. To keep the existing upgrade path and its tests byte-for-byte identical:\n- `operation` defaults to `\"upgrade\"` everywhere.\n- For `upgrade`, the `target` **is** the target version, so the historical binding is preserved.\n- A stored approval record with **no** `operation` key is treated as `\"upgrade\"`.\n\nCross-operation isolation is enforced: a `scale` approval can never authorize a `teardown`, a `launch` approval can never authorize an `upgrade`, and an approval for `scale --nodes 5` can't authorize `scale --nodes 9`. (Covered by `tests/test_operations.py`.)\n\nAll operations are **Terraform-driven** through the same `TERRAFORM_DIR` — the agent never hand-rolls AWS API creates/destroys. Launch/scale/addon/teardown live in `app/tools/operation_tools.py` and return the same deterministic `SUCCESS` / `BLOCKED` / `FAILED` prefixes the apply path keys off. (The phased live-monitor path that can emit `COMPLETED WITH ALARM` remains specific to `upgrade`.)\n\n## What it upgrades\n\nThe cluster is managed by Terraform (point `TERRAFORM_DIR` at your EKS Terraform — see **Terraform setup** below). A version upgrade is a single variable change:\n\n```hcl\nvariable \"eks_version\" {\n  default = \"1.33\"   # ← the agent proposes bumping this to the human's target\n}\n```\n\nTerraform then upgrades the control plane and node groups through the official `terraform-aws-modules/eks/aws` module.\n\n## Project layout\n\n```\nguarded-eks-upgrade-agent/\n├── app/\n│   ├── main.py            # human-driven entrypoint (input → checks → approval → apply → validate)\n│   ├── crew.py            # CrewAI agents + tasks + crew builder\n│   ├── guardrails.py      # deterministic (non-LLM) blocking checks\n│   ├── approval_gate.py   # evidence-hash + TTL + account-bound approvals, two-person for prod\n│   ├── config.py          # settings + guardrail/approval/availability policy\n│   ├── tools/\n│   │   ├── eks_tools.py       # pre-upgrade checks (version, APIs, addons, PDB, capacity, surge quota)\n│   │   ├── upgrade_tools.py   # phased terraform plan/apply (apply is gated) + sequence gate\n│   │   └── health_tools.py    # baseline, live availability monitor, halt-on-breach, validation loop\n│   ├── requirements.txt\n│   └── .env.example\n├── bin/                   # helper scripts: setup.sh/.ps1, run.sh/.ps1, test.sh\n├── tests/                 # pytest — runnable proof of the safety logic (no cluster needed)\n├── .github/workflows/eks-upgrade.yml   # CI with a human approval Environment gate\n├── docs/DEPLOYMENT-GUIDE.md · docs/TESTING-GUIDE.md\n├── DEMO-GUIDE.md\n├── .gitignore\n└── README.md\n```\n\n## Setup\n\n### Requirements\n- **Python 3.10–3.13** for the live agent. ⚠️ **Not 3.14** — CrewAI requires\n  `>=3.10,<3.14`, so it will not install on Python 3.14. (The unit tests in\n  `tests/` run on any version, including 3.14, via a tool shim.)\n- `aws` CLI (configured), `kubectl`, and `terraform` on your PATH.\n- An LLM key (e.g. `OPENROUTER_API_KEY`, or Groq etc. per `CREWAI_LLM`).\n\n### Quickest path — the setup script (recommended)\nThe scripts in `bin/` handle the Python-version / venv / install dance for you.\nThey auto-pick a compatible Python (3.13 → 3.12 → 3.11), create `app/venv`,\ninstall dependencies, create `app/.env` from the example, and run a safe\n`--status` check.\n\n**Git Bash:**\n```bash\nbash bin/setup.sh\n```\n**PowerShell:**\n```powershell\n./bin/setup.ps1\n```\n\nThen edit `app/.env` and wire kubectl to your cluster:\n```bash\n# app/.env\nEKS_CLUSTER_NAME=expense-dev\nAWS_REGION=us-east-1\nTERRAFORM_DIR=/absolute/path/to/your/Terraform   # dir with the eks_version variable\nCREWAI_LLM=openrouter/openrouter/free            # + the matching API key\n\naws eks update-kubeconfig --name expense-dev --region us-east-1\n```\n\n### Manual setup (if you prefer)\n```bash\ncd app\npy -3.13 -m venv venv                 # 3.13/3.12/3.11 — NOT 3.14\nsource venv/Scripts/activate          # Git Bash;  PowerShell: venv\\Scripts\\activate\npip install -r requirements.txt\ncp .env.example .env                  # then edit it\npython main.py --status\n```\n\n### Running the unit tests (no cluster, no CrewAI needed)\n```bash\npip install pytest python-dotenv\npython -m pytest tests/ -v            # 97 passing — proves the safety logic\n# or:  bash bin/test.sh\n```\n\n## Terraform setup\n\nThis project drives an existing Terraform config that manages the EKS cluster —\nit does **not** duplicate that infrastructure code (duplication causes drift).\nPoint it at your Terraform in one of two ways:\n\n**Option A — set the path (recommended for local runs):**\n```bash\n# in app/.env\nTERRAFORM_DIR=/absolute/path/to/your/Terraform   # the dir with the eks_version variable\n```\n\n**Option B — vendor a copy into this repo** (for a self-contained CI run):\n```bash\n# copy ONLY the .tf files + tfvars, NOT .terraform/, *.tfstate, or backups\nmkdir Terraform\ncp /path/to/source/Terraform/*.tf Terraform/\ncp -r /path/to/source/Terraform/tfvars Terraform/\ncp /path/to/source/Terraform/.terraform.lock.hcl Terraform/\n# then remove any committed backend state / init dir before committing\n```\nThe GitHub Actions workflow expects the Terraform at `./Terraform` (Option B).\n\nThe only variable the agent changes is `eks_version` — it passes\n`-var=\"eks_version=<target>\"` to plan/apply. Everything else in your Terraform\nstays as-is.\n\n## Usage\n\n> `bin/run.sh` (Bash) / `bin/run.ps1` (PowerShell) run the agent through the\n> venv automatically — **no need to activate it**. Or activate the venv and call\n> `python main.py` directly. Both forms are shown below.\n\n### Local (interactive, single approver)\n```bash\n# via the run script (recommended — uses the venv automatically)\nbash bin/run.sh --target-version 1.34\n# 1. Read-only pre-checks + terraform plan run and print evidence\n# 2. You type the cluster name to confirm; deterministic guardrails run\n# 3. You type APPROVE (agents cannot self-approve)\nbash bin/run.sh --apply --target-version 1.34\n# 4. Executor re-verifies approval, applies (control plane -> nodes), validates\n\n# equivalent, with the venv activated:\n#   cd app && python main.py --target-version 1.34\n#             python main.py --apply --target-version 1.34\n```\n\n### Two-person approval (production)\n```bash\nbash bin/run.sh --target-version 1.34 --actor alice          # opens request + 1st approval\nbash bin/run.sh --target-version 1.34 --actor bob --approve  # DISTINCT 2nd approver\nbash bin/run.sh --status                                     # shows 2/2 APPROVED\nbash bin/run.sh --apply --target-version 1.34\n```\n\n### Utility commands\n```bash\nbash bin/run.sh --status     # show gate state + who has approved (safe, no cluster calls)\nbash bin/run.sh --reset      # clear the current decision\n```\n\nPowerShell equivalents: `./bin/run.ps1 --status`, `./bin/run.ps1 --target-version 1.34`, etc.\n\n### CI/CD (GitHub Actions, approval via Environment)\nTrigger **Actions → EKS Upgrade (Agent + Human Approval) → Run workflow**, enter\nthe target version. Then:\n1. **Job 1 (precheck-plan)** runs read-only pre-checks + `terraform plan`. Always safe.\n2. **Job 2 (apply)** is gated behind the `production-eks-upgrade` **GitHub Environment**.\n   It pauses until a **required reviewer approves**. Only then does it apply.\n\nSet it up once: **Settings → Environments → New environment →\n`production-eks-upgrade` → add Required reviewers**. That reviewer approval is\nthe human-in-the-loop gate in CI (equivalent to the `APPROVE` prompt locally).\n\n**Required repo config (Settings → Secrets and variables → Actions):**\n- Variables: `AWS_ROLE_ARN`, `AWS_REGION`, `EKS_CLUSTER_NAME`, `CREWAI_LLM`\n- Secrets: `OPENROUTER_API_KEY`\n\n## Safety notes (read before running against a real cluster)\n\n- **EKS upgrades are not reversible.** You cannot downgrade a control plane. Treat every run as one-way.\n- **Test on a throwaway cluster first**, never a cluster you care about.\n- The approval gate is deliberate friction. Do not automate away the `APPROVE` prompt — it is the whole safety model.\n\n---\n\n*This is a portfolio/demonstration project showing safe, human-approved automation of a high-risk operation. It reuses the CrewAI agent patterns and the approval-gate concept from the companion `devops-microservices-crewAi` and `End-End-Project-Automate` projects.*\n","readmeExcerpt":"devops-microservices-crewAi This repo contains **two related things** that share one EKS cluster and Terraform/AWS foundation: 1. **A microservices platform** — three Flask services (order, payment, user) built and deployed to EKS via GitHub Actions → ECR → Helm (GitOps). This is the day-to-day app CI/CD. See **$1** below. 2. **A human-approved EKS-upgrade agent** (CrewAI) — an AI multi-agent workflow that safely upg","codeSnippets":[],"executableExamples":[{"language":"text","snippet":"app/{order,payment,user}-service/   # Flask app + Dockerfile + requirements.txt\ncharts/microservice/                # shared Helm chart + per-service values\nTerraform/                          # VPC, EKS, IAM/IRSA, ECR, backend\n.github/workflows/ci-cd.yml         # build -> push to ECR -> GitOps commit\n.github/workflows/codeql.yml        # static analysis"},{"language":"bash","snippet":"cd Terraform\nterraform init -backend-config=tfvars/dev/backend.tfvars\nterraform plan  -var-file=tfvars/dev/dev.tfvars\nterraform apply -var-file=tfvars/dev/dev.tfvars"},{"language":"bash","snippet":"aws eks update-kubeconfig --name expense-dev --region us-east-1\nkubectl get nodes"},{"language":"bash","snippet":"helm repo add eks https://aws.github.io/eks-charts && helm repo update\nhelm install aws-load-balancer-controller eks/aws-load-balancer-controller \\\n  -n kube-system -f alb-controller-values.yaml"},{"language":"bash","snippet":"helm upgrade --install order-service charts/microservice \\\n  -f charts/microservice/values.yaml -f charts/microservice/values-order.yaml -n default\n# repeat for payment-service / user-service"},{"language":"bash","snippet":"kubectl get pods\nkubectl get ingress   # ALB DNS name"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[],"languages":["python"],"docsSourceLabel":"GITHUB REPOS","editorialOverview":"Production-ready microservices platform on AWS EKS using GitHub Actions for CI and ArgoCD for CD (GitOps). This project uses CrewAI with a hierarchical process (manager agent auto-delegates) to run a team of specialized DevOps agents that monitor infrastructure, detect issues, and optimize costs. devops-microservices-crewAi This repo contains **two related things** that share one EKS cluster and Terraform/AWS foundation: 1. **A microservices platform** — three Flask services (order, payment, user) built and deployed to EKS via GitHub Actions → ECR → Helm (GitOps). This is the day-to-day app CI/CD. See **$1** below. 2. **A human-approved EKS-upgrade agent** (CrewAI) — an AI multi-agent workflow that safely upg","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":419,"uniquenessScore":66,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T11:50:37.719Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T11:50:37.719Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T04:07:48.499Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/github_repos","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}