{"id":"8a798746-e359-4cde-9bdc-603f88d86827","entityType":"agent","slug":"crewai-ismail-2001-agent-bench","name":"agent-bench","canonicalUrl":"https://www.xpersona.co/agent/crewai-ismail-2001-agent-bench","canonicalPath":"/agent/crewai-ismail-2001-agent-bench","generatedAt":"2026-10-09T01:03:02.528Z","source":"GITHUB_OPENCLEW","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-05-18T06:45:23.701Z","emptyReason":null},"description":"Industrial-grade benchmarking engine for AI agents. Define test scenarios in YAML, run high-performance parallel evaluations, and generate premium glassmorphism reports. Supports LangGraph, CrewAI, AutoGen, and custom agent stacks. <div align=\"center\"> <img src=\"https://raw.githubusercontent.com/lucide-icons/lucide/main/icons/layers.svg\" width=\"80\" height=\"80\" /> <h1>agentbench</h1> <p><strong>Industrial-Grade Pytest for AI Agents.</strong></p> <div> <a href=\"https://github.com/Ismail-2001/agent-bench/actions\"> <img src=\"https://img.shields.io/badge/CI-Passing-success?style=for-the-badge&logo=github-actions&logoColor=white\" alt=\"CI Status\" /> <","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 2 GitHub stars reported by the source. Last updated 5/18/2026.","installCommand":"git clone https://github.com/Ismail-2001/agent-bench.git","sourceUrl":"https://github.com/Ismail-2001/agent-bench","homepage":null,"primaryLinks":[{"label":"View Source","url":"https://github.com/Ismail-2001/agent-bench","kind":"source"}],"safetyScore":66,"overallRank":24.4,"popularityScore":12,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Industrial-grade benchmarking engine for AI agents. Define test scenarios in YAML, run high-performance parallel evaluations, and generate premium glassmorphism"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-05-18T06:45:23.701Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[{"label":"crewai","status":"self-declared"},{"label":"multi-agent","status":"self-declared"}],"verifiedCount":0,"selfDeclaredCount":3,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"},{"key":"crewai","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"},{"key":"multi-agent","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile capability:crewai|supported|profile capability:multi-agent|supported|profile"}},"adoption":{"evidence":{"source":"GITHUB OPENCLEW","verified":false,"confidence":"medium","updatedAt":"2026-05-18T06:45:23.701Z","emptyReason":null},"stars":2,"forks":0,"downloads":null,"packageName":null,"latestVersion":null,"tractionLabel":"2 GitHub stars"},"release":{"evidence":{"source":"agent-index","verified":false,"confidence":"medium","updatedAt":"2026-05-18T06:45:23.700Z","emptyReason":null},"lastUpdatedAt":"2026-05-18T06:45:23.701Z","lastCrawledAt":"2026-05-18T06:45:23.700Z","lastIndexedAt":null,"nextCrawlAt":"2026-05-25T06:45:23.700Z","lastVerifiedAt":null,"highlights":[]},"execution":{"evidence":{"source":"GITHUB OPENCLEW","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"git clone https://github.com/Ismail-2001/agent-bench.git","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/crewai-ismail-2001-agent-bench/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/crewai-ismail-2001-agent-bench/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/crewai-ismail-2001-agent-bench/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/crewai-ismail-2001-agent-bench/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/crewai-ismail-2001-agent-bench/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/crewai-ismail-2001-agent-bench/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"GITHUB_OPENCLEW","generatedAt":"2026-10-09T01:03:02.527Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/crewai-ismail-2001-agent-bench/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/crewai-ismail-2001-agent-bench/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/crewai-ismail-2001-agent-bench/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/crewai-ismail-2001-agent-bench/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"GITHUB OPENCLEW","verified":false,"confidence":"high","updatedAt":"2026-05-18T06:45:23.701Z","emptyReason":null},"readme":"<div align=\"center\">\n  <img src=\"https://raw.githubusercontent.com/lucide-icons/lucide/main/icons/layers.svg\" width=\"80\" height=\"80\" />\n  <h1>agentbench</h1>\n  <p><strong>Industrial-Grade Pytest for AI Agents.</strong></p>\n\n  <div>\n    <a href=\"https://github.com/Ismail-2001/agent-bench/actions\">\n      <img src=\"https://img.shields.io/badge/CI-Passing-success?style=for-the-badge&logo=github-actions&logoColor=white\" alt=\"CI Status\" />\n    </a>\n    <a href=\"https://pypi.org/project/agentbench/\">\n      <img src=\"https://img.shields.io/badge/pypi-v0.1.1-blue?style=for-the-badge&logo=pypi&logoColor=white\" alt=\"PyPI Version\" />\n    </a>\n    <a href=\"LICENSE\">\n      <img src=\"https://img.shields.io/badge/License-MIT-yellow.svg?style=for-the-badge\" alt=\"License\" />\n    </a>\n  </div>\n\n  <p>Define test scenarios in YAML. Benchmark any agent — LangGraph, CrewAI, AutoGen, or custom. Get premium reports with pass/fail, tokens, latency, cost, and failure analysis.</p>\n\n  <h4>\n    <a href=\"#-quick-start\">Quick Start</a>\n    <span> · </span>\n    <a href=\"#-features\">Features</a>\n    <span> · </span>\n    <a href=\"#-architecture\">Architecture</a>\n    <span> · </span>\n    <a href=\"#-reports\">Reports</a>\n    <span> · </span>\n    <a href=\"#-contributing\">Contributing</a>\n  </h4>\n</div>\n\n---\n\n## 🏗️ The Enterprise Challenge\n\nIn 2026, **52% of organizations** still don't run automated evaluations on their multi-step agent workflows. Existing tools are either ecosystem-locked (LangSmith) or too academic (THUDM/AgentBench).\n\n**agentbench** fills the gap: a free, open-source CLI engine that brings **deterministic and LLM-based testing** to the modern agent stack. Think of it as `pytest` meets `k6` for autonomous AI.\n\n---\n\n## ⚡ Quick Start\n\n### 1. Install via `uv` or `pip`\n```bash\npip install agentbench\n```\n\n### 2. Define a Scenario (`research.yaml`)\n```yaml\nname: \"basic-research\"\ntasks:\n  - id: \"compare-frameworks\"\n    input: \"Compare LangGraph and CrewAI for production systems in 2026.\"\n    criteria:\n      - type: contains_all\n        values: [\"LangGraph\", \"CrewAI\"]\n      - type: min_length\n        value: 200\n      - type: llm_judge\n        prompt: \"Does this provide a technical comparison? Score 0-10.\"\n        threshold: 7\n    limits:\n      max_tokens: 50000\n      max_latency_seconds: 60\n```\n\n### 3. Run with Your Agent\n```bash\nagentbench run --scenario scenarios/research.yaml --agent my_module:MyAgentAdapter --format html\n```\n\n---\n\n## 🎨 Professional Visualization\n\nOur reporter generates a **premium, glassmorphism-styled HTML dashboard** for every run.\n\n- **Dynamic Charts**: Visualize pass/fail trends and latency spikes.\n- **Deep Observability**: Click into any task to see raw inputs, outputs, and failing criteria.\n- **Cost Metrics**: Real-time token counting and cost estimation.\n\n> [!NOTE]\n> View a live example of the report aesthetics in the [documentation](docs/reporting.md).\n\n---\n\n## 🧩 Architecture\n\n```mermaid\ngraph TD\n    A[Scenario Loader] --> B[Parallel Runner]\n    B --> C[Agent Adapter]\n    C --> D[LangGraph / CrewAI / AutoGen]\n    B --> E[Evaluation Engine]\n    E --> F[Deterministic Evaluators]\n    E --> G[LLM-Judge / Semantic Check]\n    B --> H[Reporters]\n    H --> I[Rich CLI Table]\n    H --> J[Glassmorphism HTML]\n    H --> K[JSON Metadata]\n```\n\n---\n\n## 🚀 Key Features (FAANG Grade)\n\n- **⚡ Parallel Task Execution**: Benchmark large scenarios 10x faster with managed `asyncio` concurrency.\n- **🛡️ Built-in Scenario Packs**: Standardized benchmarks for `tool-use`, `research`, and `error-recovery`.\n- **👁️ Structured Observability**: High-fidelity logging with `structlog` for easy ingestion into Datadog/Splunk.\n- **🔌 Framework Agnostic**: A simple `AgentAdapter` interface allows you to test any agent in seconds.\n- **🐳 DevOps Ready**: Includes an optimized `Dockerfile` (using `uv`) and a comprehensive `Makefile`.\n\n---\n\n## 📊 Core Metrics Measured\n\n| Metric | Accuracy | How It's Measured |\n|--------|----------|-------------------|\n| **Pass/Fail** | 100% | All criteria must satisfy (deterministic + LLM) |\n| **Tokens** | 100% | Precise counting via `tiktoken` |\n| **Latency** | High | Monotonic wall-clock time from call to return |\n| **Cost** | Est. | Calculated from token count × model rates |\n| **Consistency** | High | Pass rate across multiple runs (optional) |\n\n---\n\n## 🤝 Contributing\n\nWe welcome contributions from the community! Please read our [CONTRIBUTING.md](CONTRIBUTING.md) to get started.\n\n**High-impact areas:**\n- **New evaluators**: (e.g., Trajectory efficiency, Tool-calling accuracy)\n- **Framework adapters**: (Pre-built adapters for popular SDKs)\n- **Reporters**: (Markdown, PDF, or Grafana dashboards)\n\n---\n\n## 📜 License\n\n[MIT](LICENSE) — Test everything. Trust nothing.\n\n---\n\n<div align=\"center\">\n  <sub>Built with ❤️ by <strong>Ismail Sajid</strong> (Re-architected for FAANG by Antigravity AI)</sub>\n</div>\n","readmeExcerpt":"<div align=\"center\"> <img src=\"https://raw.githubusercontent.com/lucide-icons/lucide/main/icons/layers.svg\" width=\"80\" height=\"80\" /> <h1>agentbench</h1> <p><strong>Industrial-Grade Pytest for AI Agents.</strong></p> <div> <a href=\"https://github.com/Ismail-2001/agent-bench/actions\"> <img src=\"https://img.shields.io/badge/CI-Passing-success?style=for-the-badge&logo=github-actions&logoColor=white\" alt=\"CI Status\" /> <","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"pip install agentbench"},{"language":"yaml","snippet":"name: \"basic-research\"\ntasks:\n  - id: \"compare-frameworks\"\n    input: \"Compare LangGraph and CrewAI for production systems in 2026.\"\n    criteria:\n      - type: contains_all\n        values: [\"LangGraph\", \"CrewAI\"]\n      - type: min_length\n        value: 200\n      - type: llm_judge\n        prompt: \"Does this provide a technical comparison? Score 0-10.\"\n        threshold: 7\n    limits:\n      max_tokens: 50000\n      max_latency_seconds: 60"},{"language":"bash","snippet":"agentbench run --scenario scenarios/research.yaml --agent my_module:MyAgentAdapter --format html"},{"language":"mermaid","snippet":"graph TD\n    A[Scenario Loader] --> B[Parallel Runner]\n    B --> C[Agent Adapter]\n    C --> D[LangGraph / CrewAI / AutoGen]\n    B --> E[Evaluation Engine]\n    E --> F[Deterministic Evaluators]\n    E --> G[LLM-Judge / Semantic Check]\n    B --> H[Reporters]\n    H --> I[Rich CLI Table]\n    H --> J[Glassmorphism HTML]\n    H --> K[JSON Metadata]"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[],"languages":["python"],"docsSourceLabel":"GITHUB OPENCLEW","editorialOverview":"Industrial-grade benchmarking engine for AI agents. Define test scenarios in YAML, run high-performance parallel evaluations, and generate premium glassmorphism reports. Supports LangGraph, CrewAI, AutoGen, and custom agent stacks. <div align=\"center\"> <img src=\"https://raw.githubusercontent.com/lucide-icons/lucide/main/icons/layers.svg\" width=\"80\" height=\"80\" /> <h1>agentbench</h1> <p><strong>Industrial-Grade Pytest for AI Agents.</strong></p> <div> <a href=\"https://github.com/Ismail-2001/agent-bench/actions\"> <img src=\"https://img.shields.io/badge/CI-Passing-success?style=for-the-badge&logo=github-actions&logoColor=white\" alt=\"CI Status\" /> <","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":396,"uniquenessScore":70,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-05-18T06:45:23.701Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-05-18T06:45:23.701Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T01:03:02.528Z","emptyReason":null},"items":[{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-10T18:48:31.762Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/github_openclew","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}