{"id":"4eaae90d-6332-4173-a79d-4518df5555e3","slug":"crewai-haaswithsai-mac-bench-multi-agent-coordination-failure-b","name":"MAC-Bench-Multi-Agent-Coordination-Failure-Benchmark","description":"MAC-Bench: A diagnostic benchmark and evaluation suite to detect and analyze behavioral coordination failures across multi-agent LLM systems (LangGraph, AutoGen, CrewAI).","canonicalUrl":"https://www.xpersona.co/agent/crewai-haaswithsai-mac-bench-multi-agent-coordination-failure-b","sourceUrl":"https://github.com/HaaswithSai/MAC-Bench-Multi-Agent-Coordination-Failure-Benchmark","homepage":null,"source":"GITHUB_REPOS","vendor":{"slug":"haaswithsai","label":"Haaswithsai","url":"https://github.com/HaaswithSai/MAC-Bench-Multi-Agent-Coordination-Failure-Benchmark"},"protocols":["OPENCLEW"],"capabilities":["crewai","multi-agent"],"trustScore":null,"trustConfidence":"unknown","artifactCount":0,"benchmarkCount":0,"lastRelease":null,"freshnessAt":"2026-10-09T15:16:50.204Z","freshnessLabel":"Oct 9, 2026","securityReviewed":true,"openapiReady":false,"stats":[{"label":"Trust score","value":"Unknown"},{"label":"Compatibility","value":"OpenClaw"},{"label":"Freshness","value":"Oct 9, 2026"},{"label":"Vendor","value":"Haaswithsai"},{"label":"Artifacts","value":"0"},{"label":"Benchmarks","value":"0"},{"label":"Last release","value":"Unpublished"}],"factsPreview":[{"factKey":"vendor","category":"vendor","label":"Vendor","value":"Haaswithsai","href":"https://github.com/HaaswithSai/MAC-Bench-Multi-Agent-Coordination-Failure-Benchmark","sourceUrl":"https://github.com/HaaswithSai/MAC-Bench-Multi-Agent-Coordination-Failure-Benchmark","sourceType":"profile","confidence":"medium","observedAt":"2026-10-09T15:16:50.206Z","isPublic":true},{"factKey":"protocols","category":"compatibility","label":"Protocol compatibility","value":"OpenClaw","href":"https://www.xpersona.co/api/v1/agents/crewai-haaswithsai-mac-bench-multi-agent-coordination-failure-b/contract","sourceUrl":"https://www.xpersona.co/api/v1/agents/crewai-haaswithsai-mac-bench-multi-agent-coordination-failure-b/contract","sourceType":"contract","confidence":"medium","observedAt":"2026-10-09T15:16:50.206Z","isPublic":true},{"factKey":"docs_crawl","category":"integration","label":"Crawlable docs","value":"6 indexed pages on the official domain","href":"https://github.com/login?return_to=https%3A%2F%2Fgithub.com%2Fopenclaw%2Fskills%2Ftree%2Fmain%2Fskills%2Fasleep123%2Fcaldav-calendar","sourceUrl":"https://github.com/login?return_to=https%3A%2F%2Fgithub.com%2Fopenclaw%2Fskills%2Ftree%2Fmain%2Fskills%2Fasleep123%2Fcaldav-calendar","sourceType":"search_document","confidence":"medium","observedAt":"2026-04-15T05:03:46.393Z","isPublic":true},{"factKey":"handshake_status","category":"security","label":"Handshake status","value":"UNKNOWN","href":"https://www.xpersona.co/api/v1/agents/crewai-haaswithsai-mac-bench-multi-agent-coordination-failure-b/trust","sourceUrl":"https://www.xpersona.co/api/v1/agents/crewai-haaswithsai-mac-bench-multi-agent-coordination-failure-b/trust","sourceType":"trust","confidence":"medium","observedAt":null,"isPublic":true}],"highlights":["Trust evidence available"],"agentCard":{"name":"MAC-Bench-Multi-Agent-Coordination-Failure-Benchmark","description":"MAC-Bench: A diagnostic benchmark and evaluation suite to detect and analyze behavioral coordination failures across multi-agent LLM systems (LangGraph, AutoGen, CrewAI).","source":"GITHUB_REPOS","sourceId":"crewai:1329918766","repository":"https://github.com/HaaswithSai/MAC-Bench-Multi-Agent-Coordination-Failure-Benchmark","documentation":"https://www.xpersona.co/agent/crewai-haaswithsai-mac-bench-multi-agent-coordination-failure-b","protocols":["OPENCLEW"],"capabilities":["crewai","multi-agent"],"languages":["python"],"install":{"command":"git clone https://github.com/HaaswithSai/MAC-Bench-Multi-Agent-Coordination-Failure-Benchmark.git","ecosystem":"git"},"examples":[{"kind":"example","language":"bash","snippet":"# 1. Clone the repository\ngit clone https://github.com/<your-username>/mac-bench.git\ncd mac-bench\n\n# 2. Install package and dependencies\npip install -e .\n\n# 3. Launch the interactive dashboard\nstreamlit run frontend/dashboard.py"},{"kind":"example","language":"mermaid","snippet":"flowchart TD\n    subgraph Tasks[\"Task Suite (50 Tasks across 5 Archetypes)\"]\n        T1[\"🔒 Role Locked\"]\n        T2[\"🔁 Sequential Handoff\"]\n        T3[\"✅ Verification Gauntlet\"]\n        T4[\"⚖️ Debate & Deliberation\"]\n        T5[\"⚡ Resource Contention\"]\n    end\n\n    subgraph Frameworks[\"Framework Under Test\"]\n        FW1[\"LangGraph\"]\n        FW2[\"AutoGen\"]\n        FW3[\"CrewAI\"]\n    end\n\n    subgraph Proxy[\"Interceptor Proxy (FastAPI)\"]\n        PRX[\"Transparent OpenAI/LiteLLM Proxy\"]\n        LOG[\"Turn-Level JSONL Logger\\n(Tokens, Tool Calls, Timestamps)\"]\n    end\n\n    subgraph Evaluation[\"Diagnostic Swarm (19 Detectors)\"]\n        DET1[\"Rule & Programmatic Engines\\n(Loop, Stagnation, Dominance, Handoff)\"]\n        DET2[\"LLM-as-a-Judge Detectors\\n(Echo Chamber, Role Drift, Siloing)\"]\n        DET3[\"Task Evaluators\\n(Ground Truth Exact / Semantic Match)\"]\n    end\n\n    subgraph Presentation[\"UI & Analytics\"]\n        DASH[\"Streamlit Dashboard\"]\n        RPT[\"Validation Reports (IAA)\"]\n    end\n\n    Tasks --> Frameworks\n    Frameworks -->|API Requests| PRX\n    PRX --> LOG\n    LOG --> Evaluation\n    Evaluation --> DASH\n    Evaluation --> RPT"}]}}