{"id":"cacd548e-64c1-4ed8-9c9e-12665a94db44","slug":"crewai-autonoma-tools-crewai-evaluation","name":"crewai-evaluation","description":"A CrewAI support-triage crew tested three ways: crewai test CLI baseline scoring, DeepEval span-level assertions, and LangWatch Scenario delegation testing.","canonicalUrl":"https://www.xpersona.co/skill/crewai-autonoma-tools-crewai-evaluation","sourceUrl":"https://github.com/Autonoma-Tools/crewai-evaluation","homepage":null,"source":"GITHUB_REPOS","vendor":{"slug":"autonoma-tools","label":"Autonoma Tools","url":"https://github.com/Autonoma-Tools/crewai-evaluation"},"protocols":["OPENCLEW"],"capabilities":["crewai","multi-agent"],"trustScore":null,"trustConfidence":"unknown","artifactCount":0,"benchmarkCount":0,"lastRelease":null,"freshnessAt":"2026-10-09T16:16:47.033Z","freshnessLabel":"Oct 9, 2026","securityReviewed":true,"openapiReady":false,"stats":[{"label":"Trust score","value":"Unknown"},{"label":"Compatibility","value":"OpenClaw"},{"label":"Freshness","value":"Oct 9, 2026"},{"label":"Vendor","value":"Autonoma Tools"},{"label":"Artifacts","value":"0"},{"label":"Benchmarks","value":"0"},{"label":"Last release","value":"Unpublished"}],"factsPreview":[{"factKey":"vendor","category":"vendor","label":"Vendor","value":"Autonoma Tools","href":"https://github.com/Autonoma-Tools/crewai-evaluation","sourceUrl":"https://github.com/Autonoma-Tools/crewai-evaluation","sourceType":"profile","confidence":"medium","observedAt":"2026-10-09T16:16:47.043Z","isPublic":true},{"factKey":"protocols","category":"compatibility","label":"Protocol compatibility","value":"OpenClaw","href":"https://www.xpersona.co/api/v1/agents/crewai-autonoma-tools-crewai-evaluation/contract","sourceUrl":"https://www.xpersona.co/api/v1/agents/crewai-autonoma-tools-crewai-evaluation/contract","sourceType":"contract","confidence":"medium","observedAt":"2026-10-09T16:16:47.043Z","isPublic":true},{"factKey":"docs_crawl","category":"integration","label":"Crawlable docs","value":"6 indexed pages on the official domain","href":"https://github.com/login?return_to=https%3A%2F%2Fgithub.com%2Fopenclaw%2Fskills%2Ftree%2Fmain%2Fskills%2Fasleep123%2Fcaldav-calendar","sourceUrl":"https://github.com/login?return_to=https%3A%2F%2Fgithub.com%2Fopenclaw%2Fskills%2Ftree%2Fmain%2Fskills%2Fasleep123%2Fcaldav-calendar","sourceType":"search_document","confidence":"medium","observedAt":"2026-04-15T05:03:46.393Z","isPublic":true},{"factKey":"handshake_status","category":"security","label":"Handshake status","value":"UNKNOWN","href":"https://www.xpersona.co/api/v1/agents/crewai-autonoma-tools-crewai-evaluation/trust","sourceUrl":"https://www.xpersona.co/api/v1/agents/crewai-autonoma-tools-crewai-evaluation/trust","sourceType":"trust","confidence":"medium","observedAt":null,"isPublic":true}],"highlights":["Trust evidence available"],"agentCard":{"name":"crewai-evaluation","description":"A CrewAI support-triage crew tested three ways: crewai test CLI baseline scoring, DeepEval span-level assertions, and LangWatch Scenario delegation testing.","source":"GITHUB_REPOS","sourceId":"crewai:1310900941","repository":"https://github.com/Autonoma-Tools/crewai-evaluation","documentation":"https://www.xpersona.co/skill/crewai-autonoma-tools-crewai-evaluation/agent/crewai-autonoma-tools-crewai-evaluation","protocols":["OPENCLEW"],"capabilities":["crewai","multi-agent"],"languages":["python"],"install":{"command":"git clone https://github.com/Autonoma-Tools/crewai-evaluation.git","ecosystem":"git"},"examples":[{"kind":"example","language":"bash","snippet":"git clone https://github.com/Autonoma-Tools/crewai-evaluation.git\ncd crewai-evaluation\npip install -r requirements.txt\nexport OPENAI_API_KEY=sk-...\n\n# Layer 1 - baseline quality score via the crewai test CLI\nbash scripts/baseline_eval.sh\n\n# Layer 2 - span-level tool-call assertions with DeepEval\npytest tests/test_deepeval_spans.py\n\n# Layer 3 - multi-agent delegation testing with LangWatch Scenario\npytest tests/test_delegation_scenario.py"},{"kind":"example","language":"text","snippet":"crewai-evaluation/\n├── src/\n│   └── crew.py                        # the two-agent crew every layer tests against\n├── scripts/\n│   └── baseline_eval.sh               # Layer 1: crewai test CLI baseline scoring\n├── tests/\n│   ├── test_deepeval_spans.py         # Layer 2: DeepEval span-level tool-call assertions\n│   └── test_delegation_scenario.py    # Layer 3: LangWatch Scenario delegation test\n├── examples/\n│   ├── autogen_groupchat_test.py      # AutoGen equivalent of the delegation assertion\n│   └── openai_agents_sdk_test.py      # OpenAI Agents SDK equivalent (HandoffOutputItem)\n├── .github/workflows/agent-tests.yml  # PR smoke checks + nightly delegation suite\n└── requirements.txt"}]}}