{"id":"944c9f3e-527d-4bc5-b837-85ff0a2ebe22","slug":"crewai-bullpeng72-agent-evaluator","name":"Agent-Evaluator","description":"LLM agent evaluation framework with 7 Harness Gates (A–G) and 58 metrics. Supports   LangChain, CrewAI, AutoGen, DSPy, PydanticAI. Native LLM-as-Judge, OTEL tracing, FastAPI   dashboard.","canonicalUrl":"https://www.xpersona.co/skill/crewai-bullpeng72-agent-evaluator","sourceUrl":"https://github.com/bullpeng72/Agent-Evaluator","homepage":null,"source":"GITHUB_REPOS","vendor":{"slug":"bullpeng72","label":"Bullpeng72","url":"https://github.com/bullpeng72/Agent-Evaluator"},"protocols":["OPENCLEW"],"capabilities":["crewai","multi-agent"],"trustScore":null,"trustConfidence":"unknown","artifactCount":0,"benchmarkCount":0,"lastRelease":null,"freshnessAt":"2026-10-09T12:48:04.943Z","freshnessLabel":"Oct 9, 2026","securityReviewed":true,"openapiReady":false,"stats":[{"label":"Trust score","value":"Unknown"},{"label":"Compatibility","value":"OpenClaw"},{"label":"Freshness","value":"Oct 9, 2026"},{"label":"Vendor","value":"Bullpeng72"},{"label":"Artifacts","value":"0"},{"label":"Benchmarks","value":"0"},{"label":"Last release","value":"Unpublished"}],"factsPreview":[{"factKey":"vendor","category":"vendor","label":"Vendor","value":"Bullpeng72","href":"https://github.com/bullpeng72/Agent-Evaluator","sourceUrl":"https://github.com/bullpeng72/Agent-Evaluator","sourceType":"profile","confidence":"medium","observedAt":"2026-10-09T12:48:04.956Z","isPublic":true},{"factKey":"protocols","category":"compatibility","label":"Protocol compatibility","value":"OpenClaw","href":"https://www.xpersona.co/api/v1/agents/crewai-bullpeng72-agent-evaluator/contract","sourceUrl":"https://www.xpersona.co/api/v1/agents/crewai-bullpeng72-agent-evaluator/contract","sourceType":"contract","confidence":"medium","observedAt":"2026-10-09T12:48:04.956Z","isPublic":true},{"factKey":"traction","category":"adoption","label":"Adoption signal","value":"1 GitHub stars","href":"https://github.com/bullpeng72/Agent-Evaluator","sourceUrl":"https://github.com/bullpeng72/Agent-Evaluator","sourceType":"profile","confidence":"medium","observedAt":"2026-10-09T12:48:04.956Z","isPublic":true},{"factKey":"docs_crawl","category":"integration","label":"Crawlable docs","value":"6 indexed pages on the official domain","href":"https://github.com/login?return_to=https%3A%2F%2Fgithub.com%2Fopenclaw%2Fskills%2Ftree%2Fmain%2Fskills%2Fasleep123%2Fcaldav-calendar","sourceUrl":"https://github.com/login?return_to=https%3A%2F%2Fgithub.com%2Fopenclaw%2Fskills%2Ftree%2Fmain%2Fskills%2Fasleep123%2Fcaldav-calendar","sourceType":"search_document","confidence":"medium","observedAt":"2026-04-15T05:03:46.393Z","isPublic":true},{"factKey":"handshake_status","category":"security","label":"Handshake status","value":"UNKNOWN","href":"https://www.xpersona.co/api/v1/agents/crewai-bullpeng72-agent-evaluator/trust","sourceUrl":"https://www.xpersona.co/api/v1/agents/crewai-bullpeng72-agent-evaluator/trust","sourceType":"trust","confidence":"medium","observedAt":null,"isPublic":true}],"highlights":["1 GitHub stars","Trust evidence available"],"agentCard":{"name":"Agent-Evaluator","description":"LLM agent evaluation framework with 7 Harness Gates (A–G) and 58 metrics. Supports   LangChain, CrewAI, AutoGen, DSPy, PydanticAI. Native LLM-as-Judge, OTEL tracing, FastAPI   dashboard.","source":"GITHUB_REPOS","sourceId":"crewai:1239437708","repository":"https://github.com/bullpeng72/Agent-Evaluator","documentation":"https://www.xpersona.co/skill/crewai-bullpeng72-agent-evaluator/agent/crewai-bullpeng72-agent-evaluator","protocols":["OPENCLEW"],"capabilities":["crewai","multi-agent"],"languages":["python"],"install":{"command":"git clone https://github.com/bullpeng72/Agent-Evaluator.git","ecosystem":"git"},"examples":[{"kind":"example","language":"bash","snippet":"pip install agent-evaluator"},{"kind":"example","language":"python","snippet":"from agent_evaluator import QuickEval\n\neval = QuickEval(\"results/\")\n\n@eval.qa\ndef my_agent(question: str, ground_truth: str = \"\") -> str:\n    return llm.invoke(question)          # your agent code — unchanged\n\nmy_agent(\"What is the capital of South Korea?\", ground_truth=\"Seoul\")\n\neval.save()                                        # results/quickeval.json + .html\neval.gate(tcr=85, accuracy=70, hallucination=5)    # CI/CD gate — sys.exit(1) if unmet"}]}}