{"id":"3ec05000-5453-47d6-99d1-6770c455ac3d","slug":"clawhub-gechengling-ai-agent-evaluator","name":"AI Agent Evaluator","description":"AI-powered agent evaluation and benchmarking assistant — design evaluation suites, run structured assessments (task completion rate, latency, safety, reasoning accuracy), compare multi-agent frameworks (CrewAI, LangChain, AutoGen), generate benchmark reports, and guide developers in selecting the right evaluation methodology. Built for AI engineers, product managers, and ML teams shipping agent-based applications to production. Keywords: AI agent evaluation, agent benchmarking, LLM testing, CrewAI, AutoGen, LangChain, SWE-bench, AgentBench, AI quality assurance, agent reliability.","canonicalUrl":"https://www.xpersona.co/agent/clawhub-gechengling-ai-agent-evaluator","sourceUrl":"https://clawhub.ai/gechengling/ai-agent-evaluator","homepage":"https://clawhub.ai/gechengling/skills/ai-agent-evaluator","source":"CLAWHUB","vendor":{"slug":"clawhub","label":"Clawhub","url":"https://clawhub.ai/gechengling/skills/ai-agent-evaluator"},"protocols":["OPENCLEW"],"capabilities":[],"trustScore":null,"trustConfidence":"unknown","artifactCount":0,"benchmarkCount":0,"lastRelease":"3.0.3","freshnessAt":"2026-10-10T16:23:34.879Z","freshnessLabel":"Oct 10, 2026","securityReviewed":true,"openapiReady":false,"stats":[{"label":"Trust score","value":"Unknown"},{"label":"Compatibility","value":"OpenClaw"},{"label":"Freshness","value":"Oct 10, 2026"},{"label":"Vendor","value":"Clawhub"},{"label":"Artifacts","value":"0"},{"label":"Benchmarks","value":"0"},{"label":"Last release","value":"3.0.3"}],"factsPreview":[{"factKey":"vendor","category":"vendor","label":"Vendor","value":"Clawhub","href":"https://clawhub.ai/gechengling/skills/ai-agent-evaluator","sourceUrl":"https://clawhub.ai/gechengling/skills/ai-agent-evaluator","sourceType":"profile","confidence":"medium","observedAt":"2026-10-10T16:23:34.880Z","isPublic":true},{"factKey":"protocols","category":"compatibility","label":"Protocol compatibility","value":"OpenClaw","href":"https://www.xpersona.co/api/v1/agents/clawhub-gechengling-ai-agent-evaluator/contract","sourceUrl":"https://www.xpersona.co/api/v1/agents/clawhub-gechengling-ai-agent-evaluator/contract","sourceType":"contract","confidence":"medium","observedAt":"2026-10-10T16:23:34.880Z","isPublic":true},{"factKey":"traction","category":"adoption","label":"Adoption signal","value":"1.3K downloads","href":"https://clawhub.ai/gechengling/ai-agent-evaluator","sourceUrl":"https://clawhub.ai/gechengling/ai-agent-evaluator","sourceType":"profile","confidence":"medium","observedAt":"2026-10-10T16:23:34.880Z","isPublic":true},{"factKey":"latest_release","category":"release","label":"Latest release","value":"3.0.3","href":"https://clawhub.ai/gechengling/ai-agent-evaluator","sourceUrl":"https://clawhub.ai/gechengling/ai-agent-evaluator","sourceType":"release","confidence":"medium","observedAt":"2026-09-15T14:25:03.729Z","isPublic":true},{"factKey":"handshake_status","category":"security","label":"Handshake status","value":"UNKNOWN","href":"https://www.xpersona.co/api/v1/agents/clawhub-gechengling-ai-agent-evaluator/trust","sourceUrl":"https://www.xpersona.co/api/v1/agents/clawhub-gechengling-ai-agent-evaluator/trust","sourceType":"trust","confidence":"medium","observedAt":null,"isPublic":true}],"highlights":["1.3K downloads","Trust evidence available"],"agentCard":{"name":"AI Agent Evaluator","description":"AI-powered agent evaluation and benchmarking assistant — design evaluation suites, run structured assessments (task completion rate, latency, safety, reasoning accuracy), compare multi-agent frameworks (CrewAI, LangChain, AutoGen), generate benchmark reports, and guide developers in selecting the right evaluation methodology. Built for AI engineers, product managers, and ML teams shipping agent-based applications to production. Keywords: AI agent evaluation, agent benchmarking, LLM testing, CrewAI, AutoGen, LangChain, SWE-bench, AgentBench, AI quality assurance, agent reliability.","source":"CLAWHUB","sourceId":"clawhub:s17ewqc4f2s6gpcbm88hy7fgvn85kg1g:ai-agent-evaluator","homepage":"https://clawhub.ai/gechengling/skills/ai-agent-evaluator","repository":"https://clawhub.ai/gechengling/ai-agent-evaluator","documentation":"https://www.xpersona.co/agent/clawhub-gechengling-ai-agent-evaluator","protocols":["OPENCLEW"]}}