{"id":"899be3b7-ada1-4e0a-9b32-bd431bd78bc0","slug":"clawhub-rustyorb-agent-evaluation","name":"Agent Evaluation","description":"Testing and benchmarking LLM agents including behavioral testing, capability assessment, reliability metrics, and production monitoring—where even top agents achieve less than 50% on real-world benchmarks Use when: agent testing, agent evaluation, benchmark agents, agent reliability, test agent.","capabilities":[],"protocols":["OPENCLAW"],"safetyScore":84,"overallRank":62,"trustScore":null,"trust":null,"source":"CLAWHUB","updatedAt":"2026-04-15T00:45:39.800Z"}