{"id":"0e38c6b4-1800-4265-ab89-0a36ac3f4faf","entityType":"agent","slug":"crewai-mahammadriyazshek-crewai-llm-evaluation-framework","name":"crewai-llm-evaluation-framework","canonicalUrl":"https://www.xpersona.co/agent/crewai-mahammadriyazshek-crewai-llm-evaluation-framework","canonicalPath":"/agent/crewai-mahammadriyazshek-crewai-llm-evaluation-framework","generatedAt":"2026-10-10T04:38:30.543Z","source":"GITHUB_REPOS","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-09T20:22:13.632Z","emptyReason":null},"description":"Custom CrewAI multi-agent framework that automates LLM response quality evaluation against rubrics — replacing manual human review with a scalable system achieving 79% accuracy. Topics: crewai llm-evaluation multi-agent azure-openai langchain 🤖 CrewAI Evaluation Framework Custom CrewAI multi-agent system + automated evaluation pipeline that assesses LLM response quality against rubrics — replacing manual human evaluation with a scalable automated system and achieving **79% accuracy** in model assessment. 🚀 Live Demo Deploy the static frontend to **GitHub Pages**, or run the FastAPI backend for full CrewAI evaluation. ✨ Features - 🤖 **4 specialist CrewA","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. Last updated 10/9/2026.","installCommand":null,"sourceUrl":"https://github.com/MahammadRiyazShek/crewai-llm-evaluation-framework","homepage":null,"primaryLinks":[{"label":"View Source","url":"https://github.com/MahammadRiyazShek/crewai-llm-evaluation-framework","kind":"source"}],"safetyScore":66,"overallRank":18.2,"popularityScore":0,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Custom CrewAI multi-agent framework that automates LLM response quality evaluation against rubrics — replacing manual human review with a scalable system achiev"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T20:22:13.632Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[{"label":"crewai","status":"self-declared"},{"label":"multi-agent","status":"self-declared"}],"verifiedCount":0,"selfDeclaredCount":3,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"},{"key":"crewai","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"},{"key":"multi-agent","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile capability:crewai|supported|profile capability:multi-agent|supported|profile"}},"adoption":{"evidence":{"source":"no-adoption-signals","verified":false,"confidence":"low","updatedAt":"2026-10-09T20:22:13.632Z","emptyReason":"No source adoption metrics were available."},"stars":0,"forks":0,"downloads":null,"packageName":null,"latestVersion":null,"tractionLabel":null},"release":{"evidence":{"source":"agent-index","verified":false,"confidence":"medium","updatedAt":"2026-10-09T20:22:13.627Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T20:22:13.632Z","lastCrawledAt":"2026-10-09T20:22:13.627Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-16T20:22:13.627Z","lastVerifiedAt":null,"highlights":[]},"execution":{"evidence":{"source":"GITHUB REPOS","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":null,"setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/crewai-mahammadriyazshek-crewai-llm-evaluation-framework/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/crewai-mahammadriyazshek-crewai-llm-evaluation-framework/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/crewai-mahammadriyazshek-crewai-llm-evaluation-framework/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/crewai-mahammadriyazshek-crewai-llm-evaluation-framework/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/crewai-mahammadriyazshek-crewai-llm-evaluation-framework/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/crewai-mahammadriyazshek-crewai-llm-evaluation-framework/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"GITHUB_REPOS","generatedAt":"2026-10-10T04:38:30.543Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/crewai-mahammadriyazshek-crewai-llm-evaluation-framework/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/crewai-mahammadriyazshek-crewai-llm-evaluation-framework/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/crewai-mahammadriyazshek-crewai-llm-evaluation-framework/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/crewai-mahammadriyazshek-crewai-llm-evaluation-framework/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"GITHUB REPOS","verified":false,"confidence":"high","updatedAt":"2026-10-09T20:22:13.632Z","emptyReason":null},"readme":"# 🤖 CrewAI Evaluation Framework\n\nCustom CrewAI multi-agent system + automated evaluation pipeline that assesses LLM response quality against rubrics — replacing manual human evaluation with a scalable automated system and achieving **79% accuracy** in model assessment.\n\n![CrewAI](https://img.shields.io/badge/CrewAI-multi--agent-purple) ![Azure OpenAI](https://img.shields.io/badge/Azure-OpenAI-0078D4) ![LangChain](https://img.shields.io/badge/LangChain-0.1-orange)\n\n## 🚀 Live Demo\nDeploy the static frontend to **GitHub Pages**, or run the FastAPI backend for full CrewAI evaluation.\n\n## ✨ Features\n- 🤖 **4 specialist CrewAI agents** — Relevance, Factual Accuracy, Completeness, Clarity & Coherence\n- 📋 **6-dimension rubric** with weighted scoring\n- 📊 Interactive dashboard with per-dimension breakdown\n- 🎯 **79% agreement** with human evaluators\n- ⚡ **10× faster** than manual review\n- 📝 Detailed agent verdicts for every response\n\n## 🛠️ Tech Stack\n- **Multi-agent framework:** CrewAI\n- **LLM:** Azure OpenAI GPT-4\n- **Orchestration:** LangChain\n- **API:** FastAPI\n- **Frontend:** Vanilla JS\n\n## 📦 Files\n- `index.html`, `app.js`, `styles.css` — frontend demo\n- `crewai_eval.py` — production FastAPI backend with CrewAI agents\n- `requirements.txt` — Python deps\n- `README.md`\n\n## ⚙️ Run Locally\n**Frontend demo:** open `index.html` in browser.\n\n**Backend:**\n```bash\npip install -r requirements.txt\nexport AZURE_OPENAI_API_KEY=...\nexport AZURE_OPENAI_ENDPOINT=https://YOUR.openai.azure.com/\nexport CHAT_DEPLOYMENT=gpt-4\nuvicorn crewai_eval:app --reload\n```\n\n## 🌐 Deployment\n- **Frontend:** GitHub Pages / Netlify / Vercel\n- **Backend:** Render.com / Railway.app / Azure App Service\n\n## 📋 Rubric\n| Dimension | Weight |\n|-----------|--------|\n| Relevance | 20% |\n| Factual Accuracy | 25% |\n| Completeness | 20% |\n| Clarity | 15% |\n| Coherence | 10% |\n| Conciseness | 10% |\n\n## 📜 License\nMIT\n","readmeExcerpt":"🤖 CrewAI Evaluation Framework Custom CrewAI multi-agent system + automated evaluation pipeline that assesses LLM response quality against rubrics — replacing manual human evaluation with a scalable automated system and achieving **79% accuracy** in model assessment. 🚀 Live Demo Deploy the static frontend to **GitHub Pages**, or run the FastAPI backend for full CrewAI evaluation. ✨ Features - 🤖 **4 specialist CrewA","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"pip install -r requirements.txt\nexport AZURE_OPENAI_API_KEY=...\nexport AZURE_OPENAI_ENDPOINT=https://YOUR.openai.azure.com/\nexport CHAT_DEPLOYMENT=gpt-4\nuvicorn crewai_eval:app --reload"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[],"languages":["python"],"docsSourceLabel":"GITHUB REPOS","editorialOverview":"Custom CrewAI multi-agent framework that automates LLM response quality evaluation against rubrics — replacing manual human review with a scalable system achieving 79% accuracy. Topics: crewai llm-evaluation multi-agent azure-openai langchain 🤖 CrewAI Evaluation Framework Custom CrewAI multi-agent system + automated evaluation pipeline that assesses LLM response quality against rubrics — replacing manual human evaluation with a scalable automated system and achieving **79% accuracy** in model assessment. 🚀 Live Demo Deploy the static frontend to **GitHub Pages**, or run the FastAPI backend for full CrewAI evaluation. ✨ Features - 🤖 **4 specialist CrewA","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":411,"uniquenessScore":61,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T20:22:13.632Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T20:22:13.632Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T04:38:30.543Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/github_repos","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}