{"id":"cdfae7b1-2ebc-4273-bd88-b695b9a24cd3","entityType":"agent","slug":"crewai-stavyagour-ai-dqm-system","name":"AI_DQM_SYSTEM","canonicalUrl":"https://www.xpersona.co/agent/crewai-stavyagour-ai-dqm-system","canonicalPath":"/agent/crewai-stavyagour-ai-dqm-system","generatedAt":"2026-10-10T08:46:14.908Z","source":"GITHUB_REPOS","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-09T23:24:37.702Z","emptyReason":null},"description":"An AI-driven, multi-agent Data Quality Management (DQM) system using CrewAI, Groq, and Machine Learning to autonomously profile, validate, and cleanse datasets across 6 quality dimensions. AI-Driven Data Quality Management System A Python-based, multi-agent pipeline that leverages CrewAI, Groq LLMs, and Machine Learning to automatically profile, audit, and remediate enterprise datasets. Overview This system automates the traditionally manual process of Data Quality Management (DQM). By combining large language models (LLMs) with statistical machine learning algorithms, the pipeline assesses data across","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. Last updated 10/9/2026.","installCommand":null,"sourceUrl":"https://github.com/stavyagour/AI_DQM_SYSTEM","homepage":null,"primaryLinks":[{"label":"View Source","url":"https://github.com/stavyagour/AI_DQM_SYSTEM","kind":"source"}],"safetyScore":66,"overallRank":18,"popularityScore":0,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"An AI-driven, multi-agent Data Quality Management (DQM) system using CrewAI, Groq, and Machine Learning to autonomously profile, validate, and cleanse datasets "},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T23:24:37.702Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[{"label":"crewai","status":"self-declared"},{"label":"multi-agent","status":"self-declared"}],"verifiedCount":0,"selfDeclaredCount":3,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"},{"key":"crewai","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"},{"key":"multi-agent","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile capability:crewai|supported|profile capability:multi-agent|supported|profile"}},"adoption":{"evidence":{"source":"no-adoption-signals","verified":false,"confidence":"low","updatedAt":"2026-10-09T23:24:37.702Z","emptyReason":"No source adoption metrics were available."},"stars":0,"forks":0,"downloads":null,"packageName":null,"latestVersion":null,"tractionLabel":null},"release":{"evidence":{"source":"agent-index","verified":false,"confidence":"medium","updatedAt":"2026-10-09T23:24:37.696Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T23:24:37.702Z","lastCrawledAt":"2026-10-09T23:24:37.696Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-16T23:24:37.696Z","lastVerifiedAt":null,"highlights":[]},"execution":{"evidence":{"source":"GITHUB REPOS","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":null,"setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/crewai-stavyagour-ai-dqm-system/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/crewai-stavyagour-ai-dqm-system/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/crewai-stavyagour-ai-dqm-system/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/crewai-stavyagour-ai-dqm-system/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/crewai-stavyagour-ai-dqm-system/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/crewai-stavyagour-ai-dqm-system/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"GITHUB_REPOS","generatedAt":"2026-10-10T08:46:14.908Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/crewai-stavyagour-ai-dqm-system/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/crewai-stavyagour-ai-dqm-system/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/crewai-stavyagour-ai-dqm-system/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/crewai-stavyagour-ai-dqm-system/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"GITHUB REPOS","verified":false,"confidence":"high","updatedAt":"2026-10-09T23:24:37.702Z","emptyReason":null},"readme":"# AI-Driven Data Quality Management System\n\nA Python-based, multi-agent pipeline that leverages CrewAI, Groq LLMs, and Machine Learning to automatically profile, audit, and remediate enterprise datasets.\n\n## Overview\n\nThis system automates the traditionally manual process of Data Quality Management (DQM). By combining large language models (LLMs) with statistical machine learning algorithms, the pipeline assesses data across multiple dimensions and applies automated fixes to improve overall data health.\n\n## Key Features\n\n*   **Multi-Agent Architecture**: Utilizes CrewAI to orchestrate specialized agents (e.g., Data Completeness Specialist, ML Accuracy Specialist) to audit data.\n*   **6 Quality Dimensions**: Evaluates datasets based on Completeness, Validity, Uniqueness, Consistency, Accuracy, and Timeliness.\n*   **Machine Learning Anomaly Detection**: Integrates Scikit-Learn's `IsolationForest` to detect multidimensional statistical outliers that standard rule-based engines miss.\n*   **Automated Remediation Engine**: Automatically applies programmatic fixes such as median imputation for missing numeric values, categorization for missing text, and winsorization for statistical anomalies.\n*   **Business Impact Reporting**: Generates a detailed console report comparing baseline and post-remediation data quality scores, highlighting error reduction and manual effort saved.\n\n## Architecture & Technologies\n\n*   **Core Logic**: Python 3.x, Pandas, Numpy\n*   **Agent Orchestration**: CrewAI\n*   **LLM Provider**: Groq (using Llama-3-70b) via OpenAI-compatible endpoint\n*   **Machine Learning**: Scikit-Learn (Isolation Forest, StandardScaler)\n\n## Prerequisites\n\nTo run this pipeline, you need:\n1. Python 3.9 or higher installed on your system.\n2. A valid Groq API key for the LLM agents.\n\n## Installation\n\nInstall the required Python packages using the provided `DQM_requirements.txt` file:\n\n```bash\npip install -r DQM_requirements.txt\n```\n\n*Note: CrewAI will use its native OpenAI package to communicate with the Groq API via a custom base URL.*\n\n## Usage\n\n1. Set your Groq API key as an environment variable in your terminal:\n\nFor Windows (PowerShell):\n```powershell\n$env:GROQ_API_KEY=\"gsk_your_actual_api_key_here\"\n```\n\nFor Mac/Linux:\n```bash\nexport GROQ_API_KEY=\"gsk_your_actual_api_key_here\"\n```\n\n2. Run the main pipeline script:\n\n```bash\npython dqm_system.py\n```\n\n## Pipeline Execution Flow\n\n1. **Data Generation**: Generates a synthetic customer dataset with intentionally injected dirty data (nulls, outliers, bad formats).\n2. **Agent Profiling**: CrewAI agents run programmatic internal tools to assess data completeness and statistical accuracy.\n3. **Remediation**: The remediation engine steps in to repair identified critical and high-severity issues.\n4. **Reporting**: A final summary report is printed to standard output.","readmeExcerpt":"AI-Driven Data Quality Management System A Python-based, multi-agent pipeline that leverages CrewAI, Groq LLMs, and Machine Learning to automatically profile, audit, and remediate enterprise datasets. Overview This system automates the traditionally manual process of Data Quality Management (DQM). By combining large language models (LLMs) with statistical machine learning algorithms, the pipeline assesses data across","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"pip install -r DQM_requirements.txt"},{"language":"powershell","snippet":"$env:GROQ_API_KEY=\"gsk_your_actual_api_key_here\""},{"language":"bash","snippet":"export GROQ_API_KEY=\"gsk_your_actual_api_key_here\""},{"language":"bash","snippet":"python dqm_system.py"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[],"languages":["python"],"docsSourceLabel":"GITHUB REPOS","editorialOverview":"An AI-driven, multi-agent Data Quality Management (DQM) system using CrewAI, Groq, and Machine Learning to autonomously profile, validate, and cleanse datasets across 6 quality dimensions. AI-Driven Data Quality Management System A Python-based, multi-agent pipeline that leverages CrewAI, Groq LLMs, and Machine Learning to automatically profile, audit, and remediate enterprise datasets. Overview This system automates the traditionally manual process of Data Quality Management (DQM). By combining large language models (LLMs) with statistical machine learning algorithms, the pipeline assesses data across","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":396,"uniquenessScore":62,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T23:24:37.702Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T23:24:37.702Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T08:46:14.908Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/github_repos","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}