{"id":"f9c9ac4a-200f-4e64-8a4a-13fa738022c1","entityType":"agent","slug":"crewai-bechir23-agentic-ai-data-science-assistant","name":"Agentic-AI-Data-Science-Assistant-","canonicalUrl":"https://www.xpersona.co/agent/crewai-bechir23-agentic-ai-data-science-assistant","canonicalPath":"/agent/crewai-bechir23-agentic-ai-data-science-assistant","generatedAt":"2026-10-09T08:58:57.109Z","source":"GITHUB_OPENCLEW","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-05-13T06:46:26.724Z","emptyReason":null},"description":"Comparative study of two Agentic AI architectures for automated data science: hidden-tool agents vs transparent code-generating agents. Built with CrewAI, OpenAI GPT-4o, tested on Titanic & House Prices datasets. What This Project Is About During this practical work, I explored how AI agents can automate data analysis tasks. I built and tested two different approaches to see which one works better for real-world data science problems. Think of it as having virtual data science assistants that can handle everything from data exploration to model training and report writing. I used the famous Titanic dataset (predicting passeng","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. Last updated 5/13/2026.","installCommand":"git clone https://github.com/bechir23/Agentic-AI-Data-Science-Assistant-.git","sourceUrl":"https://github.com/bechir23/Agentic-AI-Data-Science-Assistant-","homepage":null,"primaryLinks":[{"label":"View Source","url":"https://github.com/bechir23/Agentic-AI-Data-Science-Assistant-","kind":"source"}],"safetyScore":66,"overallRank":18.2,"popularityScore":0,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Comparative study of two Agentic AI architectures for automated data science: hidden-tool agents vs transparent code-generating agents. Built with CrewAI, OpenA"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-05-13T06:46:26.724Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[{"label":"crewai","status":"self-declared"},{"label":"multi-agent","status":"self-declared"}],"verifiedCount":0,"selfDeclaredCount":3,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"},{"key":"crewai","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"},{"key":"multi-agent","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile capability:crewai|supported|profile capability:multi-agent|supported|profile"}},"adoption":{"evidence":{"source":"no-adoption-signals","verified":false,"confidence":"low","updatedAt":"2026-05-13T06:46:26.724Z","emptyReason":"No source adoption metrics were available."},"stars":0,"forks":0,"downloads":null,"packageName":null,"latestVersion":null,"tractionLabel":null},"release":{"evidence":{"source":"agent-index","verified":false,"confidence":"medium","updatedAt":"2026-05-13T06:46:26.724Z","emptyReason":null},"lastUpdatedAt":"2026-05-13T06:46:26.724Z","lastCrawledAt":"2026-05-13T06:46:26.724Z","lastIndexedAt":null,"nextCrawlAt":"2026-05-20T06:46:26.724Z","lastVerifiedAt":null,"highlights":[]},"execution":{"evidence":{"source":"GITHUB OPENCLEW","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"git clone https://github.com/bechir23/Agentic-AI-Data-Science-Assistant-.git","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/crewai-bechir23-agentic-ai-data-science-assistant/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/crewai-bechir23-agentic-ai-data-science-assistant/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/crewai-bechir23-agentic-ai-data-science-assistant/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/crewai-bechir23-agentic-ai-data-science-assistant/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/crewai-bechir23-agentic-ai-data-science-assistant/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/crewai-bechir23-agentic-ai-data-science-assistant/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"GITHUB_OPENCLEW","generatedAt":"2026-10-09T08:58:57.109Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/crewai-bechir23-agentic-ai-data-science-assistant/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/crewai-bechir23-agentic-ai-data-science-assistant/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/crewai-bechir23-agentic-ai-data-science-assistant/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/crewai-bechir23-agentic-ai-data-science-assistant/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"GITHUB OPENCLEW","verified":false,"confidence":"high","updatedAt":"2026-05-13T06:46:26.724Z","emptyReason":null},"readme":"\r\n## What This Project Is About\r\n\r\nDuring this practical work, I explored how AI agents can automate data analysis tasks. I built and tested two different approaches to see which one works better for real-world data science problems. Think of it as having virtual data science assistants that can handle everything from data exploration to model training and report writing.\r\n\r\nI used the famous Titanic dataset (predicting passenger survival) and a house pricing dataset to put both systems through their paces. The goal was simple: let the AI agents do the heavy lifting while I evaluate how well they perform and where they struggle.\r\n\r\n## The Two Systems I Tested\r\n\r\n### System 1: The Behind-the-Scenes Approach\r\n\r\nThis system works like a traditional pipeline with four specialized agents working in sequence. Each agent has specific tools at their disposal, but all the Python code runs in the background where you can't see it.\r\n\r\n**How it works:**\r\n- **Project Planner**: Analyzes the business problem and decides on an approach\r\n- **Data Analyst**: Explores the dataset using pandas (you get statistics, but don't see the actual code)\r\n- **Modeler**: Trains machine learning models with scikit-learn\r\n- **Report Writer**: Puts everything together into a nice LaTeX report\r\n\r\n**The good parts:**\r\n- Super easy to use - just run it and wait for results\r\n- Works autonomously without needing much intervention\r\n- Produces clean, professional reports\r\n\r\n**The not-so-good parts:**\r\n- You can't see what's happening under the hood\r\n- If something goes wrong, it's hard to debug\r\n- Sometimes it makes mistakes (like including ID columns in training) and you won't notice until you check the results carefully\r\n\r\n**Files to run:**\r\n```powershell\r\npython main_classification.py    # For Titanic survival prediction\r\npython main_regression.py        # For house price prediction\r\n```\r\n\r\n### System 2: The Transparent Code Generator\r\n\r\nThis one takes a completely different approach. Instead of hiding everything, it generates Python code that you can actually read, modify, and reuse. It's like having a coding buddy who writes the analysis for you.\r\n\r\n**How it works:**\r\n- **Code Planner**: Figures out what code needs to be written\r\n- **Code Generator**: Actually writes complete Python scripts\r\n- **Code Executor**: Runs the code and checks for errors\r\n- **Results Interpreter**: Explains what the results mean\r\n\r\n**The good parts:**\r\n- Total transparency - you see every line of code\r\n- Can fix itself when it hits errors (I watched it correct 4 mistakes autonomously!)\r\n- You can extract the code and use it for other projects\r\n- Great for learning because you see the methodology\r\n\r\n**The not-so-good parts:**\r\n- Uses more API tokens because it generates longer responses\r\n- Quality depends on how well you describe what you want\r\n- Takes longer to run because of the self-correction iterations\r\n\r\n**Files to run:**\r\n```powershell\r\npython main_code_interpreter.py classification    # For Titanic\r\npython main_code_interpreter.py regression        # For house prices\r\n```\r\n\r\n## What I Actually Discovered\r\n\r\n### The PassengerId Bug\r\n\r\nBoth systems initially made the same rookie mistake: they included the PassengerId column (just a number from 1 to 891) in the training features. This created fake correlations and inflated the accuracy scores. System 2 made it way easier to spot this bug because I could literally read the code line by line. With System 1, I had to dig through tool outputs to figure out what was happening.\r\n\r\n### Self-Correction (Self-healing)\r\n\r\nThe coolest thing I observed was System 2's ability to debug itself. During one test, it hit four errors in a row:\r\n1. Syntax error with a broken f-string\r\n2. Warning about escape sequences\r\n3. Tried to extract titles from the Name column... after already deleting it\r\n4. Finally figured out it needed to extract titles BEFORE dropping columns\r\n\r\nEach iteration consumed API tokens, but watching an AI agent reason through its mistakes and fix them was genuinely impressive. \r\n\r\n### API Limits and Costs\r\n\r\nI'm using OpenAI GPT-4o for this project, which doesn't have the strict rate limits that free services have. However, I did initially try Groq's free tier (100k tokens/day) and hit the limit pretty quickly - a single run with the self-correction iterations consumed about 40k tokens!\r\n\r\nFor production use or if you want to avoid API costs entirely, switching to Ollama with a local model would be the way to go. The code supports all these options through a simple config change in `.env`.\r\n\r\n## Quick Start Guide\r\n\r\n### Prerequisites\r\n\r\nYou'll need Python 3.12 and an OpenAI API key (I'm using GPT-4o for this project).\r\n\r\n### Setup\r\n\r\n```powershell\r\n# Clone and navigate to the project\r\ncd TP_Agentic_AI\r\n\r\n# Create virtual environment\r\npython -m venv .venv\r\n.\\.venv\\Scripts\\Activate.ps1\r\n\r\n# Install dependencies\r\npip install -r requirements.txt\r\n\r\n# Configure your API key\r\n# Edit .env and add your OPENAI_API_KEY\r\n# The project is configured with LLM_MODE=openai by default\r\n```\r\n\r\n### Run System 1 (Hidden Tools)\r\n\r\n```powershell\r\npython main_classification.py\r\n# Wait 3-5 minutes, generates outputs/titanic_report.tex\r\n```\r\n\r\n### Run System 2 (Visible Code)\r\n\r\n```powershell\r\npython main_code_interpreter.py classification\r\n# Takes longer (5-10 min) but shows all code generation\r\n# Generates outputs/titanic_code_report.tex\r\n```\r\n\r\n### Compile Reports to PDF\r\n\r\n```powershell\r\n# Using WSL with pdflatex installed\r\nwsl pdflatex -interaction=nonstopmode outputs/titanic_report.tex\r\n```\r\n## Project Structure\r\n\r\n```\r\nTP_Agentic_AI/\r\n├── agents.py                      # System 1 agents (4 agents with hidden tools)\r\n├── agents_code_interpreter.py     # System 2 agents (code generators)\r\n├── crew_setup.py                  # System 1 task definitions\r\n├── tools.py                       # Python execution tools for both systems\r\n├── llama_llm.py                   # LLM configuration (OpenAI/Groq/Ollama)\r\n├── main_classification.py         # System 1 entry point (Titanic)\r\n├── main_regression.py             # System 1 entry point (House Prices)\r\n├── main_code_interpreter.py       # System 2 entry point (both datasets)\r\n├── data/\r\n│   ├── titanic.csv               # Classification dataset (891 samples)\r\n│   └── house_prices.csv          # Regression dataset (20640 samples)\r\n├── outputs/\r\n│   ├── titanic_report.tex        # System 1 classification report\r\n│   └── titanic_code_report.tex   # System 2 classification report\r\n└── Analysis_Crew_Systems.pdf     # Comparative analysis\r\n```\r\n\r\n## LLM Configuration\r\n\r\nFor this project, I'm using **OpenAI GPT-4o** as the primary language model. The `.env` file is configured with:\r\n\r\n```\r\nLLM_MODE=openai\r\nOPENAI_API_KEY=your_key_here\r\n```\r\n\r\n### Alternative LLM Options\r\n\r\nThe system supports multiple LLM providers through `llama_llm.py`. You can switch by changing `LLM_MODE` in `.env`:\r\n\r\n**Option 1: OpenAI (Current Setup)**\r\n```bash\r\nLLM_MODE=openai\r\nOPENAI_API_KEY=sk-...\r\n```\r\n- Best quality and reliability\r\n- Costs money but generous rate limits\r\n- GPT-4o gives excellent results\r\n\r\n**Option 2: Groq (Free Alternative)**\r\n```bash\r\nLLM_MODE=groq\r\nGROQ_API_KEY=gsk_...\r\n```\r\n- Free tier with 100k tokens/day\r\n- Fast inference with Llama 3.3 70B\r\n- Hit rate limits during testing\r\n\r\n**Option 3: HuggingFace**\r\n```bash\r\nLLM_MODE=huggingface\r\nHUGGINGFACE_API_KEY=hf_...\r\n```\r\n- Access to Llama 3.3 70B Instruct\r\n- Free tier available\r\n- Good for experimentation\r\n\r\n**Option 4: Ollama (Local)**\r\n```bash\r\nLLM_MODE=ollama\r\n# No API key needed, runs on your machine\r\n```\r\n\r\n## Technologies Used\r\n\r\n- **CrewAI 1.6.1**: Multi-agent orchestration framework\r\n- **OpenAI GPT-4o**: Primary LLM used for testing both systems\r\n- **Python 3.12**: Core programming language\r\n- **Pandas & Scikit-learn**: Data analysis and machine learning\r\n- **LaTeX**: Professional report generation\r\n\r\n\r\n**Note:** The critical analysis (`Analysis_Crew_Systems.pdf`) contains a detailed comparison of both systems based on actual test results.\r\n\r\n\r\n","readmeExcerpt":"What This Project Is About During this practical work, I explored how AI agents can automate data analysis tasks. I built and tested two different approaches to see which one works better for real-world data science problems. Think of it as having virtual data science assistants that can handle everything from data exploration to model training and report writing. I used the famous Titanic dataset (predicting passeng","codeSnippets":[],"executableExamples":[],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[],"languages":["python"],"docsSourceLabel":"GITHUB OPENCLEW","editorialOverview":"Comparative study of two Agentic AI architectures for automated data science: hidden-tool agents vs transparent code-generating agents. Built with CrewAI, OpenAI GPT-4o, tested on Titanic & House Prices datasets. What This Project Is About During this practical work, I explored how AI agents can automate data analysis tasks. I built and tested two different approaches to see which one works better for real-world data science problems. Think of it as having virtual data science assistants that can handle everything from data exploration to model training and report writing. I used the famous Titanic dataset (predicting passeng","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":432,"uniquenessScore":64,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-05-13T06:46:26.724Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-05-13T06:46:26.724Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T08:58:57.109Z","emptyReason":null},"items":[{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-10T18:48:31.762Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/github_openclew","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}