{"id":"4ce98fa5-6c3b-4549-91f4-b88615a7c7a1","slug":"crewai-fabriciomazzarotto-netflix-analytics-pipeline","name":"netflix-analytics-pipeline","description":"End-to-end Netflix-like data pipeline (5M+ events) using PySpark, Delta Lake, Databricks and AI agents (CrewAI + Claude)","canonicalUrl":"https://www.xpersona.co/skill/crewai-fabriciomazzarotto-netflix-analytics-pipeline","sourceUrl":"https://github.com/fabriciomazzarotto/netflix-analytics-pipeline","homepage":null,"source":"GITHUB_OPENCLEW","vendor":{"slug":"fabriciomazzarotto","label":"Fabriciomazzarotto","url":"https://github.com/fabriciomazzarotto/netflix-analytics-pipeline"},"protocols":["OPENCLEW"],"capabilities":["crewai","multi-agent"],"trustScore":null,"trustConfidence":"unknown","artifactCount":0,"benchmarkCount":0,"lastRelease":null,"freshnessAt":"2026-05-18T06:45:10.220Z","freshnessLabel":"May 18, 2026","securityReviewed":true,"openapiReady":false,"stats":[{"label":"Trust score","value":"Unknown"},{"label":"Compatibility","value":"OpenClaw"},{"label":"Freshness","value":"May 18, 2026"},{"label":"Vendor","value":"Fabriciomazzarotto"},{"label":"Artifacts","value":"0"},{"label":"Benchmarks","value":"0"},{"label":"Last release","value":"Unpublished"}],"factsPreview":[{"factKey":"vendor","label":"Vendor","value":"Fabriciomazzarotto","category":"vendor","href":"https://github.com/fabriciomazzarotto/netflix-analytics-pipeline","sourceUrl":"https://github.com/fabriciomazzarotto/netflix-analytics-pipeline","sourceType":"profile","confidence":"medium","observedAt":"2026-05-18T06:45:10.220Z","isPublic":true,"metadata":{}},{"factKey":"protocols","label":"Protocol compatibility","value":"OpenClaw","category":"compatibility","href":"https://www.xpersona.co/api/v1/agents/crewai-fabriciomazzarotto-netflix-analytics-pipeline/contract","sourceUrl":"https://www.xpersona.co/api/v1/agents/crewai-fabriciomazzarotto-netflix-analytics-pipeline/contract","sourceType":"contract","confidence":"medium","observedAt":"2026-05-18T06:45:10.220Z","isPublic":true,"metadata":{}},{"factKey":"docs_crawl","label":"Crawlable docs","value":"6 indexed pages on the official domain","category":"integration","href":"https://github.com/login?return_to=https%3A%2F%2Fgithub.com%2Fopenclaw%2Fskills%2Ftree%2Fmain%2Fskills%2Fasleep123%2Fcaldav-calendar","sourceUrl":"https://github.com/login?return_to=https%3A%2F%2Fgithub.com%2Fopenclaw%2Fskills%2Ftree%2Fmain%2Fskills%2Fasleep123%2Fcaldav-calendar","sourceType":"search_document","confidence":"medium","observedAt":"2026-04-15T05:03:46.393Z","isPublic":true,"metadata":{}},{"factKey":"handshake_status","label":"Handshake status","value":"UNKNOWN","category":"security","href":"https://www.xpersona.co/api/v1/agents/crewai-fabriciomazzarotto-netflix-analytics-pipeline/trust","sourceUrl":"https://www.xpersona.co/api/v1/agents/crewai-fabriciomazzarotto-netflix-analytics-pipeline/trust","sourceType":"trust","confidence":"medium","observedAt":null,"isPublic":true,"metadata":{}}],"highlights":["Trust evidence available"],"agentCard":{"name":"netflix-analytics-pipeline","description":"End-to-end Netflix-like data pipeline (5M+ events) using PySpark, Delta Lake, Databricks and AI agents (CrewAI + Claude)","source":"GITHUB_OPENCLEW","sourceId":"crewai:1217237444","repository":"https://github.com/fabriciomazzarotto/netflix-analytics-pipeline","documentation":"https://www.xpersona.co/skill/crewai-fabriciomazzarotto-netflix-analytics-pipeline/agent/crewai-fabriciomazzarotto-netflix-analytics-pipeline","protocols":["OPENCLEW"],"capabilities":["crewai","multi-agent"],"languages":["python"],"install":{"command":"git clone https://github.com/fabriciomazzarotto/netflix-analytics-pipeline.git","ecosystem":"git"},"examples":[{"kind":"example","language":"text","snippet":"Dados Sintéticos (5M+ eventos)\n         │\n         ▼\n    ┌─────────┐\n    │  BRONZE │  Ingestão append-only com lineage (_ingested_at, _source_file, _batch_id)\n    │  Delta  │  Particionado por data_evento\n    └────┬────┘\n         │\n         ▼\n    ┌─────────┐\n    │  SILVER │  Sessionização via window functions · Validação · DLQ (rejeições)\n    │  Delta  │  MERGE idempotente · Quality Report · MAX_REJECTION_RATE=0.25\n    └────┬────┘\n         │\n         ▼\n    ┌─────────┐\n    │  GOLD   │  6 marts analíticos: engagement, conteúdo, LTV, churn, receita, binge\n    │ Parquet │  Broadcast joins · AQE · Window functions para rankings e churn\n    └────┬────┘\n         │\n         ▼\n    ┌─────────┐\n    │ DIAMOND │  4 marts executivos no Databricks SQL Warehouse\n    │   SQL   │  dm_executive_kpis · dm_content_top10 · dm_churn_cohort · dm_revenue_forecast\n    └────┬────┘\n         │\n         ▼\n    ┌─────────┐\n    │  CrewAI │  2 workflows multi-agente via Claude Sonnet\n    │ Agentes │  Negócio (5 agentes) · Engenharia (5 agentes)\n    └─────────┘"},{"kind":"example","language":"text","snippet":"netflix-pipeline/\n├── config/\n│   ├── settings.py          # Configurações por ambiente (dev/databricks)\n│   └── schemas.py           # Schemas Spark das camadas Bronze/Silver/Gold\n│\n├── data/\n│   ├── generate_data.py     # Gerador: 5M+ eventos com problemas de qualidade\n│   ├── dicionario_dados.json # Referência anti-alucinação para agentes CrewAI\n│   └── output/              # Artefatos Delta/Parquet gerados localmente\n│\n├── pipeline/\n│   ├── ingest.py            # Bronze: ingestão append-only com lineage\n│   ├── quality.py           # Silver: sessionização, validação e DLQ\n│   ├── transform.py         # Gold: 6 marts com métricas Netflix\n│   ├── delta_utils.py       # Utilitários Delta Lake (MERGE, vacuum, optimize)\n│   └── watermark.py         # Controle de watermark para ingestão idempotente\n│\n├── databricks/\n│   ├── client.py            # Cliente REST API Databricks\n│   ├── upload_gold.py       # Upload Gold → SQL Warehouse (INSERT/MERGE)\n│   ├── create_diamond_layer.py # Camada Diamond (4 marts executivos)\n│   ├── jobs.py              # Jobs API 2.1 (CRUD + polling)\n│   ├── setup_jobs.py        # Job diário às 06:00 BRT + notebook Diamond\n│   ├── setup_secrets.py     # Scope 'netflix-pipeline' + secrets\n│   ├── setup_monitoring.py  # Health check notebook + alertas por email\n│   ├── setup_dashboard.py   # Dashboard matplotlib 5 seções\n│   ├── setup_unity_catalog.py # Comments + tags em 10 tabelas\n│   ├── setup_all.py         # Orquestrador de setup completo\n│   └── run_full_pipeline.py # Pipeline completo local → Databricks\n│\n├── crew/\n│   ├── ferramentas.py       # 6 ferramentas analíticas (leitura Gold via pandas)\n│   ├── agentes.py           # 5 agentes de negócio\n│   ├── tarefas.py           # 5 tarefas do workflow de negócio\n│   ├── ferramentas_engenharia.py # 7 ferramentas técnicas (schemas, código, perfil)\n│   ├── agentes_engenharia.py    # 5 agentes de engenharia\n│   └── tarefas_engenharia.py   # 5 tarefas do workflow de engenharia\n│\n├── tests/\n│   ├─"}]}}