{"id":"88b8f5e5-de00-4fe5-812c-b17cde0127df","entityType":"agent","slug":"clawhub-skills-1kalin-afrexai-ml-engineering","name":"afrexai-ml-engineering","canonicalUrl":"https://www.xpersona.co/agent/clawhub-skills-1kalin-afrexai-ml-engineering","canonicalPath":"/agent/clawhub-skills-1kalin-afrexai-ml-engineering","generatedAt":"2026-10-09T19:56:08.927Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":null},"description":"ML & AI Engineering System ML & AI Engineering System Complete methodology for building, deploying, and operating production ML/AI systems — from experiment to scale. --- Phase 1: Problem Framing Before writing any code, define the ML problem precisely. ML Problem Brief ML vs Rules Decision | Signal | Use Rules | Use ML | |--------|-----------|--------| | Logic is explainable in <10 rules | ✅ | ❌ | | Pattern is too complex for humans | ❌ | ✅ |","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. Last updated 4/15/2026.","installCommand":"clawhub skill install skills:1kalin:afrexai-ml-engineering","sourceUrl":"https://github.com/openclaw/skills/tree/main/skills/1kalin/afrexai-ml-engineering","homepage":null,"primaryLinks":[{"label":"View on ClawHub","url":"https://github.com/openclaw/skills/tree/main/skills/1kalin/afrexai-ml-engineering","kind":"source"}],"safetyScore":84,"overallRank":62,"popularityScore":50,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"ML & AI Engineering System ML & AI Engineering System Complete methodology for building, deploying, and operating production ML/AI systems — from experiment to "},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[{"label":"update","status":"self-declared"}],"verifiedCount":0,"selfDeclaredCount":2,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"},{"key":"update","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile capability:update|supported|profile"}},"adoption":{"evidence":{"source":"no-adoption-signals","verified":false,"confidence":"low","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":"No source adoption metrics were available."},"stars":null,"forks":null,"downloads":null,"packageName":null,"latestVersion":null,"tractionLabel":null},"release":{"evidence":{"source":"agent-index","verified":false,"confidence":"medium","updatedAt":"2026-02-25T06:17:44.083Z","emptyReason":null},"lastUpdatedAt":"2026-04-15T00:45:39.800Z","lastCrawledAt":"2026-02-25T06:17:44.083Z","lastIndexedAt":null,"nextCrawlAt":"2026-02-26T06:17:44.083Z","lastVerifiedAt":null,"highlights":[]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install skills:1kalin:afrexai-ml-engineering","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-ml-engineering/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-ml-engineering/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-ml-engineering/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-ml-engineering/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-ml-engineering/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-ml-engineering/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-09T19:56:08.927Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-ml-engineering/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-ml-engineering/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-ml-engineering/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-skills-1kalin-afrexai-ml-engineering/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":null},"readme":"# ML & AI Engineering System\n\nComplete methodology for building, deploying, and operating production ML/AI systems — from experiment to scale.\n\n---\n\n## Phase 1: Problem Framing\n\nBefore writing any code, define the ML problem precisely.\n\n### ML Problem Brief\n\n```yaml\nproblem_brief:\n  business_objective: \"\"          # What business metric improves?\n  success_metric: \"\"              # Quantified target (e.g., \"reduce churn 15%\")\n  baseline: \"\"                    # Current performance without ML\n  ml_task_type: \"\"                # classification | regression | ranking | generation | clustering | anomaly_detection | recommendation\n  prediction_target: \"\"           # What exactly are we predicting?\n  prediction_consumer: \"\"         # Who/what uses the prediction? (API | dashboard | email | automated action)\n  latency_requirement: \"\"         # real-time (<100ms) | near-real-time (<1s) | batch (minutes-hours)\n  data_available: \"\"              # What data exists today?\n  data_gaps: \"\"                   # What's missing?\n  ethical_considerations: \"\"      # Bias risks, fairness requirements, privacy\n  kill_criteria:                  # When to abandon the ML approach\n    - \"Baseline heuristic achieves >90% of ML performance\"\n    - \"Data quality too poor after 2 weeks of cleaning\"\n    - \"Model can't beat random by >10% on holdout set\"\n```\n\n### ML vs Rules Decision\n\n| Signal | Use Rules | Use ML |\n|--------|-----------|--------|\n| Logic is explainable in <10 rules | ✅ | ❌ |\n| Pattern is too complex for humans | ❌ | ✅ |\n| Training data >1,000 labeled examples | — | ✅ |\n| Needs to adapt to new patterns | ❌ | ✅ |\n| Must be 100% auditable/deterministic | ✅ | ❌ |\n| Pattern changes faster than you can update rules | ❌ | ✅ |\n\n**Rule of thumb:** Start with rules/heuristics. Only add ML when rules fail to capture the pattern.\n\n---\n\n## Phase 2: Data Engineering for ML\n\n### Data Quality Assessment\n\nScore each data source (0-5 per dimension):\n\n| Dimension | 0 (Terrible) | 5 (Excellent) |\n|-----------|--------------|----------------|\n| **Completeness** | >50% missing | <1% missing |\n| **Accuracy** | Known errors, no validation | Validated against source of truth |\n| **Consistency** | Different formats, duplicates | Standardized, deduplicated |\n| **Timeliness** | Months stale | Real-time or daily refresh |\n| **Relevance** | Weak proxy for target | Direct signal for prediction |\n| **Volume** | <100 samples | >10,000 samples per class |\n\n**Minimum score to proceed:** 18/30. Below 18 → fix data first, don't build models.\n\n### Feature Engineering Patterns\n\n```yaml\nfeature_types:\n  numerical:\n    - raw_value           # Use as-is if normally distributed\n    - log_transform       # Right-skewed distributions (revenue, counts)\n    - standardize         # z-score for algorithms sensitive to scale (SVM, KNN, neural nets)\n    - bin_to_categorical  # When relationship is non-linear and data is limited\n  categorical:\n    - one_hot             # <20 categories, tree-based models handle natively\n    - target_encoding     # High-cardinality (>20 categories), use with K-fold to prevent leakage\n    - embedding           # Very high-cardinality (user IDs, product IDs) with deep learning\n  temporal:\n    - lag_features        # Value at t-1, t-7, t-30\n    - rolling_statistics  # Mean, std, min, max over windows\n    - time_since_event    # Days since last purchase, hours since login\n    - cyclical_encoding   # sin/cos for hour-of-day, day-of-week, month\n  text:\n    - tfidf               # Simple, interpretable, good baseline\n    - sentence_embeddings # semantic similarity, modern NLP\n    - llm_extraction      # Use LLM to extract structured fields from unstructured text\n  interaction:\n    - ratios              # Feature A / Feature B (e.g., clicks/impressions = CTR)\n    - differences         # Feature A - Feature B (e.g., price - competitor_price)\n    - polynomial          # A * B, A^2 (use sparingly, high-cardinality features)\n```\n\n### Feature Store Design\n\n```yaml\nfeature_store:\n  offline_store:         # For training — batch computed, stored in data warehouse\n    storage: \"BigQuery | Snowflake | S3+Parquet\"\n    compute: \"Spark | dbt | SQL\"\n    refresh: \"daily | hourly\"\n  online_store:          # For serving — low-latency lookups\n    storage: \"Redis | DynamoDB | Feast online\"\n    latency_target: \"<10ms p99\"\n    refresh: \"streaming | near-real-time\"\n  registry:              # Feature metadata\n    naming: \"{entity}_{feature_name}_{window}_{aggregation}\"  # e.g., user_purchase_count_30d_sum\n    documentation:\n      - description\n      - data_type\n      - source_table\n      - owner\n      - created_date\n      - known_issues\n```\n\n### Data Leakage Prevention Checklist\n\n- [ ] No future information in features (time-travel check)\n- [ ] Train/val/test split done BEFORE feature engineering\n- [ ] Target encoding uses only training fold statistics\n- [ ] No features derived from the target variable\n- [ ] Temporal splits for time-series (no random shuffle)\n- [ ] Holdout set created BEFORE any EDA\n- [ ] Duplicates removed BEFORE splitting (same entity not in train AND test)\n- [ ] Normalization/scaling fit on train, applied to val/test\n\n---\n\n## Phase 3: Experiment Management\n\n### Experiment Tracking Template\n\n```yaml\nexperiment:\n  id: \"EXP-{YYYY-MM-DD}-{NNN}\"\n  hypothesis: \"\"                 # \"Adding user tenure features will improve churn prediction AUC by >2%\"\n  dataset_version: \"\"            # Hash or version of training data\n  features_used: []              # List of feature names\n  model_type: \"\"                 # Algorithm name\n  hyperparameters: {}            # All hyperparams logged\n  training_time: \"\"              # Wall clock\n  metrics:\n    primary: {}                  # The one metric that matters\n    secondary: {}                # Supporting metrics\n  baseline_comparison: \"\"        # Delta vs baseline\n  verdict: \"promoted | archived | iterate\"\n  notes: \"\"\n  artifacts:\n    - model_path: \"\"\n    - notebook_path: \"\"\n    - confusion_matrix: \"\"\n```\n\n### Model Selection Guide\n\n| Task | Start With | Scale To | Avoid |\n|------|-----------|----------|-------|\n| Tabular classification | XGBoost/LightGBM | Neural nets only if >100K samples | Deep learning on <10K samples |\n| Tabular regression | XGBoost/LightGBM | CatBoost for high-cardinality cats | Linear regression without feature engineering |\n| Image classification | Fine-tune ResNet/EfficientNet | Vision Transformer if >100K images | Training from scratch |\n| Text classification | Fine-tune BERT/RoBERTa | LLM few-shot if labeled data scarce | Bag-of-words for nuanced tasks |\n| Text generation | GPT-4/Claude API | Fine-tuned Llama/Mistral for cost | Training from scratch |\n| Time series | Prophet/ARIMA baseline → LightGBM | Temporal Fusion Transformer | LSTM without strong reason |\n| Recommendation | Collaborative filtering baseline | Two-tower neural | Complex models on <1K users |\n| Anomaly detection | Isolation Forest | Autoencoder if high-dimensional | Supervised methods without labeled anomalies |\n| Search/ranking | BM25 baseline → Learning to Rank | Cross-encoder reranking | Pure keyword without semantic |\n\n### Hyperparameter Tuning Strategy\n\n1. **Manual first** — understand 3-5 most impactful parameters\n2. **Bayesian optimization** (Optuna) — 50-100 trials for production models\n3. **Grid search** — only for final fine-tuning of 2-3 parameters\n4. **Random search** — better than grid for >4 parameters\n\n**Key hyperparameters by model:**\n\n| Model | Critical Params | Typical Range |\n|-------|----------------|---------------|\n| XGBoost | learning_rate, max_depth, n_estimators, min_child_weight | 0.01-0.3, 3-10, 100-1000, 1-10 |\n| LightGBM | learning_rate, num_leaves, feature_fraction, min_data_in_leaf | 0.01-0.3, 15-255, 0.5-1.0, 5-100 |\n| Neural Net | learning_rate, batch_size, hidden_dims, dropout | 1e-5 to 1e-2, 32-512, arch-dependent, 0.1-0.5 |\n| Random Forest | n_estimators, max_depth, min_samples_leaf | 100-1000, 5-30, 1-20 |\n\n---\n\n## Phase 4: Model Evaluation\n\n### Metric Selection by Task\n\n| Task | Primary Metric | When to Use | Watch Out For |\n|------|---------------|-------------|---------------|\n| Binary classification (balanced) | F1-score | Equal importance of precision/recall | — |\n| Binary classification (imbalanced) | PR-AUC | Rare positive class (<5%) | ROC-AUC hides poor performance on minority |\n| Multi-class | Macro F1 | All classes equally important | Micro F1 if class frequency = importance |\n| Regression | MAE | Outliers should not dominate | RMSE penalizes large errors more |\n| Ranking | NDCG@K | Top-K results matter most | MAP if binary relevance |\n| Generation | Human eval + automated | Quality is subjective | BLEU/ROUGE alone are insufficient |\n| Anomaly detection | Precision@K | False positives are expensive | Recall if missing anomalies is dangerous |\n\n### Evaluation Rigor Checklist\n\n- [ ] Metrics computed on TRUE holdout (never seen during training OR tuning)\n- [ ] Cross-validation for small datasets (<10K samples)\n- [ ] Stratified splits for imbalanced classes\n- [ ] Temporal split for time-dependent data\n- [ ] Confidence intervals reported (bootstrap or cross-val)\n- [ ] Performance broken down by important segments (geography, user cohort, etc.)\n- [ ] Fairness metrics across protected groups\n- [ ] Comparison against simple baseline (majority class, mean prediction, rules)\n- [ ] Error analysis: examined top 50 worst predictions manually\n- [ ] Calibration plot for probabilistic predictions\n\n### Offline-to-Online Gap Analysis\n\nBefore deploying, verify these don't cause train-serving skew:\n\n| Check | Offline | Online | Action |\n|-------|---------|--------|--------|\n| Feature computation | Batch SQL | Real-time API | Verify same logic, test with replay |\n| Data freshness | Point-in-time snapshot | Latest value | Document acceptable staleness |\n| Missing values | Imputed in pipeline | May be truly missing | Handle gracefully in serving |\n| Feature distributions | Training period | Current period | Monitor drift post-deploy |\n\n---\n\n## Phase 5: Model Deployment\n\n### Deployment Pattern Decision Tree\n\n```\nIs latency < 100ms required?\n├── Yes → Is model < 500MB?\n│   ├── Yes → Embedded serving (FastAPI + model in memory)\n│   └── No → Model server (Triton, TorchServe, vLLM)\n└── No → Is it a batch prediction?\n    ├── Yes → Batch pipeline (Spark, Airflow + offline inference)\n    └── No → Async queue (Celery/SQS → worker → result store)\n```\n\n### Production Serving Checklist\n\n```yaml\nserving_config:\n  model:\n    format: \"\"                    # ONNX | TorchScript | SavedModel | safetensors\n    version: \"\"                   # Semantic version\n    size_mb: null\n    load_time_seconds: null\n  infrastructure:\n    compute: \"\"                   # CPU | GPU (T4/A10/A100/H100)\n    instances: null               # Min/max for autoscaling\n    autoscale_metric: \"\"          # RPS | latency_p99 | GPU_utilization\n    autoscale_target: null\n  api:\n    endpoint: \"\"\n    input_schema: {}              # Pydantic model or JSON schema\n    output_schema: {}\n    timeout_ms: null\n    rate_limit: null\n  reliability:\n    health_check: \"/health\"\n    readiness_check: \"/ready\"     # Model loaded and warm\n    graceful_shutdown: true\n    circuit_breaker: true\n    fallback: \"\"                  # Rules-based fallback when model is down\n```\n\n### Containerization Template\n\n```dockerfile\n# Multi-stage build for minimal image\nFROM python:3.11-slim AS builder\nWORKDIR /app\nCOPY requirements.txt .\nRUN pip install --no-cache-dir -r requirements.txt\n\nFROM python:3.11-slim\nWORKDIR /app\nCOPY --from=builder /usr/local/lib/python3.11/site-packages /usr/local/lib/python3.11/site-packages\nCOPY --from=builder /usr/local/bin /usr/local/bin\nCOPY model/ ./model/\nCOPY src/ ./src/\n\n# Non-root user\nRUN useradd -m appuser && chown -R appuser /app\nUSER appuser\n\n# Health check\nHEALTHCHECK --interval=30s --timeout=5s CMD curl -f http://localhost:8080/health || exit 1\n\nEXPOSE 8080\nCMD [\"uvicorn\", \"src.serve:app\", \"--host\", \"0.0.0.0\", \"--port\", \"8080\"]\n```\n\n### A/B Testing for Models\n\n```yaml\nab_test:\n  name: \"\"\n  hypothesis: \"\"\n  primary_metric: \"\"              # Business metric (revenue, engagement, etc.)\n  guardrail_metrics: []           # Metrics that must NOT degrade\n  traffic_split:\n    control: 50                   # Current model\n    treatment: 50                 # New model\n  minimum_sample_size: null       # Power analysis: use statsmodels or online calculator\n  minimum_runtime_days: null      # At least 1 full business cycle (7 days min)\n  decision_criteria:\n    ship: \"Treatment > control by >X% with p<0.05 AND no guardrail regression\"\n    iterate: \"Promising signal but not significant — extend test or refine model\"\n    kill: \"No improvement after 2x minimum runtime OR guardrail breach\"\n```\n\n---\n\n## Phase 6: LLM Engineering\n\n### LLM Application Architecture\n\n```\n┌─────────────────────────────────────────────┐\n│              Application Layer               │\n│  (Prompt templates, chains, output parsers)  │\n├─────────────────────────────────────────────┤\n│              Orchestration Layer              │\n│  (Routing, fallback, retry, caching)         │\n├─────────────────────────────────────────────┤\n│              Model Layer                     │\n│  (API calls, fine-tuned models, embeddings)  │\n├─────────────────────────────────────────────┤\n│              Data Layer                      │\n│  (Vector store, context retrieval, memory)   │\n└─────────────────────────────────────────────┘\n```\n\n### Model Selection for LLM Tasks\n\n| Task | Best Option | Cost-Effective Option | When to Fine-Tune |\n|------|------------|----------------------|-------------------|\n| General reasoning | Claude Opus / GPT-4o | Claude Sonnet / GPT-4o-mini | Never for general reasoning |\n| Classification | Fine-tuned small model | Few-shot with Sonnet | >1,000 labeled examples + high volume |\n| Extraction | Structured output API | Regex + LLM fallback | Consistent format needed at scale |\n| Summarization | Claude Sonnet | GPT-4o-mini | Domain-specific style needed |\n| Code generation | Claude Sonnet | Codestral / DeepSeek | Internal codebase conventions |\n| Embeddings | text-embedding-3-large | text-embedding-3-small | Domain-specific vocab (medical, legal) |\n\n### RAG System Architecture\n\n```yaml\nrag_pipeline:\n  ingestion:\n    chunking:\n      strategy: \"semantic\"         # semantic | fixed_size | recursive\n      chunk_size: 512              # tokens (512-1024 for most use cases)\n      overlap: 50                  # tokens overlap between chunks\n      metadata_to_preserve:\n        - source_document\n        - page_number\n        - section_heading\n        - date_created\n    embedding:\n      model: \"text-embedding-3-large\"\n      dimensions: 1536             # Or 256/512 with Matryoshka for cost savings\n    vector_store: \"Pinecone | Weaviate | pgvector | Qdrant\"\n  retrieval:\n    strategy: \"hybrid\"             # dense | sparse | hybrid (recommended)\n    top_k: 10                      # Retrieve more, then rerank\n    reranking:\n      model: \"Cohere rerank | cross-encoder\"\n      top_n: 3                     # Final context chunks\n    filters: []                    # Metadata filters (date range, source, etc.)\n  generation:\n    model: \"\"\n    system_prompt: |\n      Answer based ONLY on the provided context.\n      If the context doesn't contain the answer, say \"I don't have enough information.\"\n      Cite sources using [Source: document_name, page X].\n    temperature: 0.1               # Low for factual, higher for creative\n    max_tokens: null\n```\n\n### RAG Quality Checklist\n\n- [ ] Chunking preserves semantic meaning (not cutting mid-sentence)\n- [ ] Metadata enables filtering (dates, sources, categories)\n- [ ] Retrieval returns relevant chunks (test with 50+ queries manually)\n- [ ] Reranking improves precision (compare with/without)\n- [ ] System prompt prevents hallucination (tested with adversarial queries)\n- [ ] Sources are cited and verifiable\n- [ ] Handles \"I don't know\" gracefully\n- [ ] Latency acceptable (<3s for interactive, <30s for complex)\n- [ ] Cost per query tracked and within budget\n\n### LLM Cost Optimization\n\n| Strategy | Savings | Trade-off |\n|----------|---------|-----------|\n| Prompt caching | 50-90% on repeated prefixes | Requires cache-friendly prompt design |\n| Model routing (small → large) | 40-70% | Slightly higher latency, need router logic |\n| Batch API | 50% | Hours of delay, batch-only workloads |\n| Shorter prompts | Linear with token reduction | May reduce quality |\n| Fine-tuned small model | 80-95% vs large model API | Training cost + maintenance |\n| Semantic caching | 50-80% for similar queries | May return stale/wrong cached result |\n| Output token limits | Proportional | May truncate useful information |\n\n---\n\n## Phase 7: Model Monitoring\n\n### Monitoring Dashboard\n\n```yaml\nmonitoring:\n  model_performance:\n    metrics:\n      - name: \"primary_metric\"         # Same as offline evaluation\n        threshold: null                 # Alert if below\n        window: \"1h | 1d | 7d\"\n      - name: \"prediction_distribution\"\n        alert: \"KL divergence > 0.1 from training distribution\"\n    latency:\n      p50_ms: null\n      p95_ms: null\n      p99_ms: null\n      alert_threshold_ms: null\n    throughput:\n      requests_per_second: null\n      error_rate_threshold: 0.01       # Alert if >1% errors\n  data_drift:\n    method: \"PSI | KS-test | JS-divergence\"\n    features_to_monitor: []            # Top 10 most important features\n    check_frequency: \"hourly | daily\"\n    alert_threshold: null              # PSI > 0.2 = significant drift\n  concept_drift:\n    method: \"performance_degradation\"\n    ground_truth_delay: \"\"             # How long until we get labels?\n    proxy_metrics: []                  # Metrics available before ground truth\n    retraining_trigger: \"\"             # When to retrain\n```\n\n### Drift Response Playbook\n\n| Drift Type | Detection | Severity | Response |\n|------------|-----------|----------|----------|\n| **Feature drift** (input distribution shifts) | PSI > 0.1 | Warning | Investigate cause, monitor performance |\n| **Feature drift** (PSI > 0.25) | PSI > 0.25 | Critical | Retrain on recent data within 24h |\n| **Concept drift** (relationship changes) | Performance drop >5% | Critical | Retrain with new labels, review features |\n| **Label drift** (target distribution changes) | Chi-square test | Warning | Verify label quality, check for data issues |\n| **Prediction drift** (output distribution shifts) | KL divergence | Warning | May indicate upstream data issue |\n\n### Automated Retraining Pipeline\n\n```yaml\nretraining:\n  triggers:\n    - type: \"scheduled\"\n      frequency: \"weekly | monthly\"\n    - type: \"performance\"\n      condition: \"primary_metric < threshold for 24h\"\n    - type: \"drift\"\n      condition: \"PSI > 0.2 on any top-10 feature\"\n  pipeline:\n    1_data_validation:\n      - check_completeness\n      - check_distribution_shift\n      - check_label_quality\n    2_training:\n      - use_latest_N_months_data\n      - same_hyperparameters_as_production   # Unless scheduled tuning\n      - log_all_metrics\n    3_evaluation:\n      - compare_vs_production_model\n      - must_beat_production_on_primary_metric\n      - must_not_regress_on_guardrail_metrics\n      - evaluate_on_golden_test_set\n    4_deployment:\n      - canary_deployment: 5%\n      - monitor_for: \"4h minimum\"\n      - auto_rollback_if: \"error_rate > 2x baseline\"\n      - gradual_rollout: \"5% → 25% → 50% → 100%\"\n    5_notification:\n      - log_retraining_event\n      - notify_team_on_failure\n      - update_model_registry\n```\n\n---\n\n## Phase 8: MLOps Infrastructure\n\n### ML Platform Components\n\n| Component | Purpose | Tools |\n|-----------|---------|-------|\n| Experiment tracking | Log runs, compare results | MLflow, W&B, Neptune |\n| Feature store | Centralized feature management | Feast, Tecton, Hopsworks |\n| Model registry | Version, stage, approve models | MLflow Registry, SageMaker |\n| Pipeline orchestration | DAG-based ML workflows | Airflow, Prefect, Dagster, Kubeflow |\n| Model serving | Low-latency inference | Triton, TorchServe, vLLM, BentoML |\n| Monitoring | Drift, performance, data quality | Evidently, Whylogs, Great Expectations |\n| Vector store | Embedding storage for RAG | Pinecone, Weaviate, pgvector, Qdrant |\n| GPU management | Training and inference compute | K8s + GPU operator, RunPod, Modal |\n\n### CI/CD for ML\n\n```yaml\nml_cicd:\n  on_code_change:\n    - lint_and_type_check\n    - unit_tests (data transforms, feature logic)\n    - integration_tests (pipeline end-to-end on sample data)\n  on_data_change:\n    - data_validation (Great Expectations / custom)\n    - feature_pipeline_run\n    - smoke_test_predictions\n  on_model_change:\n    - full_evaluation_suite\n    - bias_and_fairness_check\n    - performance_regression_test\n    - model_size_and_latency_check\n    - security_scan (model file, dependencies)\n    - staging_deployment\n    - integration_test_in_staging\n    - approval_gate (manual for major versions)\n    - canary_deployment\n```\n\n### Model Registry Workflow\n\n```\n┌──────────────┐      ┌──────────────┐      ┌──────────────┐\n│  Development │ ───→ │   Staging    │ ───→ │  Production  │\n│              │      │              │      │              │\n│ - Experiment │      │ - Eval suite │      │ - Canary     │\n│ - Log metrics│      │ - Load test  │      │ - Monitor    │\n│ - Compare    │      │ - Approval   │      │ - Rollback   │\n└──────────────┘      └──────────────┘      └──────────────┘\n```\n\n**Promotion criteria:**\n- Dev → Staging: Beats current production on offline metrics\n- Staging → Production: Passes load test + integration test + human approval\n- Auto-rollback: Error rate >2x OR latency >2x OR primary metric drops >5%\n\n---\n\n## Phase 9: Responsible AI\n\n### Bias Detection Checklist\n\n- [ ] Training data represents all demographic groups proportionally\n- [ ] Performance metrics broken down by protected attributes\n- [ ] Equal opportunity: similar true positive rates across groups\n- [ ] Calibration: predicted probabilities match actual rates per group\n- [ ] No proxy features for protected attributes (ZIP code → race)\n- [ ] Fairness metric selected and threshold defined BEFORE training\n- [ ] Disparate impact ratio >0.8 (80% rule)\n- [ ] Edge cases tested: what happens with unusual inputs?\n\n### Model Card Template\n\n```yaml\nmodel_card:\n  model_name: \"\"\n  version: \"\"\n  date: \"\"\n  owner: \"\"\n  description: \"\"\n  intended_use: \"\"\n  out_of_scope_uses: \"\"\n  training_data:\n    source: \"\"\n    size: \"\"\n    date_range: \"\"\n    known_biases: \"\"\n  evaluation:\n    metrics: {}\n    datasets: []\n    sliced_metrics: {}             # Performance by subgroup\n  limitations: []\n  ethical_considerations: []\n  maintenance:\n    retraining_schedule: \"\"\n    monitoring: \"\"\n    contact: \"\"\n```\n\n---\n\n## Phase 10: Cost & Performance Optimization\n\n### GPU Selection Guide\n\n| Use Case | GPU | VRAM | Cost/hr (cloud) | Best For |\n|----------|-----|------|-----------------|----------|\n| Fine-tune 7B model | A10G | 24GB | ~$1 | LoRA/QLoRA fine-tuning |\n| Fine-tune 70B model | A100 80GB | 80GB | ~$4 | Full fine-tuning medium models |\n| Serve 7B model | T4 | 16GB | ~$0.50 | Inference at scale |\n| Serve 70B model | A100 40GB | 40GB | ~$2 | Large model inference |\n| Train from scratch | H100 | 80GB | ~$8 | Pre-training, large-scale training |\n\n### Inference Optimization Techniques\n\n| Technique | Speedup | Quality Impact | Complexity |\n|-----------|---------|---------------|------------|\n| Quantization (INT8) | 2-3x | <1% degradation | Low |\n| Quantization (INT4/GPTQ) | 3-4x | 1-3% degradation | Medium |\n| Batching | 2-10x throughput | None | Low |\n| KV-cache optimization | 20-40% memory savings | None | Medium |\n| Speculative decoding | 2-3x for LLMs | None (mathematically exact) | High |\n| Model distillation | 5-10x smaller model | 2-5% degradation | High |\n| ONNX Runtime | 1.5-3x | None | Low |\n| TensorRT | 2-5x | <1% | Medium |\n| vLLM (PagedAttention) | 2-4x throughput for LLMs | None | Low |\n\n### Cost Tracking Template\n\n```yaml\nml_costs:\n  training:\n    compute_cost_per_run: null\n    runs_per_month: null\n    data_storage_monthly: null\n    experiment_tracking: null\n  inference:\n    cost_per_1k_predictions: null\n    daily_volume: null\n    monthly_cost: null\n    cost_per_query_breakdown:\n      compute: null\n      model_api_calls: null\n      vector_db: null\n      data_transfer: null\n  optimization_targets:\n    cost_per_prediction: null      # Target\n    monthly_budget: null\n    cost_reduction_goal: \"\"\n```\n\n---\n\n## Phase 11: ML System Quality Rubric\n\nScore your ML system (0-100):\n\n| Dimension | Weight | 0-2 (Poor) | 3-4 (Good) | 5 (Excellent) |\n|-----------|--------|-----------|------------|----------------|\n| **Problem framing** | 15% | No clear business metric | Defined success metric | Kill criteria + baseline + ROI estimate |\n| **Data quality** | 15% | Ad-hoc data, no validation | Automated quality checks | Feature store + lineage + versioning |\n| **Experiment rigor** | 15% | No tracking, one-off notebooks | MLflow/W&B tracking | Reproducible pipelines + proper evaluation |\n| **Model performance** | 15% | Barely beats baseline | Significant improvement | Calibrated, fair, robust to edge cases |\n| **Deployment** | 10% | Manual deployment | CI/CD for models | Canary + auto-rollback + A/B testing |\n| **Monitoring** | 15% | No monitoring | Basic metrics dashboard | Drift detection + auto-retraining + alerts |\n| **Documentation** | 5% | Nothing documented | Model card exists | Full model card + runbooks + decision log |\n| **Cost efficiency** | 10% | No cost tracking | Budget exists | Optimized inference + cost-per-prediction tracking |\n\n**Scoring:**\n- 80-100: Production-grade ML system\n- 60-79: Good foundations, missing operational maturity\n- 40-59: Prototype quality, not ready for production\n- <40: Science project, needs fundamental rework\n\n---\n\n## Common Mistakes\n\n| Mistake | Fix |\n|---------|-----|\n| Optimizing model before fixing data | Data quality > model complexity. Always. |\n| Using accuracy on imbalanced data | Use PR-AUC, F1, or domain-specific metric |\n| No baseline comparison | Always start with simple heuristic baseline |\n| Training on future data | Temporal splits for time-series, strict leakage checks |\n| Deploying without monitoring | No model in production without drift detection |\n| Fine-tuning when prompting works | Try few-shot prompting first — fine-tune only for scale/cost |\n| GPU for everything | CPU inference is often sufficient and 10x cheaper |\n| Ignoring calibration | If probabilities matter (risk scoring), calibrate |\n| One-time model deployment | ML is a continuous system — plan for retraining from day 1 |\n| Premature scaling | Prove value with batch predictions before building real-time serving |\n\n---\n\n## Quick Commands\n\n- \"Frame ML problem\" → Phase 1 brief\n- \"Assess data quality\" → Phase 2 scoring\n- \"Select model\" → Phase 3 guide\n- \"Evaluate model\" → Phase 4 checklist\n- \"Deploy model\" → Phase 5 serving config\n- \"Build RAG\" → Phase 6 RAG architecture\n- \"Set up monitoring\" → Phase 7 dashboard\n- \"Optimize costs\" → Phase 10 tracking\n- \"Score ML system\" → Phase 11 rubric\n- \"Detect drift\" → Phase 7 playbook\n- \"A/B test model\" → Phase 5 template\n- \"Create model card\" → Phase 9 template\n","readmeExcerpt":"ML & AI Engineering System Complete methodology for building, deploying, and operating production ML/AI systems — from experiment to scale. --- Phase 1: Problem Framing Before writing any code, define the ML problem precisely. ML Problem Brief ML vs Rules Decision | Signal | Use Rules | Use ML | |--------|-----------|--------| | Logic is explainable in <10 rules | ✅ | ❌ | | Pattern is too complex for humans | ❌ | ✅ |","codeSnippets":[],"executableExamples":[{"language":"yaml","snippet":"problem_brief:\n  business_objective: \"\"          # What business metric improves?\n  success_metric: \"\"              # Quantified target (e.g., \"reduce churn 15%\")\n  baseline: \"\"                    # Current performance without ML\n  ml_task_type: \"\"                # classification | regression | ranking | generation | clustering | anomaly_detection | recommendation\n  prediction_target: \"\"           # What exactly are we predicting?\n  prediction_consumer: \"\"         # Who/what uses the prediction? (API | dashboard | email | automated action)\n  latency_requirement: \"\"         # real-time (<100ms) | near-real-time (<1s) | batch (minutes-hours)\n  data_available: \"\"              # What data exists today?\n  data_gaps: \"\"                   # What's missing?\n  ethical_considerations: \"\"      # Bias risks, fairness requirements, privacy\n  kill_criteria:                  # When to abandon the ML approach\n    - \"Baseline heuristic achieves >90% of ML performance\"\n    - \"Data quality too poor after 2 weeks of cleaning\"\n    - \"Model can't beat random by >10% on holdout set\""},{"language":"yaml","snippet":"feature_types:\n  numerical:\n    - raw_value           # Use as-is if normally distributed\n    - log_transform       # Right-skewed distributions (revenue, counts)\n    - standardize         # z-score for algorithms sensitive to scale (SVM, KNN, neural nets)\n    - bin_to_categorical  # When relationship is non-linear and data is limited\n  categorical:\n    - one_hot             # <20 categories, tree-based models handle natively\n    - target_encoding     # High-cardinality (>20 categories), use with K-fold to prevent leakage\n    - embedding           # Very high-cardinality (user IDs, product IDs) with deep learning\n  temporal:\n    - lag_features        # Value at t-1, t-7, t-30\n    - rolling_statistics  # Mean, std, min, max over windows\n    - time_since_event    # Days since last purchase, hours since login\n    - cyclical_encoding   # sin/cos for hour-of-day, day-of-week, month\n  text:\n    - tfidf               # Simple, interpretable, good baseline\n    - sentence_embeddings # semantic similarity, modern NLP\n    - llm_extraction      # Use LLM to extract structured fields from unstructured text\n  interaction:\n    - ratios              # Feature A / Feature B (e.g., clicks/impressions = CTR)\n    - differences         # Feature A - Feature B (e.g., price - competitor_price)\n    - polynomial          # A * B, A^2 (use sparingly, high-cardinality features)"},{"language":"yaml","snippet":"feature_store:\n  offline_store:         # For training — batch computed, stored in data warehouse\n    storage: \"BigQuery | Snowflake | S3+Parquet\"\n    compute: \"Spark | dbt | SQL\"\n    refresh: \"daily | hourly\"\n  online_store:          # For serving — low-latency lookups\n    storage: \"Redis | DynamoDB | Feast online\"\n    latency_target: \"<10ms p99\"\n    refresh: \"streaming | near-real-time\"\n  registry:              # Feature metadata\n    naming: \"{entity}_{feature_name}_{window}_{aggregation}\"  # e.g., user_purchase_count_30d_sum\n    documentation:\n      - description\n      - data_type\n      - source_table\n      - owner\n      - created_date\n      - known_issues"},{"language":"yaml","snippet":"experiment:\n  id: \"EXP-{YYYY-MM-DD}-{NNN}\"\n  hypothesis: \"\"                 # \"Adding user tenure features will improve churn prediction AUC by >2%\"\n  dataset_version: \"\"            # Hash or version of training data\n  features_used: []              # List of feature names\n  model_type: \"\"                 # Algorithm name\n  hyperparameters: {}            # All hyperparams logged\n  training_time: \"\"              # Wall clock\n  metrics:\n    primary: {}                  # The one metric that matters\n    secondary: {}                # Supporting metrics\n  baseline_comparison: \"\"        # Delta vs baseline\n  verdict: \"promoted | archived | iterate\"\n  notes: \"\"\n  artifacts:\n    - model_path: \"\"\n    - notebook_path: \"\"\n    - confusion_matrix: \"\""},{"language":"text","snippet":"Is latency < 100ms required?\n├── Yes → Is model < 500MB?\n│   ├── Yes → Embedded serving (FastAPI + model in memory)\n│   └── No → Model server (Triton, TorchServe, vLLM)\n└── No → Is it a batch prediction?\n    ├── Yes → Batch pipeline (Spark, Airflow + offline inference)\n    └── No → Async queue (Celery/SQS → worker → result store)"},{"language":"yaml","snippet":"serving_config:\n  model:\n    format: \"\"                    # ONNX | TorchScript | SavedModel | safetensors\n    version: \"\"                   # Semantic version\n    size_mb: null\n    load_time_seconds: null\n  infrastructure:\n    compute: \"\"                   # CPU | GPU (T4/A10/A100/H100)\n    instances: null               # Min/max for autoscaling\n    autoscale_metric: \"\"          # RPS | latency_p99 | GPU_utilization\n    autoscale_target: null\n  api:\n    endpoint: \"\"\n    input_schema: {}              # Pydantic model or JSON schema\n    output_schema: {}\n    timeout_ms: null\n    rate_limit: null\n  reliability:\n    health_check: \"/health\"\n    readiness_check: \"/ready\"     # Model loaded and warm\n    graceful_shutdown: true\n    circuit_breaker: true\n    fallback: \"\"                  # Rules-based fallback when model is down"}],"parameters":{},"dependencies":[],"permissions":[],"extractedFiles":[],"languages":["typescript"],"docsSourceLabel":"CLAWHUB","editorialOverview":"ML & AI Engineering System ML & AI Engineering System Complete methodology for building, deploying, and operating production ML/AI systems — from experiment to scale. --- Phase 1: Problem Framing Before writing any code, define the ML problem precisely. ML Problem Brief ML vs Rules Decision | Signal | Use Rules | Use ML | |--------|-----------|--------| | Logic is explainable in <10 rules | ✅ | ❌ | | Pattern is too complex for humans | ❌ | ✅ |","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":369,"uniquenessScore":66,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T19:56:08.927Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}