{"id":"1c67021b-12ce-4f12-a072-c25f221475b4","entityType":"agent","slug":"clawhub-ivangdavila-data-analysis-2","name":"Data Analysis","canonicalUrl":"https://www.xpersona.co/agent/clawhub-ivangdavila-data-analysis-2","canonicalPath":"/agent/clawhub-ivangdavila-data-analysis-2","generatedAt":"2026-10-09T12:13:42.606Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-05-11T06:25:23.421Z","emptyReason":null},"description":"Data analysis and visualization. Query databases, generate reports, automate spreadsheets, and turn raw data into clear, actionable insights. Use when (1) yo... Skill: Data Analysis Owner: ivangdavila Summary: Data analysis and visualization. Query databases, generate reports, automate spreadsheets, and turn raw data into clear, actionable insights. Use when (1) yo... Tags: latest:1.0.2 Version history: v1.0.2 | 2026-03-11T15:11:50.484Z | user Added metric contracts, chart guidance, and decision brief templates for more reliable analysis. v1.0.1 | 2026-03-11T14:45:27.317Z |","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 31.3K downloads reported by the source. Last updated 5/11/2026.","installCommand":"clawhub skill install s178jdk12x4qj3gs2se3etxf3h83h7ft:data-analysis","sourceUrl":"https://clawhub.ai/ivangdavila/data-analysis","homepage":"https://clawhub.ai/ivangdavila/data-analysis","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/ivangdavila/data-analysis","kind":"source"}],"safetyScore":84,"overallRank":62,"popularityScore":90,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Data analysis and visualization. Query databases, generate reports, automate spreadsheets, and turn raw data into clear, actionable insights. Use when (1) yo..."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-05-11T06:25:23.421Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-05-11T06:25:23.421Z","emptyReason":null},"stars":null,"forks":null,"downloads":31296,"packageName":null,"latestVersion":"1.0.2","tractionLabel":"31.3K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-05-11T06:25:23.298Z","emptyReason":null},"lastUpdatedAt":"2026-05-11T06:25:23.421Z","lastCrawledAt":"2026-05-11T06:25:23.298Z","lastIndexedAt":null,"nextCrawlAt":"2026-05-12T06:25:23.298Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.2","createdAt":"2026-03-11T15:11:50.484Z","changelog":"Added metric contracts, chart guidance, and decision brief templates for more reliable analysis.","fileCount":7,"zipByteSize":11148},{"version":"1.0.1","createdAt":"2026-03-11T14:45:27.317Z","changelog":"Clearer promise and activation cues for analysis tasks.","fileCount":4,"zipByteSize":6595},{"version":"1.0.0","createdAt":"2026-02-12T14:10:27.744Z","changelog":"Initial release","fileCount":4,"zipByteSize":6322}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s178jdk12x4qj3gs2se3etxf3h83h7ft:data-analysis","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-ivangdavila-data-analysis-2/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-ivangdavila-data-analysis-2/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-ivangdavila-data-analysis-2/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-ivangdavila-data-analysis-2/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-ivangdavila-data-analysis-2/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-ivangdavila-data-analysis-2/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-09T12:13:42.605Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-ivangdavila-data-analysis-2/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-ivangdavila-data-analysis-2/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-ivangdavila-data-analysis-2/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-ivangdavila-data-analysis-2/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-05-11T06:25:23.421Z","emptyReason":null},"readme":"Skill: Data Analysis\n\nOwner: ivangdavila\n\nSummary: Data analysis and visualization. Query databases, generate reports, automate spreadsheets, and turn raw data into clear, actionable insights. Use when (1) yo...\n\nTags: latest:1.0.2\n\nVersion history:\n\nv1.0.2 | 2026-03-11T15:11:50.484Z | user\n\nAdded metric contracts, chart guidance, and decision brief templates for more reliable analysis.\n\nv1.0.1 | 2026-03-11T14:45:27.317Z | user\n\nClearer promise and activation cues for analysis tasks.\n\nv1.0.0 | 2026-02-12T14:10:27.744Z | user\n\nInitial release\n\nArchive index:\n\nArchive v1.0.2: 7 files, 11148 bytes\n\nFiles: chart-selection.md (1756b), decision-briefs.md (1144b), metric-contracts.md (1580b), pitfalls.md (3961b), SKILL.md (7910b), techniques.md (4994b), _meta.json (132b)\n\nFile v1.0.2:SKILL.md\n\n---\nname: Data Analysis\nslug: data-analysis\nversion: 1.0.2\nhomepage: https://clawic.com/skills/data-analysis\ndescription: \"Data analysis and visualization. Query databases, generate reports, automate spreadsheets, and turn raw data into clear, actionable insights. Use when (1) you need to analyze, visualize, or explain data; (2) the user wants reports, dashboards, or metrics turned into a decision; (3) the work involves SQL, Python, spreadsheets, BI tools, or notebooks; (4) you need to compare segments, cohorts, funnels, experiments, or time periods; (5) the user explicitly installs or references the skill for the current task.\"\nchangelog: Added metric contracts, chart guidance, and decision brief templates for more reliable analysis.\nmetadata: {\"clawdbot\":{\"emoji\":\"D\",\"requires\":{\"bins\":[]},\"os\":[\"linux\",\"darwin\",\"win32\"]}}\n---\n\n## When to Use\n\nUse this skill when the user needs to analyze, explain, or visualize data from SQL, spreadsheets, notebooks, dashboards, exports, or ad hoc tables.\n\nUse it for KPI debugging, experiment readouts, funnel or cohort analysis, anomaly reviews, executive reporting, and quality checks on metrics or query logic.\n\nPrefer this skill over generic coding or spreadsheet help when the hard part is analytical judgment: metric definition, comparison design, interpretation, or recommendation.\n\nUser asks about: analyzing data, finding patterns, understanding metrics, testing hypotheses, cohort analysis, A/B testing, churn analysis, or statistical significance.\n\n## Core Principle\n\nAnalysis without a decision is just arithmetic. Always clarify: **What would change if this analysis shows X vs Y?**\n\n## Methodology First\n\nBefore touching data:\n1. **What decision** is this analysis supporting?\n2. **What would change your mind?** (the real question)\n3. **What data do you actually have** vs what you wish you had?\n4. **What timeframe** is relevant?\n\n## Statistical Rigor Checklist\n\n- [ ] Sample size sufficient? (small N = wide confidence intervals)\n- [ ] Comparison groups fair? (same time period, similar conditions)\n- [ ] Multiple comparisons? (20 tests = 1 \"significant\" by chance)\n- [ ] Effect size meaningful? (statistically significant != practically important)\n- [ ] Uncertainty quantified? (\"12-18% lift\" not just \"15% lift\")\n\n## Architecture\n\nThis skill does not require local folders, persistent memory, or setup state.\n\nUse the included reference files as lightweight guides:\n- `metric-contracts.md` for KPI definitions and caveats\n- `chart-selection.md` for visual choice and chart anti-patterns\n- `decision-briefs.md` for stakeholder-facing outputs\n- `pitfalls.md` and `techniques.md` for analytical rigor and method choice\n\n## Quick Reference\n\nLoad only the smallest relevant file to keep context focused.\n\n| Topic | File |\n|-------|------|\n| Metric definition contracts | `metric-contracts.md` |\n| Visual selection and chart anti-patterns | `chart-selection.md` |\n| Decision-ready output formats | `decision-briefs.md` |\n| Failure modes to catch early | `pitfalls.md` |\n| Method selection by question type | `techniques.md` |\n\n## Core Rules\n\n### 1. Start from the decision, not the dataset\n- Identify the decision owner, the question that could change a decision, and the deadline before doing analysis.\n- If no decision would change, reframe the request before computing anything.\n\n### 2. Lock the metric contract before calculating\n- Define entity, grain, numerator, denominator, time window, timezone, filters, exclusions, and source of truth.\n- If any of those are ambiguous, state the ambiguity explicitly before presenting results.\n\n### 3. Separate extraction, transformation, and interpretation\n- Keep query logic, cleanup assumptions, and analytical conclusions distinguishable.\n- Never hide business assumptions inside SQL, formulas, or notebook code without naming them in the write-up.\n\n### 4. Choose visuals to answer a question\n- Select charts based on the analytical question: trend, comparison, distribution, relationship, composition, funnel, or cohort retention.\n- Do not add charts that make the deck look fuller but do not change the decision.\n\n### 5. Brief every result in decision format\n- Every output should include the answer, evidence, confidence, caveats, and recommended next action.\n- If the output is going to a stakeholder, translate the method into business implications instead of leading with technical detail.\n\n### 6. Stress-test claims before recommending action\n- Segment by obvious confounders, compare the right baseline, quantify uncertainty, and check sensitivity to exclusions or time windows.\n- Strong-looking numbers without robustness checks are not decision-ready.\n\n### 7. Escalate when the data cannot support the claim\n- Block or downgrade conclusions when sample size is weak, the source is unreliable, definitions drifted, or confounding is unresolved.\n- It is better to say \"unknown yet\" than to produce false confidence.\n\n## Common Traps\n\n- Reusing a KPI name after changing numerator, denominator, or exclusions -> trend comparisons become invalid.\n- Comparing daily, weekly, and monthly grains in one chart -> movement looks real but is mostly aggregation noise.\n- Showing percentages without underlying counts -> leadership overreacts to tiny denominators.\n- Using a pretty chart instead of the right chart -> the output looks polished but hides the actual decision signal.\n- Hunting for interesting cuts after seeing the result -> narrative follows chance instead of evidence.\n- Shipping automated reports without metric owners or caveats -> bad numbers spread faster than they can be corrected.\n- Treating observational patterns as causal proof -> action plans get built on correlation alone.\n\n## Approach Selection\n\n| Question type | Approach | Key output |\n|---------------|----------|------------|\n| \"Is X different from Y?\" | Hypothesis test | p-value + effect size + CI |\n| \"What predicts Z?\" | Regression/correlation | Coefficients + R² + residual check |\n| \"How do users behave over time?\" | Cohort analysis | Retention curves by cohort |\n| \"Are these groups different?\" | Segmentation | Profiles + statistical comparison |\n| \"What's unusual?\" | Anomaly detection | Flagged points + context |\n\nFor technique details and when to use each, see `techniques.md`.\n\n## Output Standards\n\n1. **Lead with the insight**, not the methodology\n2. **Quantify uncertainty** - ranges, not point estimates\n3. **State limitations** - what this analysis can't tell you\n4. **Recommend next steps** - what would strengthen the conclusion\n\n## Red Flags to Escalate\n\n- User wants to \"prove\" a predetermined conclusion\n- Sample size too small for reliable inference\n- Data quality issues that invalidate analysis\n- Confounders that can't be controlled for\n\n## External Endpoints\n\nThis skill makes no external network requests.\n\n| Endpoint | Data Sent | Purpose |\n|----------|-----------|---------|\n| None | None | N/A |\n\nNo data is sent externally.\n\n## Security & Privacy\n\nData that leaves your machine:\n- Nothing by default.\n\nData that stays local:\n- Nothing by default.\n\nThis skill does NOT:\n- Access undeclared external endpoints.\n- Store credentials or raw exports in hidden local memory files.\n- Create or depend on local folder systems for persistence.\n- Create automations or background jobs without explicit user confirmation.\n- Rewrite its own instruction source files.\n\n## Related Skills\nInstall with `clawhub install <slug>` if user confirms:\n- `sql` - query design and review for reliable data extraction.\n- `csv` - cleanup and normalization for tabular inputs before analysis.\n- `dashboard` - implementation patterns for KPI visualization layers.\n- `report` - structured stakeholder-facing deliverables after analysis.\n- `business-intelligence` - KPI systems and operating cadence beyond one-off analysis.\n\n## Feedback\n\n- If useful: `clawhub star data-analysis`\n- Stay updated: `clawhub sync`\n\nFile v1.0.2:_meta.json\n\n{\n  \"ownerId\": \"kn73vp5rarc3b14rc7wjcw8f8580t5d1\",\n  \"slug\": \"data-analysis\",\n  \"version\": \"1.0.2\",\n  \"publishedAt\": 1773241910484\n}\n\nFile v1.0.2:chart-selection.md\n\n# Chart Selection\n\nChoose visuals based on the question, not on what is easiest to render.\n\n## Question to Chart Map\n\n| Question | Preferred chart | Notes |\n|----------|-----------------|-------|\n| How is a metric changing over time? | line chart | annotate structural breaks and missing data |\n| Which groups are highest or lowest? | sorted bar chart | keep a shared baseline |\n| How is the distribution shaped? | histogram or box plot | avoid average-only summaries |\n| Are two variables related? | scatter plot | show trend and outliers separately |\n| How do parts contribute to the whole? | stacked bar with totals | keep category count low |\n| Where are users dropping? | funnel chart | define the time window explicitly |\n| How do cohorts retain over time? | cohort table or heatmap | show cohort size alongside retention |\n\n## Default Rules\n\n- Bars start at zero unless there is a strong reason not to.\n- Show underlying counts next to percentages when denominators are small.\n- Prefer direct labels over legends when possible.\n- Use one chart per decision question, not one chart per available metric.\n\n## Visual Anti-Patterns\n\n- Pie charts with many slices -> comparisons become guesswork.\n- Dual-axis charts -> viewers infer relationships that are not there.\n- Cumulative-only charts -> hide recent deterioration or recovery.\n- Truncated bar axes -> exaggerate small differences.\n- Stacked areas with many categories -> impossible to compare layers.\n\n## Before Shipping a Chart\n\nCheck:\n\n1. What decision question this chart answers.\n2. Whether the baseline is visible.\n3. Whether the grain and time window match the narrative.\n4. Whether annotations explain outages, launches, or missing data.\n5. Whether a table would be clearer than the chart.\n\nFile v1.0.2:decision-briefs.md\n\n# Decision Briefs\n\nUse these templates to turn analysis into action instead of dumping findings.\n\n## Standard Decision Brief\n\n1. Decision question.\n2. Short answer.\n3. Evidence: key numbers and comparison baseline.\n4. Confidence: high, medium, or low, with one sentence why.\n5. Caveats and what could still change the conclusion.\n6. Recommended next action, owner, and due date.\n\n## Experiment Readout\n\n- Hypothesis:\n- Primary metric and guardrails:\n- Estimated effect and uncertainty:\n- Segment differences:\n- Ship, iterate, or stop:\n- Follow-up test:\n\n## Anomaly Note\n\n- What moved:\n- Since when:\n- Likely drivers:\n- Data quality checks passed or failed:\n- Immediate action:\n- What to watch next:\n\n## Executive Summary\n\n- One-sentence answer.\n- Two or three supporting bullets with numbers.\n- One caveat.\n- One decision or escalation request.\n\n## Writing Rules\n\n- Lead with the answer, not the method.\n- Translate statistics into business implications.\n- Separate observations from recommendations.\n- If confidence is low, say what would raise confidence.\n- Avoid dumping every cut you explored; keep only evidence that changes the decision.\n\nFile v1.0.2:metric-contracts.md\n\n# Metric Contracts\n\nUse this when a KPI, dashboard tile, or report number could be interpreted in more than one way.\n\n## Contract Template\n\nCapture each metric in this order before trusting comparisons:\n\n1. Business question the metric is meant to answer.\n2. Entity and grain: user, account, order, session, day, week, month.\n3. Numerator and denominator with exact inclusion logic.\n4. Filters and exclusions: internal traffic, refunds, test accounts, paused users.\n5. Time window, timezone, and refresh cadence.\n6. Source of truth and owner.\n7. Known caveats, version changes, and safe interpretation range.\n\n## Minimum Contract Output\n\n| Field | Example |\n|-------|---------|\n| Metric | Paid conversion rate |\n| Question | Is onboarding quality improving? |\n| Grain | weekly |\n| Numerator | first paid subscriptions |\n| Denominator | qualified onboarding starts |\n| Filters | excludes employees and QA accounts |\n| Timezone | UTC |\n| Source | warehouse.subscriptions_daily |\n| Owner | Growth lead |\n| Caveat | Launch week excluded because tracking was partial |\n\n## Stop Conditions\n\nDo not present a metric as stable if:\n\n- Numerator or denominator changed between periods.\n- Source ownership is unclear.\n- Filters were applied ad hoc and not documented.\n- Time windows or timezones differ across comparisons.\n- A dashboard label hides a formula change.\n\n## Fast Questions to Ask\n\n- \"What exactly counts in the numerator?\"\n- \"Who is excluded and why?\"\n- \"What is the comparison baseline?\"\n- \"Has this definition changed over time?\"\n- \"Who would dispute this number internally?\"\n\nFile v1.0.2:pitfalls.md\n\n# Analytical Pitfalls — Detailed Examples\n\n## Simpson's Paradox\n\n**What it is:** A trend that appears in aggregated data reverses when you segment by a key variable.\n\n**Example:**\n- Overall: Treatment A has 80% success, Treatment B has 85% -> \"B is better\"\n- But segmented by severity:\n  - Mild cases: A=90%, B=85% -> A is better\n  - Severe cases: A=70%, B=65% -> A is better\n- Paradox: A is better in BOTH groups, but B looks better overall because B got more mild cases\n\n**How to catch:** Always segment by obvious confounders (user type, time period, source, severity) before concluding.\n\n---\n\n## Survivorship Bias\n\n**What it is:** Drawing conclusions only from \"survivors\" while ignoring those who dropped out.\n\n**Example:**\n- \"Users who completed onboarding have 80% retention!\"\n- Problem: You're only looking at users who already demonstrated commitment by completing onboarding\n- The 60% who abandoned onboarding aren't in your \"user\" dataset\n\n**How to catch:** Ask \"Who is NOT in this dataset that should be?\" Include churned users, failed attempts, non-converters.\n\n---\n\n## Comparing Unequal Periods\n\n**What it is:** Comparing metrics across time periods of different lengths or characteristics.\n\n**Examples:**\n- February (28 days) vs January (31 days) revenue\n- Holiday week vs normal week traffic\n- Q4 (holiday season) vs Q1 for e-commerce\n\n**How to catch:**\n- Normalize to per-day, per-user, or per-session\n- Compare same period last year (YoY) not sequential months\n- Flag seasonal factors explicitly\n\n---\n\n## p-Hacking (Multiple Comparisons)\n\n**What it is:** Running many statistical tests until finding a \"significant\" result, then reporting only that one.\n\n**Example:**\n- Test 20 different user segments for conversion difference\n- At p=0.05, expect 1 \"significant\" result by chance alone\n- Report: \"Segment X shows significant improvement!\" (cherry-picked)\n\n**How to catch:**\n- Apply Bonferroni correction (divide alpha by number of tests)\n- Pre-register hypotheses before looking at data\n- Report ALL tests run, not just significant ones\n\n---\n\n## Spurious Correlation in Time Series\n\n**What it is:** Two variables both trending over time appear correlated, but the relationship is meaningless.\n\n**Example:**\n- \"Revenue and employee count are 95% correlated!\"\n- Both grew over time. Controlling for time, there's no relationship.\n- Classic: \"Ice cream sales correlate with drowning deaths\" (both rise in summer)\n\n**How to catch:**\n- Detrend both series before correlating\n- Check if relationship holds within time periods\n- Ask: \"Is there a causal mechanism, or just shared time trend?\"\n\n---\n\n## Aggregating Percentages\n\n**What it is:** Averaging percentages instead of recalculating from underlying totals.\n\n**Example:**\n- Store A: 10/100 = 10% conversion\n- Store B: 5/10 = 50% conversion\n- Wrong: \"Average conversion is 30%\"\n- Right: 15/110 = 13.6% conversion\n\n**How to catch:** Never average percentages. Sum numerators, sum denominators, recalculate.\n\n---\n\n## Selection Bias in A/B Tests\n\n**What it is:** Treatment and control groups differ systematically before treatment is applied.\n\n**Examples:**\n- Users who opted into new feature vs those who didn't\n- Early adopters (Monday signups) vs late week (Friday signups)\n- Users who saw the experiment (loaded fast enough) vs those who didn't\n\n**How to catch:**\n- Verify pre-experiment metrics are balanced\n- Use intention-to-treat analysis\n- Check for differential attrition\n\n---\n\n## Confusing Causation\n\n**What it is:** Assuming X causes Y when the relationship might be: Y causes X, Z causes both, or it's coincidental.\n\n**Example:**\n- \"Power users have higher retention\"\n- Did power usage cause retention? Or did retained users become power users over time? Or does a third factor (job role) drive both?\n\n**How to catch:**\n- Can you run an experiment? (randomize treatment)\n- Is there a natural experiment? (policy change, feature rollout)\n- At minimum: control for obvious confounders\n\nFile v1.0.2:techniques.md\n\n# Analysis Techniques — When to Use Each\n\n## Hypothesis Testing\n\n**Use when:** Comparing two groups to determine if a difference is real or random chance.\n\n**Technique selection:**\n| Data type | Groups | Test |\n|-----------|--------|------|\n| Continuous | 2 | t-test (if normal) or Mann-Whitney |\n| Continuous | 3+ | ANOVA or Kruskal-Wallis |\n| Proportions | 2 | Chi-square or Fisher's exact |\n| Paired data | 2 | Paired t-test or Wilcoxon signed-rank |\n\n**Key outputs:**\n- p-value (probability of seeing this difference by chance)\n- Effect size (how big is the difference - Cohen's d, odds ratio)\n- Confidence interval (range of plausible true values)\n\n**Watch out for:**\n- Large samples make everything \"significant\" - focus on effect size\n- Multiple comparisons inflate false positives\n- Normality assumptions (use non-parametric if violated)\n\n---\n\n## Cohort Analysis\n\n**Use when:** Understanding how user behavior changes over time, segmented by when they started.\n\n**Types:**\n- **Retention cohorts:** % of users still active N days after signup\n- **Revenue cohorts:** Revenue per cohort over time\n- **Behavioral cohorts:** Feature adoption by signup cohort\n\n**Setup:**\n1. Define cohort (usually signup week/month)\n2. Define event (login, purchase, specific action)\n3. Define time windows (day 1, 7, 30, 90)\n4. Build matrix: cohort × time period\n\n**Key outputs:**\n- Retention curves (line chart by cohort)\n- Cohort comparison (are newer cohorts performing better?)\n- Time-to-event patterns\n\n**Watch out for:**\n- Cohort size differences (small cohorts = noisy data)\n- Seasonality (December cohort behaves differently)\n- Definition consistency (what counts as \"active\"?)\n\n---\n\n## Funnel Analysis\n\n**Use when:** Understanding conversion through a multi-step process.\n\n**Setup:**\n1. Define stages (visit -> signup -> activate -> purchase)\n2. Count users at each stage\n3. Calculate drop-off rates between stages\n\n**Key outputs:**\n- Conversion rates per stage\n- Biggest drop-off points\n- Segment comparison (mobile vs desktop funnels)\n\n**Watch out for:**\n- Time window (did they convert eventually, or just not today?)\n- Stage ordering (users don't always follow linear paths)\n- Defining \"same session\" vs \"ever\"\n\n---\n\n## Regression Analysis\n\n**Use when:** Understanding what predicts an outcome, controlling for other factors.\n\n**Types:**\n- **Linear:** Continuous outcome (revenue, time spent)\n- **Logistic:** Binary outcome (churned/retained, converted/didn't)\n- **Poisson:** Count outcome (purchases, logins)\n\n**Key outputs:**\n- Coefficients (effect of each variable, holding others constant)\n- R² (how much variance is explained)\n- p-values per variable\n- Residual plots (are assumptions met?)\n\n**Watch out for:**\n- Multicollinearity (correlated predictors)\n- Omitted variable bias (missing important controls)\n- Extrapolation beyond data range\n- Causation claims from observational data\n\n---\n\n## Segmentation/Clustering\n\n**Use when:** Discovering natural groups in your data.\n\n**Techniques:**\n- **K-means:** Simple, fast, assumes spherical clusters\n- **Hierarchical:** Shows cluster relationships, good for exploration\n- **RFM:** Business-specific (Recency, Frequency, Monetary)\n\n**Process:**\n1. Select features (what defines a segment?)\n2. Normalize features (so scale doesn't dominate)\n3. Choose number of clusters (elbow method, silhouette score)\n4. Profile each cluster (what makes them different?)\n\n**Key outputs:**\n- Cluster profiles (avg values per segment)\n- Segment sizes\n- Distinguishing characteristics\n\n**Watch out for:**\n- Garbage in, garbage out (feature selection matters)\n- Cluster count is subjective\n- Stability (do clusters hold with different random seeds?)\n\n---\n\n## Anomaly Detection\n\n**Use when:** Finding unusual data points that warrant investigation.\n\n**Approaches:**\n- **Statistical:** Points beyond 2-3 standard deviations\n- **IQR method:** Below Q1-1.5×IQR or above Q3+1.5×IQR\n- **Isolation Forest:** For multivariate anomalies\n- **Domain rules:** Negative revenue, future dates, impossible values\n\n**Key outputs:**\n- Flagged records with anomaly scores\n- Context (why is this unusual?)\n- Severity (how far from normal?)\n\n**Watch out for:**\n- Seasonality (Black Friday isn't an anomaly)\n- Trends (growth makes old \"normal\" look like anomalies)\n- False positives (investigate before acting)\n\n---\n\n## Time Series Analysis\n\n**Use when:** Understanding patterns in data over time.\n\n**Components:**\n- **Trend:** Long-term direction\n- **Seasonality:** Repeating patterns (daily, weekly, yearly)\n- **Noise:** Random variation\n\n**Techniques:**\n- **Moving averages:** Smooth out noise\n- **Decomposition:** Separate trend, seasonal, residual\n- **Year-over-year:** Compare same period last year\n\n**Key outputs:**\n- Trend direction and strength\n- Seasonal patterns identified\n- Forecast with uncertainty bands\n\n**Watch out for:**\n- Comparing different lengths (months vary in days)\n- Holidays/events (one-time vs recurring)\n- Structural breaks (COVID, product changes)\n\nArchive v1.0.1: 4 files, 6595 bytes\n\nFiles: pitfalls.md (3965b), SKILL.md (3518b), techniques.md (5001b), _meta.json (132b)\n\nFile v1.0.1:SKILL.md\n\n---\nname: Data Analysis\nslug: data-analysis\nversion: 1.0.1\nhomepage: https://clawic.com/skills/data-analysis\ndescription: \"Data analysis and visualization. Query databases, generate reports, automate spreadsheets, and turn raw data into clear, actionable insights. Use when (1) you need to analyze, visualize, or explain data; (2) the user wants reports, dashboards, or metrics turned into a decision; (3) the work involves SQL, Python, spreadsheets, BI tools, or notebooks; (4) you need to compare segments, cohorts, funnels, experiments, or time periods; (5) the user explicitly installs or references the skill for the current task.\"\nchangelog: \"Clearer promise and activation cues for analysis tasks.\"\n---\n\n## When to Load\n\nUser asks about: analyzing data, finding patterns, understanding metrics, testing hypotheses, cohort analysis, A/B testing, churn analysis, statistical significance.\n\n## Core Principle\n\nAnalysis without a decision is just arithmetic. Always clarify: **What would change if this analysis shows X vs Y?**\n\n## Methodology First\n\nBefore touching data:\n1. **What decision** is this analysis supporting?\n2. **What would change your mind?** (the real question)\n3. **What data do you actually have** vs what you wish you had?\n4. **What timeframe** is relevant?\n\n## Statistical Rigor Checklist\n\n- [ ] Sample size sufficient? (small N = wide confidence intervals)\n- [ ] Comparison groups fair? (same time period, similar conditions)\n- [ ] Multiple comparisons? (20 tests = 1 \"significant\" by chance)\n- [ ] Effect size meaningful? (statistically significant ≠ practically important)\n- [ ] Uncertainty quantified? (\"12-18% lift\" not just \"15% lift\")\n\n## Analytical Pitfalls to Catch\n\n| Pitfall | What it looks like | How to avoid |\n|---------|-------------------|--------------|\n| Simpson's Paradox | Trend reverses when you segment | Always check by key dimensions |\n| Survivorship bias | Only analyzing current users | Include churned/failed in dataset |\n| Comparing unequal periods | Feb (28d) vs March (31d) | Normalize to per-day or same-length windows |\n| p-hacking | Testing until something is \"significant\" | Pre-register hypotheses or adjust for multiple comparisons |\n| Correlation in time series | Both went up = \"related\" | Check if controlling for time removes relationship |\n| Aggregating percentages | Averaging percentages directly | Re-calculate from underlying totals |\n\nFor detailed examples of each pitfall, see `pitfalls.md`.\n\n## Approach Selection\n\n| Question type | Approach | Key output |\n|---------------|----------|------------|\n| \"Is X different from Y?\" | Hypothesis test | p-value + effect size + CI |\n| \"What predicts Z?\" | Regression/correlation | Coefficients + R² + residual check |\n| \"How do users behave over time?\" | Cohort analysis | Retention curves by cohort |\n| \"Are these groups different?\" | Segmentation | Profiles + statistical comparison |\n| \"What's unusual?\" | Anomaly detection | Flagged points + context |\n\nFor technique details and when to use each, see `techniques.md`.\n\n## Output Standards\n\n1. **Lead with the insight**, not the methodology\n2. **Quantify uncertainty** — ranges, not point estimates\n3. **State limitations** — what this analysis can't tell you\n4. **Recommend next steps** — what would strengthen the conclusion\n\n## Red Flags to Escalate\n\n- User wants to \"prove\" a predetermined conclusion\n- Sample size too small for reliable inference\n- Data quality issues that invalidate analysis\n- Confounders that can't be controlled for\n\nFile v1.0.1:_meta.json\n\n{\n  \"ownerId\": \"kn73vp5rarc3b14rc7wjcw8f8580t5d1\",\n  \"slug\": \"data-analysis\",\n  \"version\": \"1.0.1\",\n  \"publishedAt\": 1773240327317\n}\n\nFile v1.0.1:pitfalls.md\n\n# Analytical Pitfalls — Detailed Examples\n\n## Simpson's Paradox\n\n**What it is:** A trend that appears in aggregated data reverses when you segment by a key variable.\n\n**Example:** \n- Overall: Treatment A has 80% success, Treatment B has 85% → \"B is better\"\n- But segmented by severity:\n  - Mild cases: A=90%, B=85% → A is better\n  - Severe cases: A=70%, B=65% → A is better\n- Paradox: A is better in BOTH groups, but B looks better overall because B got more mild cases\n\n**How to catch:** Always segment by obvious confounders (user type, time period, source, severity) before concluding.\n\n---\n\n## Survivorship Bias\n\n**What it is:** Drawing conclusions only from \"survivors\" while ignoring those who dropped out.\n\n**Example:**\n- \"Users who completed onboarding have 80% retention!\" \n- Problem: You're only looking at users who already demonstrated commitment by completing onboarding\n- The 60% who abandoned onboarding aren't in your \"user\" dataset\n\n**How to catch:** Ask \"Who is NOT in this dataset that should be?\" Include churned users, failed attempts, non-converters.\n\n---\n\n## Comparing Unequal Periods\n\n**What it is:** Comparing metrics across time periods of different lengths or characteristics.\n\n**Examples:**\n- February (28 days) vs January (31 days) revenue\n- Holiday week vs normal week traffic\n- Q4 (holiday season) vs Q1 for e-commerce\n\n**How to catch:** \n- Normalize to per-day, per-user, or per-session\n- Compare same period last year (YoY) not sequential months\n- Flag seasonal factors explicitly\n\n---\n\n## p-Hacking (Multiple Comparisons)\n\n**What it is:** Running many statistical tests until finding a \"significant\" result, then reporting only that one.\n\n**Example:**\n- Test 20 different user segments for conversion difference\n- At p=0.05, expect 1 \"significant\" result by chance alone\n- Report: \"Segment X shows significant improvement!\" (cherry-picked)\n\n**How to catch:**\n- Apply Bonferroni correction (divide α by number of tests)\n- Pre-register hypotheses before looking at data\n- Report ALL tests run, not just significant ones\n\n---\n\n## Spurious Correlation in Time Series\n\n**What it is:** Two variables both trending over time appear correlated, but the relationship is meaningless.\n\n**Example:**\n- \"Revenue and employee count are 95% correlated!\"\n- Both grew over time. Controlling for time, there's no relationship.\n- Classic: \"Ice cream sales correlate with drowning deaths\" (both rise in summer)\n\n**How to catch:**\n- Detrend both series before correlating\n- Check if relationship holds within time periods\n- Ask: \"Is there a causal mechanism, or just shared time trend?\"\n\n---\n\n## Aggregating Percentages\n\n**What it is:** Averaging percentages instead of recalculating from underlying totals.\n\n**Example:**\n- Store A: 10/100 = 10% conversion\n- Store B: 5/10 = 50% conversion\n- Wrong: \"Average conversion is 30%\"\n- Right: 15/110 = 13.6% conversion\n\n**How to catch:** Never average percentages. Sum numerators, sum denominators, recalculate.\n\n---\n\n## Selection Bias in A/B Tests\n\n**What it is:** Treatment and control groups differ systematically before treatment is applied.\n\n**Examples:**\n- Users who opted into new feature vs those who didn't\n- Early adopters (Monday signups) vs late week (Friday signups)\n- Users who saw the experiment (loaded fast enough) vs those who didn't\n\n**How to catch:**\n- Verify pre-experiment metrics are balanced\n- Use intention-to-treat analysis\n- Check for differential attrition\n\n---\n\n## Confusing Causation\n\n**What it is:** Assuming X causes Y when the relationship might be: Y causes X, Z causes both, or it's coincidental.\n\n**Example:**\n- \"Power users have higher retention\" \n- Did power usage cause retention? Or did retained users become power users over time? Or does a third factor (job role) drive both?\n\n**How to catch:**\n- Can you run an experiment? (randomize treatment)\n- Is there a natural experiment? (policy change, feature rollout)\n- At minimum: control for obvious confounders\n\nFile v1.0.1:techniques.md\n\n# Analysis Techniques — When to Use Each\n\n## Hypothesis Testing\n\n**Use when:** Comparing two groups to determine if a difference is real or random chance.\n\n**Technique selection:**\n| Data type | Groups | Test |\n|-----------|--------|------|\n| Continuous | 2 | t-test (if normal) or Mann-Whitney |\n| Continuous | 3+ | ANOVA or Kruskal-Wallis |\n| Proportions | 2 | Chi-square or Fisher's exact |\n| Paired data | 2 | Paired t-test or Wilcoxon signed-rank |\n\n**Key outputs:**\n- p-value (probability of seeing this difference by chance)\n- Effect size (how big is the difference — Cohen's d, odds ratio)\n- Confidence interval (range of plausible true values)\n\n**Watch out for:**\n- Large samples make everything \"significant\" — focus on effect size\n- Multiple comparisons inflate false positives\n- Normality assumptions (use non-parametric if violated)\n\n---\n\n## Cohort Analysis\n\n**Use when:** Understanding how user behavior changes over time, segmented by when they started.\n\n**Types:**\n- **Retention cohorts:** % of users still active N days after signup\n- **Revenue cohorts:** Revenue per cohort over time\n- **Behavioral cohorts:** Feature adoption by signup cohort\n\n**Setup:**\n1. Define cohort (usually signup week/month)\n2. Define event (login, purchase, specific action)\n3. Define time windows (day 1, 7, 30, 90)\n4. Build matrix: cohort × time period\n\n**Key outputs:**\n- Retention curves (line chart by cohort)\n- Cohort comparison (are newer cohorts performing better?)\n- Time-to-event patterns\n\n**Watch out for:**\n- Cohort size differences (small cohorts = noisy data)\n- Seasonality (December cohort behaves differently)\n- Definition consistency (what counts as \"active\"?)\n\n---\n\n## Funnel Analysis\n\n**Use when:** Understanding conversion through a multi-step process.\n\n**Setup:**\n1. Define stages (visit → signup → activate → purchase)\n2. Count users at each stage\n3. Calculate drop-off rates between stages\n\n**Key outputs:**\n- Conversion rates per stage\n- Biggest drop-off points\n- Segment comparison (mobile vs desktop funnels)\n\n**Watch out for:**\n- Time window (did they convert eventually, or just not today?)\n- Stage ordering (users don't always follow linear paths)\n- Defining \"same session\" vs \"ever\"\n\n---\n\n## Regression Analysis\n\n**Use when:** Understanding what predicts an outcome, controlling for other factors.\n\n**Types:**\n- **Linear:** Continuous outcome (revenue, time spent)\n- **Logistic:** Binary outcome (churned/retained, converted/didn't)\n- **Poisson:** Count outcome (purchases, logins)\n\n**Key outputs:**\n- Coefficients (effect of each variable, holding others constant)\n- R² (how much variance is explained)\n- p-values per variable\n- Residual plots (are assumptions met?)\n\n**Watch out for:**\n- Multicollinearity (correlated predictors)\n- Omitted variable bias (missing important controls)\n- Extrapolation beyond data range\n- Causation claims from observational data\n\n---\n\n## Segmentation/Clustering\n\n**Use when:** Discovering natural groups in your data.\n\n**Techniques:**\n- **K-means:** Simple, fast, assumes spherical clusters\n- **Hierarchical:** Shows cluster relationships, good for exploration\n- **RFM:** Business-specific (Recency, Frequency, Monetary)\n\n**Process:**\n1. Select features (what defines a segment?)\n2. Normalize features (so scale doesn't dominate)\n3. Choose number of clusters (elbow method, silhouette score)\n4. Profile each cluster (what makes them different?)\n\n**Key outputs:**\n- Cluster profiles (avg values per segment)\n- Segment sizes\n- Distinguishing characteristics\n\n**Watch out for:**\n- Garbage in, garbage out (feature selection matters)\n- Cluster count is subjective\n- Stability (do clusters hold with different random seeds?)\n\n---\n\n## Anomaly Detection\n\n**Use when:** Finding unusual data points that warrant investigation.\n\n**Approaches:**\n- **Statistical:** Points beyond 2-3 standard deviations\n- **IQR method:** Below Q1-1.5×IQR or above Q3+1.5×IQR\n- **Isolation Forest:** For multivariate anomalies\n- **Domain rules:** Negative revenue, future dates, impossible values\n\n**Key outputs:**\n- Flagged records with anomaly scores\n- Context (why is this unusual?)\n- Severity (how far from normal?)\n\n**Watch out for:**\n- Seasonality (Black Friday isn't an anomaly)\n- Trends (growth makes old \"normal\" look like anomalies)\n- False positives (investigate before acting)\n\n---\n\n## Time Series Analysis\n\n**Use when:** Understanding patterns in data over time.\n\n**Components:**\n- **Trend:** Long-term direction\n- **Seasonality:** Repeating patterns (daily, weekly, yearly)\n- **Noise:** Random variation\n\n**Techniques:**\n- **Moving averages:** Smooth out noise\n- **Decomposition:** Separate trend, seasonal, residual\n- **Year-over-year:** Compare same period last year\n\n**Key outputs:**\n- Trend direction and strength\n- Seasonal patterns identified\n- Forecast with uncertainty bands\n\n**Watch out for:**\n- Comparing different lengths (months vary in days)\n- Holidays/events (one-time vs recurring)\n- Structural breaks (COVID, product changes)\n\nArchive v1.0.0: 4 files, 6322 bytes\n\nFiles: pitfalls.md (3965b), SKILL.md (2960b), techniques.md (5001b), _meta.json (132b)\n\nFile v1.0.0:SKILL.md\n\n---\nname: Data Analysis\ndescription: Turn raw data into decisions with statistical rigor, proper methodology, and awareness of analytical pitfalls.\n---\n\n## When to Load\n\nUser asks about: analyzing data, finding patterns, understanding metrics, testing hypotheses, cohort analysis, A/B testing, churn analysis, statistical significance.\n\n## Core Principle\n\nAnalysis without a decision is just arithmetic. Always clarify: **What would change if this analysis shows X vs Y?**\n\n## Methodology First\n\nBefore touching data:\n1. **What decision** is this analysis supporting?\n2. **What would change your mind?** (the real question)\n3. **What data do you actually have** vs what you wish you had?\n4. **What timeframe** is relevant?\n\n## Statistical Rigor Checklist\n\n- [ ] Sample size sufficient? (small N = wide confidence intervals)\n- [ ] Comparison groups fair? (same time period, similar conditions)\n- [ ] Multiple comparisons? (20 tests = 1 \"significant\" by chance)\n- [ ] Effect size meaningful? (statistically significant ≠ practically important)\n- [ ] Uncertainty quantified? (\"12-18% lift\" not just \"15% lift\")\n\n## Analytical Pitfalls to Catch\n\n| Pitfall | What it looks like | How to avoid |\n|---------|-------------------|--------------|\n| Simpson's Paradox | Trend reverses when you segment | Always check by key dimensions |\n| Survivorship bias | Only analyzing current users | Include churned/failed in dataset |\n| Comparing unequal periods | Feb (28d) vs March (31d) | Normalize to per-day or same-length windows |\n| p-hacking | Testing until something is \"significant\" | Pre-register hypotheses or adjust for multiple comparisons |\n| Correlation in time series | Both went up = \"related\" | Check if controlling for time removes relationship |\n| Aggregating percentages | Averaging percentages directly | Re-calculate from underlying totals |\n\nFor detailed examples of each pitfall, see `pitfalls.md`.\n\n## Approach Selection\n\n| Question type | Approach | Key output |\n|---------------|----------|------------|\n| \"Is X different from Y?\" | Hypothesis test | p-value + effect size + CI |\n| \"What predicts Z?\" | Regression/correlation | Coefficients + R² + residual check |\n| \"How do users behave over time?\" | Cohort analysis | Retention curves by cohort |\n| \"Are these groups different?\" | Segmentation | Profiles + statistical comparison |\n| \"What's unusual?\" | Anomaly detection | Flagged points + context |\n\nFor technique details and when to use each, see `techniques.md`.\n\n## Output Standards\n\n1. **Lead with the insight**, not the methodology\n2. **Quantify uncertainty** — ranges, not point estimates\n3. **State limitations** — what this analysis can't tell you\n4. **Recommend next steps** — what would strengthen the conclusion\n\n## Red Flags to Escalate\n\n- User wants to \"prove\" a predetermined conclusion\n- Sample size too small for reliable inference\n- Data quality issues that invalidate analysis\n- Confounders that can't be controlled for\n\nFile v1.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn73vp5rarc3b14rc7wjcw8f8580t5d1\",\n  \"slug\": \"data-analysis\",\n  \"version\": \"1.0.0\",\n  \"publishedAt\": 1770905427744\n}\n\nFile v1.0.0:pitfalls.md\n\n# Analytical Pitfalls — Detailed Examples\n\n## Simpson's Paradox\n\n**What it is:** A trend that appears in aggregated data reverses when you segment by a key variable.\n\n**Example:** \n- Overall: Treatment A has 80% success, Treatment B has 85% → \"B is better\"\n- But segmented by severity:\n  - Mild cases: A=90%, B=85% → A is better\n  - Severe cases: A=70%, B=65% → A is better\n- Paradox: A is better in BOTH groups, but B looks better overall because B got more mild cases\n\n**How to catch:** Always segment by obvious confounders (user type, time period, source, severity) before concluding.\n\n---\n\n## Survivorship Bias\n\n**What it is:** Drawing conclusions only from \"survivors\" while ignoring those who dropped out.\n\n**Example:**\n- \"Users who completed onboarding have 80% retention!\" \n- Problem: You're only looking at users who already demonstrated commitment by completing onboarding\n- The 60% who abandoned onboarding aren't in your \"user\" dataset\n\n**How to catch:** Ask \"Who is NOT in this dataset that should be?\" Include churned users, failed attempts, non-converters.\n\n---\n\n## Comparing Unequal Periods\n\n**What it is:** Comparing metrics across time periods of different lengths or characteristics.\n\n**Examples:**\n- February (28 days) vs January (31 days) revenue\n- Holiday week vs normal week traffic\n- Q4 (holiday season) vs Q1 for e-commerce\n\n**How to catch:** \n- Normalize to per-day, per-user, or per-session\n- Compare same period last year (YoY) not sequential months\n- Flag seasonal factors explicitly\n\n---\n\n## p-Hacking (Multiple Comparisons)\n\n**What it is:** Running many statistical tests until finding a \"significant\" result, then reporting only that one.\n\n**Example:**\n- Test 20 different user segments for conversion difference\n- At p=0.05, expect 1 \"significant\" result by chance alone\n- Report: \"Segment X shows significant improvement!\" (cherry-picked)\n\n**How to catch:**\n- Apply Bonferroni correction (divide α by number of tests)\n- Pre-register hypotheses before looking at data\n- Report ALL tests run, not just significant ones\n\n---\n\n## Spurious Correlation in Time Series\n\n**What it is:** Two variables both trending over time appear correlated, but the relationship is meaningless.\n\n**Example:**\n- \"Revenue and employee count are 95% correlated!\"\n- Both grew over time. Controlling for time, there's no relationship.\n- Classic: \"Ice cream sales correlate with drowning deaths\" (both rise in summer)\n\n**How to catch:**\n- Detrend both series before correlating\n- Check if relationship holds within time periods\n- Ask: \"Is there a causal mechanism, or just shared time trend?\"\n\n---\n\n## Aggregating Percentages\n\n**What it is:** Averaging percentages instead of recalculating from underlying totals.\n\n**Example:**\n- Store A: 10/100 = 10% conversion\n- Store B: 5/10 = 50% conversion\n- Wrong: \"Average conversion is 30%\"\n- Right: 15/110 = 13.6% conversion\n\n**How to catch:** Never average percentages. Sum numerators, sum denominators, recalculate.\n\n---\n\n## Selection Bias in A/B Tests\n\n**What it is:** Treatment and control groups differ systematically before treatment is applied.\n\n**Examples:**\n- Users who opted into new feature vs those who didn't\n- Early adopters (Monday signups) vs late week (Friday signups)\n- Users who saw the experiment (loaded fast enough) vs those who didn't\n\n**How to catch:**\n- Verify pre-experiment metrics are balanced\n- Use intention-to-treat analysis\n- Check for differential attrition\n\n---\n\n## Confusing Causation\n\n**What it is:** Assuming X causes Y when the relationship might be: Y causes X, Z causes both, or it's coincidental.\n\n**Example:**\n- \"Power users have higher retention\" \n- Did power usage cause retention? Or did retained users become power users over time? Or does a third factor (job role) drive both?\n\n**How to catch:**\n- Can you run an experiment? (randomize treatment)\n- Is there a natural experiment? (policy change, feature rollout)\n- At minimum: control for obvious confounders\n\nFile v1.0.0:techniques.md\n\n# Analysis Techniques — When to Use Each\n\n## Hypothesis Testing\n\n**Use when:** Comparing two groups to determine if a difference is real or random chance.\n\n**Technique selection:**\n| Data type | Groups | Test |\n|-----------|--------|------|\n| Continuous | 2 | t-test (if normal) or Mann-Whitney |\n| Continuous | 3+ | ANOVA or Kruskal-Wallis |\n| Proportions | 2 | Chi-square or Fisher's exact |\n| Paired data | 2 | Paired t-test or Wilcoxon signed-rank |\n\n**Key outputs:**\n- p-value (probability of seeing this difference by chance)\n- Effect size (how big is the difference — Cohen's d, odds ratio)\n- Confidence interval (range of plausible true values)\n\n**Watch out for:**\n- Large samples make everything \"significant\" — focus on effect size\n- Multiple comparisons inflate false positives\n- Normality assumptions (use non-parametric if violated)\n\n---\n\n## Cohort Analysis\n\n**Use when:** Understanding how user behavior changes over time, segmented by when they started.\n\n**Types:**\n- **Retention cohorts:** % of users still active N days after signup\n- **Revenue cohorts:** Revenue per cohort over time\n- **Behavioral cohorts:** Feature adoption by signup cohort\n\n**Setup:**\n1. Define cohort (usually signup week/month)\n2. Define event (login, purchase, specific action)\n3. Define time windows (day 1, 7, 30, 90)\n4. Build matrix: cohort × time period\n\n**Key outputs:**\n- Retention curves (line chart by cohort)\n- Cohort comparison (are newer cohorts performing better?)\n- Time-to-event patterns\n\n**Watch out for:**\n- Cohort size differences (small cohorts = noisy data)\n- Seasonality (December cohort behaves differently)\n- Definition consistency (what counts as \"active\"?)\n\n---\n\n## Funnel Analysis\n\n**Use when:** Understanding conversion through a multi-step process.\n\n**Setup:**\n1. Define stages (visit → signup → activate → purchase)\n2. Count users at each stage\n3. Calculate drop-off rates between stages\n\n**Key outputs:**\n- Conversion rates per stage\n- Biggest drop-off points\n- Segment comparison (mobile vs desktop funnels)\n\n**Watch out for:**\n- Time window (did they convert eventually, or just not today?)\n- Stage ordering (users don't always follow linear paths)\n- Defining \"same session\" vs \"ever\"\n\n---\n\n## Regression Analysis\n\n**Use when:** Understanding what predicts an outcome, controlling for other factors.\n\n**Types:**\n- **Linear:** Continuous outcome (revenue, time spent)\n- **Logistic:** Binary outcome (churned/retained, converted/didn't)\n- **Poisson:** Count outcome (purchases, logins)\n\n**Key outputs:**\n- Coefficients (effect of each variable, holding others constant)\n- R² (how much variance is explained)\n- p-values per variable\n- Residual plots (are assumptions met?)\n\n**Watch out for:**\n- Multicollinearity (correlated predictors)\n- Omitted variable bias (missing important controls)\n- Extrapolation beyond data range\n- Causation claims from observational data\n\n---\n\n## Segmentation/Clustering\n\n**Use when:** Discovering natural groups in your data.\n\n**Techniques:**\n- **K-means:** Simple, fast, assumes spherical clusters\n- **Hierarchical:** Shows cluster relationships, good for exploration\n- **RFM:** Business-specific (Recency, Frequency, Monetary)\n\n**Process:**\n1. Select features (what defines a segment?)\n2. Normalize features (so scale doesn't dominate)\n3. Choose number of clusters (elbow method, silhouette score)\n4. Profile each cluster (what makes them different?)\n\n**Key outputs:**\n- Cluster profiles (avg values per segment)\n- Segment sizes\n- Distinguishing characteristics\n\n**Watch out for:**\n- Garbage in, garbage out (feature selection matters)\n- Cluster count is subjective\n- Stability (do clusters hold with different random seeds?)\n\n---\n\n## Anomaly Detection\n\n**Use when:** Finding unusual data points that warrant investigation.\n\n**Approaches:**\n- **Statistical:** Points beyond 2-3 standard deviations\n- **IQR method:** Below Q1-1.5×IQR or above Q3+1.5×IQR\n- **Isolation Forest:** For multivariate anomalies\n- **Domain rules:** Negative revenue, future dates, impossible values\n\n**Key outputs:**\n- Flagged records with anomaly scores\n- Context (why is this unusual?)\n- Severity (how far from normal?)\n\n**Watch out for:**\n- Seasonality (Black Friday isn't an anomaly)\n- Trends (growth makes old \"normal\" look like anomalies)\n- False positives (investigate before acting)\n\n---\n\n## Time Series Analysis\n\n**Use when:** Understanding patterns in data over time.\n\n**Components:**\n- **Trend:** Long-term direction\n- **Seasonality:** Repeating patterns (daily, weekly, yearly)\n- **Noise:** Random variation\n\n**Techniques:**\n- **Moving averages:** Smooth out noise\n- **Decomposition:** Separate trend, seasonal, residual\n- **Year-over-year:** Compare same period last year\n\n**Key outputs:**\n- Trend direction and strength\n- Seasonal patterns identified\n- Forecast with uncertainty bands\n\n**Watch out for:**\n- Comparing different lengths (months vary in days)\n- Holidays/events (one-time vs recurring)\n- Structural breaks (COVID, product changes)","readmeExcerpt":"Skill: Data Analysis Owner: ivangdavila Summary: Data analysis and visualization. Query databases, generate reports, automate spreadsheets, and turn raw data into clear, actionable insights. Use when (1) yo... Tags: latest:1.0.2 Version history: v1.0.2 | 2026-03-11T15:11:50.484Z | user Added metric contracts, chart guidance, and decision brief templates for more reliable analysis. v1.0.1 | 2026-03-11T14:45:27.317Z | ","codeSnippets":[],"executableExamples":[],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: Data Analysis\nslug: data-analysis\nversion: 1.0.2\nhomepage: https://clawic.com/skills/data-analysis\ndescription: \"Data analysis and visualization. Query databases, generate reports, automate spreadsheets, and turn raw data into clear, actionable insights. Use when (1) you need to analyze, visualize, or explain data; (2) the user wants reports, dashboards, or metrics turned into a decision; (3) the work involves SQL, Python, spreadsheets, BI tools, or notebooks; (4) you need to compare segments, cohorts, funnels, experiments, or time periods; (5) the user explicitly installs or references the skill for the current task.\"\nchangelog: Added metric contracts, chart guidance, and decision brief templates for more reliable analysis.\nmetadata: {\"clawdbot\":{\"emoji\":\"D\",\"requires\":{\"bins\":[]},\"os\":[\"linux\",\"darwin\",\"win32\"]}}\n---\n\n## When to Use\n\nUse this skill when the user needs to analyze, explain, or visualize data from SQL, spreadsheets, notebooks, dashboards, exports, or ad hoc tables.\n\nUse it for KPI debugging, experiment readouts, funnel or cohort analysis, anomaly reviews, executive reporting, and quality checks on metrics or query logic.\n\nPrefer this skill over generic coding or spreadsheet help when the hard part is analytical judgment: metric definition, comparison design, interpretation, or recommendation.\n\nUser asks about: analyzing data, finding patterns, understanding metrics, testing hypotheses, cohort analysis, A/B testing, churn analysis, or statistical significance.\n\n## Core Principle\n\nAnalysis without a decision is just arithmetic. Always clarify: **What would change if this analysis shows X vs Y?**\n\n## Methodology First\n\nBefore touching data:\n1. **What decision** is this analysis supporting?\n2. **What would change your mind?** (the real question)\n3. **What data do you actually have** vs what you wish you had?\n4. **What timeframe** is relevant?\n\n## Statistical Rigor Checklist\n\n- [ ] Sample size sufficient? (small N = wide confidence intervals)\n- [ ] Comparison groups fair? (same time period, similar conditions)\n- [ ] Multiple comparisons? (20 tests = 1 \"significant\" by chance)\n- [ ] Effect size meaningful? (statistically significant != practically important)\n- [ ] Uncertainty quantified? (\"12-18% lift\" not just \"15% lift\")\n\n## Architecture\n\nThis skill does not require local folders, persistent memory, or setup state.\n\nUse the included reference files as lightweight guides:\n- `metric-contracts.md` for KPI definitions and caveats\n- `chart-selection.md` for visual choice and chart anti-patterns\n- `decision-briefs.md` for stakeholder-facing outputs\n- `pitfalls.md` and `techniques.md` for analytical rigor and method choice\n\n## Quick Reference\n\nLoad only the smallest relevant file to keep context focused.\n\n| Topic | File |\n|-------|------|\n| Metric definition contracts | `metric-contracts.md` |\n| Visual selection and chart anti-patterns | `chart-selection.md` |\n| Decision-ready output formats | `decision-briefs.md` |\n| Failure modes"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn73vp5rarc3b14rc7wjcw8f8580t5d1\",\n  \"slug\": \"data-analysis\",\n  \"version\": \"1.0.2\",\n  \"publishedAt\": 1773241910484\n}"},{"path":"chart-selection.md","content":"# Chart Selection\n\nChoose visuals based on the question, not on what is easiest to render.\n\n## Question to Chart Map\n\n| Question | Preferred chart | Notes |\n|----------|-----------------|-------|\n| How is a metric changing over time? | line chart | annotate structural breaks and missing data |\n| Which groups are highest or lowest? | sorted bar chart | keep a shared baseline |\n| How is the distribution shaped? | histogram or box plot | avoid average-only summaries |\n| Are two variables related? | scatter plot | show trend and outliers separately |\n| How do parts contribute to the whole? | stacked bar with totals | keep category count low |\n| Where are users dropping? | funnel chart | define the time window explicitly |\n| How do cohorts retain over time? | cohort table or heatmap | show cohort size alongside retention |\n\n## Default Rules\n\n- Bars start at zero unless there is a strong reason not to.\n- Show underlying counts next to percentages when denominators are small.\n- Prefer direct labels over legends when possible.\n- Use one chart per decision question, not one chart per available metric.\n\n## Visual Anti-Patterns\n\n- Pie charts with many slices -> comparisons become guesswork.\n- Dual-axis charts -> viewers infer relationships that are not there.\n- Cumulative-only charts -> hide recent deterioration or recovery.\n- Truncated bar axes -> exaggerate small differences.\n- Stacked areas with many categories -> impossible to compare layers.\n\n## Before Shipping a Chart\n\nCheck:\n\n1. What decision question this chart answers.\n2. Whether the baseline is visible.\n3. Whether the grain and time window match the narrative.\n4. Whether annotations explain outages, launches, or missing data.\n5. Whether a table would be clearer than the chart."},{"path":"decision-briefs.md","content":"# Decision Briefs\n\nUse these templates to turn analysis into action instead of dumping findings.\n\n## Standard Decision Brief\n\n1. Decision question.\n2. Short answer.\n3. Evidence: key numbers and comparison baseline.\n4. Confidence: high, medium, or low, with one sentence why.\n5. Caveats and what could still change the conclusion.\n6. Recommended next action, owner, and due date.\n\n## Experiment Readout\n\n- Hypothesis:\n- Primary metric and guardrails:\n- Estimated effect and uncertainty:\n- Segment differences:\n- Ship, iterate, or stop:\n- Follow-up test:\n\n## Anomaly Note\n\n- What moved:\n- Since when:\n- Likely drivers:\n- Data quality checks passed or failed:\n- Immediate action:\n- What to watch next:\n\n## Executive Summary\n\n- One-sentence answer.\n- Two or three supporting bullets with numbers.\n- One caveat.\n- One decision or escalation request.\n\n## Writing Rules\n\n- Lead with the answer, not the method.\n- Translate statistics into business implications.\n- Separate observations from recommendations.\n- If confidence is low, say what would raise confidence.\n- Avoid dumping every cut you explored; keep only evidence that changes the decision."},{"path":"metric-contracts.md","content":"# Metric Contracts\n\nUse this when a KPI, dashboard tile, or report number could be interpreted in more than one way.\n\n## Contract Template\n\nCapture each metric in this order before trusting comparisons:\n\n1. Business question the metric is meant to answer.\n2. Entity and grain: user, account, order, session, day, week, month.\n3. Numerator and denominator with exact inclusion logic.\n4. Filters and exclusions: internal traffic, refunds, test accounts, paused users.\n5. Time window, timezone, and refresh cadence.\n6. Source of truth and owner.\n7. Known caveats, version changes, and safe interpretation range.\n\n## Minimum Contract Output\n\n| Field | Example |\n|-------|---------|\n| Metric | Paid conversion rate |\n| Question | Is onboarding quality improving? |\n| Grain | weekly |\n| Numerator | first paid subscriptions |\n| Denominator | qualified onboarding starts |\n| Filters | excludes employees and QA accounts |\n| Timezone | UTC |\n| Source | warehouse.subscriptions_daily |\n| Owner | Growth lead |\n| Caveat | Launch week excluded because tracking was partial |\n\n## Stop Conditions\n\nDo not present a metric as stable if:\n\n- Numerator or denominator changed between periods.\n- Source ownership is unclear.\n- Filters were applied ad hoc and not documented.\n- Time windows or timezones differ across comparisons.\n- A dashboard label hides a formula change.\n\n## Fast Questions to Ask\n\n- \"What exactly counts in the numerator?\"\n- \"Who is excluded and why?\"\n- \"What is the comparison baseline?\"\n- \"Has this definition changed over time?\"\n- \"Who would dispute this number internally?\""}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Data analysis and visualization. Query databases, generate reports, automate spreadsheets, and turn raw data into clear, actionable insights. Use when (1) yo... Skill: Data Analysis Owner: ivangdavila Summary: Data analysis and visualization. Query databases, generate reports, automate spreadsheets, and turn raw data into clear, actionable insights. Use when (1) yo... Tags: latest:1.0.2 Version history: v1.0.2 | 2026-03-11T15:11:50.484Z | user Added metric contracts, chart guidance, and decision brief templates for more reliable analysis. v1.0.1 | 2026-03-11T14:45:27.317Z |","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1396,"uniquenessScore":56,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-05-11T06:25:23.421Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-05-11T06:25:23.421Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T12:13:42.606Z","emptyReason":null},"items":[{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-10T18:48:31.762Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}