{"id":"152cdfe2-2c3a-4a45-813d-bfcc4ea4481d","entityType":"agent","slug":"clawhub-gitcanadabrett-data-analysis-reporting","name":"Data Analysis Reporting","canonicalUrl":"https://www.xpersona.co/agent/clawhub-gitcanadabrett-data-analysis-reporting","canonicalPath":"/agent/clawhub-gitcanadabrett-data-analysis-reporting","generatedAt":"2026-10-09T18:05:28.613Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-09T17:15:55.241Z","emptyReason":null},"description":"Turn raw business data (CSV, SQLite, spreadsheets, pasted tables) into clear analytical summaries, trend analysis, and actionable reports for small business... Skill: Data Analysis Reporting Owner: gitcanadabrett Summary: Turn raw business data (CSV, SQLite, spreadsheets, pasted tables) into clear analytical summaries, trend analysis, and actionable reports for small business... Tags: latest:0.1.0 Version history: v0.1.0 | 2026-04-07T23:17:05.424Z | user v0.1.0 — First public release. Turn raw business data into clear analytical reports with plain-language insights. Clarifi","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 2.2K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s179z9xbrcbpqvt6a22em801t983qf9y:data-analysis-reporting","sourceUrl":"https://clawhub.ai/gitcanadabrett/data-analysis-reporting","homepage":"https://clawhub.ai/gitcanadabrett/skills/data-analysis-reporting","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/gitcanadabrett/data-analysis-reporting","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/gitcanadabrett/skills/data-analysis-reporting","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":46,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Turn raw business data (CSV, SQLite, spreadsheets, pasted tables) into clear analytical summaries, trend analysis, and actionable reports for small business... "},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T17:15:55.241Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T17:15:55.241Z","emptyReason":null},"stars":null,"forks":null,"downloads":2231,"packageName":null,"latestVersion":"0.1.0","tractionLabel":"2.2K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T17:15:55.240Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T17:15:55.241Z","lastCrawledAt":"2026-10-09T17:15:55.240Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T17:15:55.240Z","lastVerifiedAt":null,"highlights":[{"version":"0.1.0","createdAt":"2026-04-07T23:17:05.424Z","changelog":"v0.1.0 — First public release. Turn raw business data into clear analytical reports with plain-language insights. Clarification-first workflow, business metrics library (MRR, churn, CAC, LTV, margins), 6-category data quality checks, 5 report templates, PII detection and exclusion protocol. 43/45 QA score across 12 test cases.","fileCount":9,"zipByteSize":24006}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s179z9xbrcbpqvt6a22em801t983qf9y:data-analysis-reporting","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-gitcanadabrett-data-analysis-reporting/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-gitcanadabrett-data-analysis-reporting/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-gitcanadabrett-data-analysis-reporting/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-gitcanadabrett-data-analysis-reporting/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-gitcanadabrett-data-analysis-reporting/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-gitcanadabrett-data-analysis-reporting/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-09T18:05:28.612Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-gitcanadabrett-data-analysis-reporting/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-gitcanadabrett-data-analysis-reporting/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-gitcanadabrett-data-analysis-reporting/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-gitcanadabrett-data-analysis-reporting/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-09T17:15:55.241Z","emptyReason":null},"readme":"Skill: Data Analysis Reporting\n\nOwner: gitcanadabrett\n\nSummary: Turn raw business data (CSV, SQLite, spreadsheets, pasted tables) into clear analytical summaries, trend analysis, and actionable reports for small business...\n\nTags: latest:0.1.0\n\nVersion history:\n\nv0.1.0 | 2026-04-07T23:17:05.424Z | user\n\nv0.1.0 — First public release. Turn raw business data into clear analytical reports with plain-language insights. Clarification-first workflow, business metrics library (MRR, churn, CAC, LTV, margins), 6-category data quality checks, 5 report templates, PII detection and exclusion protocol. 43/45 QA score across 12 test cases.\n\nArchive index:\n\nArchive v0.1.0: 9 files, 24006 bytes\n\nFiles: data-analysis-reporting-spec.md (6516b), README.md (1753b), references/business-metrics.md (7378b), references/data-quality-checks.md (6475b), references/report-templates.md (6526b), references/test-prompts.md (7543b), skill-card.md (2600b), SKILL.md (11874b), _meta.json (142b)\n\nFile v0.1.0:SKILL.md\n\n---\nname: data-analysis-reporting\ndescription: Turn raw business data (CSV, SQLite, spreadsheets, pasted tables) into clear analytical summaries, trend analysis, and actionable reports for small business operators, analysts, and decision-makers. Asks clarifying questions first, delivers plain-language insights before numbers, and labels statistical confidence explicitly. Does not provide financial advice or present projections as fact.\n---\n\n# Data Analysis & Reporting\n\nTurn raw business data into plain-language insights, trend analysis, and actionable reports. Think sharp junior analyst, not statistics engine.\n\n## Trigger conditions\n\nActivate this skill when the user:\n- Pastes or uploads tabular data (CSV, markdown table, tab-separated, pipe-delimited)\n- Asks to analyze, summarize, or report on business data\n- Asks about metrics, KPIs, trends, or performance from a dataset\n- Provides a SQLite database and asks questions about it\n- Asks for a report, executive summary, or data briefing\n- Asks to compare periods, segments, or actuals vs. targets\n\nDo NOT activate when:\n- The user wants to build a dashboard or visualization tool (suggest BI tools)\n- The user needs real-time streaming analytics\n- The user asks for financial advice, investment recommendations, or tax guidance\n- The data is code/logs/system metrics (suggest observability tools instead)\n\n## Work the request in this order\n\n1. **Clarify the question** — before touching the data, understand what the user needs to know and why. Ask up to 3 clarifying questions:\n   - \"What decision does this analysis need to support?\"\n   - \"What time period or comparison matters most?\"\n   - \"Who is the audience for this report?\"\n   If the user provides clear context, skip to step 2.\n\n2. **Ingest and validate** — parse the data, detect column types, run quality checks\n   - Auto-detect: column types (numeric, date, categorical, text)\n   - Flag: missing values, outliers, formatting inconsistencies, duplicate rows\n   - Report data quality issues before proceeding, not after\n   - If data quality is poor enough to undermine analysis, say so and recommend fixes\n\n3. **Propose an analysis plan** — tell the user what you intend to analyze and why, before doing it\n   - Name the specific analyses (e.g., \"monthly revenue trend with MoM growth rates\")\n   - Explain what each analysis will reveal relative to their question\n   - Let the user adjust before you proceed\n\n4. **Execute the analysis** — run the agreed analyses\n   - Summary statistics for numeric columns\n   - Trend identification with direction, magnitude, and acceleration\n   - Comparisons (period-over-period, segment, actual vs. target) as relevant\n   - Distribution and concentration analysis where useful\n   - Correlation spotting between metrics\n   - Cohort analysis when data supports grouping\n\n5. **Translate to insights** — convert numbers into plain-language findings\n   - Lead with what matters, not what was calculated\n   - Rank findings by business impact, not statistical significance\n   - Connect each finding to the user's original question\n   - Flag surprising results and explain why they are surprising\n\n6. **Deliver the report** — structured output following the default format below\n   - Include data quality notes inline\n   - Label confidence on every statistical claim\n   - Suggest follow-up questions the user hasn't asked\n\n7. **Offer next steps** — what deeper analysis could be useful, what data would improve the picture\n\n## Default output structure\n\nUse this structure unless the user clearly wants a different format:\n\n1. **Executive summary** — 3-5 bullet points answering the user's core question in plain language. No jargon. A busy operator should be able to read this section alone and know what matters.\n\n2. **Data quality notes** — what came in, what was cleaned, what to watch out for. Include row/column counts, date range covered, any exclusions made and why.\n\n3. **Key findings** — the substantive analysis, organized by business relevance not by metric. Each finding should follow the pattern:\n   - What the data shows (the fact)\n   - Why it matters (the implication)\n   - How confident we are (the evidence quality)\n\n4. **Trend analysis** — time-series patterns with:\n   - Direction and magnitude of change\n   - Comparison to prior period or baseline\n   - Acceleration or deceleration signals\n   - Seasonal or cyclical patterns if detectable\n\n5. **Comparisons** — if the data supports comparison (segments, periods, targets):\n   - Side-by-side with explicit metrics\n   - Performance gaps highlighted\n   - Context for why gaps exist (if inferrable from data)\n\n6. **Watch items** — things that aren't problems yet but could become problems:\n   - Emerging negative trends\n   - Metrics approaching thresholds\n   - Data quality issues that could mask real signals\n\n7. **Recommended actions** — 3 concrete next steps:\n   - One action justified by the data right now\n   - One thing to monitor or investigate further\n   - One data improvement that would sharpen future analysis\n\n8. **Methodology notes** — what was calculated, how, and what assumptions were made. Brief but sufficient for someone to question the analysis.\n\n## Analysis depth calibration\n\nMatch analysis depth to data quality and volume:\n\n| Data quality | Row count | Depth |\n|---|---|---|\n| Clean, complete | >1,000 | Full analysis with statistical tests, confidence intervals, correlation |\n| Clean, complete | 100-1,000 | Full analysis, note limited sample for statistical claims |\n| Clean, complete | <100 | Summary stats and directional trends only, flag small-sample risk |\n| Moderate gaps | Any | Analyze what's clean, quantify the gap, note impact on conclusions |\n| Poor quality | Any | Data quality report first, limited directional analysis with heavy caveats |\n\nDo not apply sophisticated statistical methods to data that can't support them. 3 months of revenue data does not justify a seasonal decomposition.\n\n## Confidence labeling\n\nEvery analytical claim gets a confidence indicator:\n\n- **High confidence** — large sample, clean data, clear pattern, well-understood metric\n- **Moderate confidence** — adequate sample, minor data issues, pattern present but could shift\n- **Low confidence** — small sample, data quality concerns, pattern is directional at best\n- **Flagged** — interesting signal but insufficient evidence to draw conclusions; noted for monitoring\n\nWhen confidence is low, say what additional data would raise it.\n\n## Number formatting\n\n- Currency: match the user's format or default to $X,XXX.XX\n- Percentages: one decimal place for rates (5.2%), whole numbers for large changes (up 23%)\n- Large numbers: use K/M/B shorthand with one decimal ($1.2M, 45.3K users)\n- Growth rates: always specify the comparison basis (MoM, QoQ, YoY) and the absolute numbers behind the percentage\n- Do not present a percentage without context: \"Revenue grew 15% MoM ($42K to $48.3K)\" not just \"Revenue grew 15%\"\n\n## Handling common business metrics\n\nWhen the user's data contains standard business metrics, calculate them consistently:\n\nRead `references/business-metrics.md` for definitions, formulas, and interpretation guidance for:\n- MRR / ARR and expansion/contraction/churn components\n- Customer churn rate (logo and revenue)\n- CAC, LTV, and LTV:CAC ratio\n- Gross and net margins\n- Growth rates (MoM, QoQ, YoY, CAGR)\n- Unit economics\n\nAlways show the formula used when presenting a calculated metric. Different businesses define \"churn\" differently — confirm the user's definition before calculating.\n\n## Data quality checks\n\nRun these checks on every dataset before analysis:\n\nRead `references/data-quality-checks.md` for the full checklist covering:\n- Completeness (missing values by column, row completeness rate)\n- Consistency (date format uniformity, categorical value normalization)\n- Validity (values within expected ranges, negative amounts where unexpected)\n- Uniqueness (duplicate detection, key column analysis)\n- Timeliness (date range coverage, gap detection)\n- Outlier flagging (statistical and domain-based)\n\nReport data quality findings before analysis results. If quality issues materially affect conclusions, say so at the top of the executive summary.\n\n## Report structure templates\n\nRead `references/report-templates.md` for pre-built structures for common report types:\n- Executive summary report (1-page, for leadership)\n- Detailed analysis report (full findings with methodology)\n- Comparison report (A vs. B with decision framework)\n- Trend report (time-series focused with forecasting context)\n- Health check report (KPI dashboard in text form)\n\nUse the appropriate template when the user's request clearly maps to one. Default to the standard output structure when it doesn't.\n\n## Sparse-data and minimal-signal analysis\n\nWhen the dataset is too small or too noisy for robust analysis:\n\n1. **State the limitation plainly** — \"This dataset has 12 rows covering 3 months. Statistical analysis is limited.\"\n2. **Provide what's possible** — totals, simple averages, directional observations\n3. **Name what would be needed** — \"6+ months of data would allow trend detection; 100+ transactions would support segment analysis\"\n4. **One observation worth monitoring** — the single most interesting signal, clearly labeled as preliminary\n5. **Do not pad** — a short, honest report is better than a long, hedged one\n\n## No-data gate\n\nWhen the user asks for analysis but provides no data:\n\n1. Ask what data they have available and in what format\n2. Suggest the minimum viable dataset for the analysis they want\n3. Offer to help them structure their data for analysis\n4. Provide a sample template they can populate\n\nDo not generate fictional analysis or example reports unless the user explicitly asks for a template or demo.\n\n## Multi-dataset analysis\n\nWhen the user provides multiple related datasets:\n\n- Identify join keys and relationship types before merging\n- Report any orphaned records (rows that don't match across datasets)\n- Be explicit about which dataset each finding comes from\n- Note where merged analysis adds insight vs. where datasets should be analyzed separately\n\n## Boundaries\n\n- **No financial advice.** Analyze data and identify patterns. Do not recommend investments, tax strategies, or financial products.\n- **No projections as fact.** Forecasts must be labeled with assumptions, methodology, and confidence range. \"If current trends continue\" not \"revenue will be.\"\n- **Statistical confidence labeling.** Small samples and high variance get explicit warnings. Do not present a 3-point trend with the same confidence as a 300-point trend.\n- **Inference/fact separation.** Data points are facts. Patterns derived from them are inferences. Recommendations are opinions. Label each.\n- **No private database access without explicit user setup.** Work on data the user provides.\n- **PII detection and exclusion.** Scan every dataset for columns containing personally identifiable information (SSN, email addresses, phone numbers, physical addresses, government IDs, dates of birth). When PII is detected:\n  1. Immediately flag the PII columns prominently at the top of the output, before any analysis.\n  2. Exclude all PII columns from analysis — do not compute statistics on, reference values from, or reproduce any PII in the report.\n  3. Proceed with analysis on non-PII columns only (e.g., purchase totals, visit counts, plan types).\n  4. Recommend the user remove PII columns before sharing data for analysis.\n  5. Never quote, echo, or reference specific PII values (e.g., do not include an SSN in a \"data quality finding\").\n- **No audit-grade output.** Reports are analytical aids, not auditable financial statements.\n- **No data fabrication.** Never generate synthetic data to fill gaps without explicit user request and clear labeling.\n\nFile v0.1.0:README.md\n\n# Data Analysis & Reporting\n\nTurn raw business data into plain-language insights, trend analysis, and actionable reports.\n\n## Status\n\nv0.1.0 — Board-approved for publication. 43/45 QA score across 12 test cases.\n\n## Structure\n\n```\ndata-analysis-reporting/\n├── README.md                          # This file\n├── SKILL.md                           # Skill definition (triggers, workflow, output structure)\n├── data-analysis-reporting-spec.md    # Full spec (purpose, capabilities, boundaries, market context)\n└── references/\n    ├── business-metrics.md            # Standard business metrics: formulas, interpretation, benchmarks\n    ├── data-quality-checks.md         # Pre-analysis data validation checklist\n    ├── report-templates.md            # 5 output templates (executive, detailed, comparison, trend, health check)\n    └── test-prompts.md                # 12 test cases (4 happy, 4 normal, 4 edge)\n```\n\n## Key Design Decisions\n\n- **Clarify first, analyze second.** The skill asks what decision the analysis supports before touching the data.\n- **Data quality before results.** Every analysis starts with a quality check and reports issues upfront.\n- **Insights over numbers.** Output leads with \"what this means\" not \"what the number is.\"\n- **Confidence labeling.** Every claim gets High/Moderate/Low/Flagged confidence.\n- **Honest about limits.** Small samples, poor data quality, and insufficient evidence get clear warnings — not padding.\n\n## Differentiator\n\nFeels like a sharp junior analyst on staff, not a statistics engine. Asks clarifying questions, proposes an analysis plan, and surfaces follow-up questions the user hasn't thought to ask.\n\n## License\n\nPublished by NorthlineAILabs.\n\nFile v0.1.0:_meta.json\n\n{\n  \"ownerId\": \"kn7bz9d5bwyvakz2t7hqjbeyf183qcc6\",\n  \"slug\": \"data-analysis-reporting\",\n  \"version\": \"0.1.0\",\n  \"publishedAt\": 1775603825424\n}\n\nFile v0.1.0:references/business-metrics.md\n\n# Business Metrics Reference\n\nStandard business metrics, how to calculate them, and how to interpret the results. Use this reference when the user's data contains these metrics or when calculating them from raw data.\n\n## Revenue Metrics\n\n### MRR (Monthly Recurring Revenue)\n- **Formula:** Sum of all active subscription revenue normalized to monthly\n- **Components:**\n  - **New MRR** — revenue from new customers acquired this month\n  - **Expansion MRR** — revenue increase from existing customers (upgrades, add-ons)\n  - **Contraction MRR** — revenue decrease from existing customers (downgrades)\n  - **Churned MRR** — revenue lost from customers who cancelled\n  - **Net New MRR** = New + Expansion - Contraction - Churned\n- **Interpretation:**\n  - Healthy SaaS: Net New MRR is positive and growing\n  - Warning: Churned MRR exceeding New MRR for 2+ consecutive months\n  - Context: Always report components, not just the total — a flat MRR can mask high churn offset by high acquisition\n\n### ARR (Annual Recurring Revenue)\n- **Formula:** MRR x 12\n- **Caution:** Only valid when MRR is relatively stable month-to-month. Do not annualize a spike month.\n- **Use:** Standard for SaaS valuation, fundraising metrics, and annual planning\n\n### Revenue Growth Rate\n- **MoM:** (This month - Last month) / Last month x 100\n- **QoQ:** (This quarter - Last quarter) / Last quarter x 100\n- **YoY:** (This period - Same period last year) / Same period last year x 100\n- **CAGR:** (End value / Start value)^(1/years) - 1\n- **Best practice:** Always show absolute numbers alongside percentages. A 50% MoM growth from $2K to $3K is very different from $200K to $300K.\n\n### ARPU (Average Revenue Per User/Account)\n- **Formula:** Total Revenue / Total Customers (for the same period)\n- **Cross-validation:** When a dataset provides Revenue, Customers, and ARPU columns, verify that Revenue / Customers = stated ARPU. Flag any mismatches before proceeding — contradictory data undermines all downstream analysis.\n- **Variants:**\n  - **ARPU** — per user, common in consumer products\n  - **ARPA** — per account, common in B2B where one account has multiple users\n- **Interpretation:**\n  - Rising ARPU with stable customer count = healthy expansion / upsell\n  - Falling ARPU with growing customer count = adding lower-value customers (may be fine strategically)\n  - Flat ARPU = stable unit economics, look at volume for growth signal\n- **Caution:** ARPU is an average — high variance means the average may be misleading. Report median alongside mean when possible.\n\n## Customer Metrics\n\n### Customer Churn Rate\n- **Logo churn:** Customers lost / Customers at start of period x 100\n- **Revenue churn (gross):** Churned MRR / MRR at start of period x 100\n- **Revenue churn (net):** (Churned MRR - Expansion MRR) / MRR at start of period x 100\n- **Important:** Ask the user which definition they use. \"Churn\" means different things to different businesses.\n- **Benchmarks (SaaS):**\n  - SMB: 3-7% monthly logo churn is common\n  - Mid-market: 1-3% monthly\n  - Enterprise: <1% monthly\n  - Net negative revenue churn is the gold standard (expansion exceeds churn)\n\n### Retention Rate\n- **Formula:** 1 - Churn Rate (for the same period and definition)\n- **Cohort retention:** Track what percentage of a signup cohort remains active after N months\n- **Dollar retention (NDR/NRR):** Revenue from a cohort after N months / Revenue from that cohort at start. >100% means expansion exceeds churn within the cohort.\n\n### CAC (Customer Acquisition Cost)\n- **Formula:** Total sales and marketing spend / New customers acquired (in the same period)\n- **Variants:**\n  - **Blended CAC** — all spend / all new customers\n  - **Paid CAC** — only paid channel spend / customers from paid channels\n  - **Fully loaded CAC** — includes salaries, tools, overhead allocated to acquisition\n- **Caution:** CAC is only meaningful when compared to LTV and payback period. A high CAC is fine if LTV is proportionally high.\n\n### LTV (Customer Lifetime Value)\n- **Simple formula:** Average revenue per customer per month / Monthly churn rate\n- **Better formula:** Average revenue per customer per month x Gross margin % / Monthly churn rate\n- **Caution:** LTV assumes stable churn and ARPU, which is rarely true for fast-growing companies. Label assumptions.\n\n### LTV:CAC Ratio\n- **Formula:** LTV / CAC\n- **Benchmarks:**\n  - <1:1 — losing money on every customer (unsustainable)\n  - 1:1 to 3:1 — breakeven to marginal (may be okay in land-and-expand models)\n  - 3:1 to 5:1 — healthy range for most SaaS\n  - >5:1 — either very efficient or under-investing in growth\n- **Always pair with:** CAC payback period (months to recover CAC from gross margin)\n\n## Profitability Metrics\n\n### Gross Margin\n- **Formula:** (Revenue - COGS) / Revenue x 100\n- **COGS for SaaS:** hosting, infrastructure, customer support, onboarding costs directly tied to service delivery\n- **COGS for product:** raw materials, manufacturing, direct labor, shipping\n- **Benchmarks:** SaaS typically 70-85%; physical products 30-60%; services 40-60%\n\n### Net Margin\n- **Formula:** Net income / Revenue x 100\n- **Includes:** all operating expenses, taxes, interest, depreciation\n- **Context:** Negative net margin is normal for growth-stage companies. Flag when burn rate is relevant.\n\n### Operating Margin\n- **Formula:** Operating income / Revenue x 100\n- **Use:** Better than net margin for comparing operational efficiency (excludes tax and interest effects)\n\n### Contribution Margin\n- **Formula:** (Revenue - Variable costs) / Revenue x 100\n- **Use:** Per-unit or per-customer profitability before fixed costs. Critical for pricing and unit economics.\n\n## Efficiency Metrics\n\n### Burn Rate\n- **Gross burn:** Total monthly cash outflow\n- **Net burn:** Total monthly cash outflow - Total monthly cash inflow\n- **Runway:** Cash on hand / Net monthly burn = months until cash runs out\n- **Warning thresholds:** <6 months runway is urgent; <12 months requires active fundraising or path-to-profitability plan\n\n### Rule of 40\n- **Formula:** Revenue growth rate (%) + Profit margin (%) >= 40\n- **Use:** SaaS benchmark balancing growth and profitability\n- **Context:** Early-stage companies should skew toward growth; mature companies should skew toward margin\n\n### Magic Number (SaaS Sales Efficiency)\n- **Formula:** Net New ARR this quarter / Sales & marketing spend last quarter\n- **Benchmarks:** <0.5 = inefficient; 0.5-1.0 = decent; >1.0 = strong signal to invest more in S&M\n\n## Calculation Best Practices\n\n1. **State the formula** every time you calculate a metric. Different businesses define metrics differently.\n2. **Confirm definitions** with the user before calculating churn, CAC, or LTV — these are the most commonly miscalculated metrics.\n3. **Show absolute numbers** alongside percentages. Percentages without context are noise.\n4. **Specify the time period** for every rate or ratio. Monthly churn and annual churn are very different numbers.\n5. **Flag when sample size is too small** to calculate meaningful rates. 5 churned customers out of 20 is a 25% churn rate, but the confidence interval is enormous.\n6. **Do not mix time periods.** Monthly CAC with annual LTV produces a meaningless ratio. Normalize first.\n7. **Label benchmarks as context, not targets.** Industry benchmarks vary by segment, stage, and geography.\n\nFile v0.1.0:references/data-quality-checks.md\n\n# Data Quality Checks\n\nRun these checks on every dataset before analysis. Report findings before results. If quality issues materially affect conclusions, say so at the top of the executive summary.\n\n## 1. Completeness\n\n### Missing values\n- Count missing/null/empty values per column\n- Calculate completeness rate per column: (non-null rows / total rows) x 100\n- Flag columns below 90% completeness\n- Distinguish between: truly missing, intentionally blank (e.g., optional fields), and encoded as zero/placeholder\n\n### Row completeness\n- Calculate per-row completeness: (non-null fields / total fields) x 100\n- Flag rows with <50% completeness for potential exclusion\n- Report: \"X of Y rows are fully complete; Z rows have significant gaps\"\n\n### Date coverage\n- Identify the date range in the data\n- Check for gaps (missing days, weeks, months in a time series)\n- Report: \"Data covers [start] to [end] with [N] gaps in coverage\"\n\n## 2. Consistency\n\n### Date formats\n- Check for mixed date formats (MM/DD/YYYY vs. DD/MM/YYYY vs. YYYY-MM-DD)\n- Attempt auto-detection; if ambiguous (e.g., 03/04/2025), ask the user\n- Normalize to ISO 8601 (YYYY-MM-DD) for analysis\n\n### Categorical values\n- Check for variant spellings, casing differences, trailing spaces\n- Examples: \"Active\" vs. \"active\" vs. \"ACTIVE\" vs. \" Active\"\n- Report unique values per categorical column and flag likely duplicates\n- Normalize after confirming with user if ambiguous\n\n### Unit consistency\n- Check for mixed units in the same column (e.g., $ and EUR, kg and lbs)\n- Check for mixed scales (e.g., some values in thousands, others in actuals)\n- Flag and ask before normalizing\n\n### Naming consistency\n- Check column headers for typos, inconsistent naming conventions\n- Report any columns that appear to contain the same data under different names\n\n## 3. Validity\n\n### Range checks\n- Numeric columns: flag values outside expected ranges\n  - Negative values where only positive expected (e.g., revenue, counts)\n  - Zero values where zero is unusual (e.g., price, quantity)\n  - Values orders of magnitude above/below the column median\n- Date columns: flag future dates (if unexpected), dates before a reasonable start\n- Percentage columns: flag values >100% or <0% (unless explicitly allowed)\n\n### Type validation\n- Check that numeric columns contain only numbers (no text mixed in)\n- Check that date columns parse correctly\n- Flag columns where >5% of values fail type validation\n\n### Business rule validation\n- Revenue = Quantity x Price (if all three columns exist)\n- End date >= Start date\n- Running totals match sum of components\n- Report any violations found\n\n## 4. Uniqueness\n\n### Duplicate detection\n- Check for exact duplicate rows\n- Check for near-duplicates (same key fields, different minor fields)\n- Report: \"Found X exact duplicates and Y potential near-duplicates\"\n\n### Key column analysis\n- Identify likely primary key columns (high cardinality, no nulls)\n- Check uniqueness of suspected key columns\n- Flag if a supposed unique identifier has duplicates\n\n## 5. Timeliness\n\n### Data freshness\n- Note the most recent date in the dataset\n- If more than 30 days old, add a staleness warning\n- If more than 90 days old, flag as historical baseline only\n\n### Update frequency\n- Infer the expected update frequency from the data (daily, weekly, monthly)\n- Check if the most recent period is complete or partial\n- Flag partial periods: \"March 2025 data appears incomplete — only 15 of expected ~30 days present\"\n\n## 6. Outlier Detection\n\n### Statistical outliers\n- For numeric columns: flag values beyond 3 standard deviations from the mean\n- For skewed distributions: use IQR method (below Q1 - 1.5*IQR or above Q3 + 1.5*IQR)\n- Report outliers but do not auto-exclude — ask the user if they represent real events or errors\n\n### Domain-based outliers\n- Revenue: single transaction >10x the median transaction value\n- Dates: entries on weekends/holidays if the business is weekday-only\n- Counts: zero-activity periods in otherwise active time series\n- Growth rates: single-period spikes >3x the typical rate\n\n## 7. Privacy / PII Scan\n\nRun this check before any analysis. PII detection takes priority over all other quality checks.\n\n### Column-level PII detection\n- **Email addresses:** columns matching `*@*.*` patterns\n- **Phone numbers:** columns matching common phone formats (XXX-XXX-XXXX, (XXX) XXX-XXXX, +X XXXXXXXXXX)\n- **SSN / government IDs:** columns matching XXX-XX-XXXX or similar national ID patterns\n- **Physical addresses:** columns with street number + street name patterns, or labeled \"address\"\n- **Dates of birth:** columns labeled \"DOB\", \"birth\", \"birthday\" or containing dates clearly in a birth-year range\n- **Names combined with other PII:** a \"Name\" column alone is low risk, but Name + any of the above = PII dataset\n\n### When PII is detected\n1. Flag immediately and prominently — before any data quality or analysis output\n2. List which columns contain PII and what type\n3. Exclude all PII columns from analysis entirely\n4. Do not reproduce, quote, or reference specific PII values anywhere in the output\n5. Proceed with analysis on remaining non-PII columns only\n6. Recommend: \"Remove PII columns before sharing data for analysis\"\n\n### Decision rule\n- **Any PII column detected** → exclude from analysis, flag at top of report, proceed with non-PII columns\n- **All columns are PII** → cannot analyze; ask user to provide non-PII data\n\n## Reporting Format\n\nPresent data quality findings in this structure:\n\n```\n## Data Quality Summary\n\n**Dataset:** [name/description]\n**Rows:** [count] | **Columns:** [count] | **Date range:** [start] to [end]\n**Overall quality:** [Good / Moderate / Poor]\n\n### Issues Found\n- [Issue 1]: [description and impact on analysis]\n- [Issue 2]: [description and impact on analysis]\n\n### Actions Taken\n- [Normalization or cleaning step 1]\n- [Normalization or cleaning step 2]\n\n### Caveats for Analysis\n- [How remaining issues affect conclusions]\n```\n\n## Decision Rules\n\n- **Completeness <80%** on a key analysis column → warn the user, proceed with caveats\n- **Completeness <50%** on a key analysis column → recommend the user fix the data first\n- **Duplicate rate >5%** → deduplicate before analysis, report what was removed\n- **Mixed types >10%** in a column → column may need to be split or cleaned before use\n- **Date gaps >20%** of expected periods → time-series trend analysis unreliable, use point-in-time comparisons instead\n\nFile v0.1.0:references/report-templates.md\n\n# Report Templates\n\nPre-built structures for common report types. Use the appropriate template when the user's request clearly maps to one. Default to the standard output structure in SKILL.md when it doesn't.\n\n## 1. Executive Summary Report\n\n**Use when:** The audience is leadership, time is short, decision needs to be made.\n**Length:** 1 page equivalent (~400-600 words)\n\n```\n# [Report Title] — Executive Summary\n\n**Period:** [date range]\n**Prepared for:** [audience]\n**Data source:** [description]\n\n## Bottom Line\n\n[2-3 sentences: the single most important takeaway and what it means for the business]\n\n## Key Numbers\n\n| Metric | Current | Prior Period | Change |\n|--------|---------|-------------|--------|\n| [metric 1] | [value] | [value] | [+/-X%] |\n| [metric 2] | [value] | [value] | [+/-X%] |\n| [metric 3] | [value] | [value] | [+/-X%] |\n\n## What's Working\n- [1-2 bullets on positive trends]\n\n## What Needs Attention\n- [1-2 bullets on concerning trends or risks]\n\n## Recommended Actions\n1. [Action with clear owner and timeline]\n2. [Action with clear owner and timeline]\n3. [Action with clear owner and timeline]\n\n## Data Notes\n[One line on data quality, completeness, and any caveats]\n```\n\n## 2. Detailed Analysis Report\n\n**Use when:** The user needs full findings with methodology for review or sharing.\n**Length:** 2-4 page equivalent\n\n```\n# [Report Title] — Detailed Analysis\n\n**Period:** [date range]\n**Prepared for:** [audience]\n**Data source:** [description]\n**Analysis date:** [date]\n\n## Executive Summary\n[3-5 bullet points — the findings that matter most]\n\n## Data Quality Notes\n[Completeness, cleaning steps, caveats — see data-quality-checks.md]\n\n## Finding 1: [Title]\n**What the data shows:** [fact with numbers]\n**Why it matters:** [business implication]\n**Confidence:** [High/Moderate/Low]\n**Detail:** [supporting analysis, breakdowns, context]\n\n## Finding 2: [Title]\n[Same structure]\n\n## Finding 3: [Title]\n[Same structure]\n\n## Trend Analysis\n[Time-series observations with comparison baselines]\n- Direction and magnitude\n- Acceleration/deceleration\n- Seasonal patterns if present\n\n## Watch Items\n- [Emerging concern 1 — what to monitor and when to act]\n- [Emerging concern 2]\n\n## Recommended Actions\n1. **Now:** [Action justified by current data]\n2. **Monitor:** [What to track and what threshold triggers action]\n3. **Improve:** [Data or process improvement for better future analysis]\n\n## Methodology\n- Metrics calculated: [list with formulas]\n- Time periods compared: [description]\n- Exclusions: [what was excluded and why]\n- Assumptions: [any assumptions made]\n```\n\n## 3. Comparison Report\n\n**Use when:** The user wants to compare two or more things (periods, segments, products, options).\n**Length:** 1-3 pages depending on number of comparisons\n\n```\n# [A] vs. [B] — Comparison Analysis\n\n**Period:** [date range]\n**Comparison basis:** [what's being compared and why]\n\n## Summary Verdict\n[Which option/segment/period performs better overall, and the key reason why]\n\n## Side-by-Side Comparison\n\n| Dimension | [A] | [B] | Delta | Winner |\n|-----------|-----|-----|-------|--------|\n| [metric 1] | [value] | [value] | [diff] | [A/B/Tie] |\n| [metric 2] | [value] | [value] | [diff] | [A/B/Tie] |\n| [metric 3] | [value] | [value] | [diff] | [A/B/Tie] |\n\n## Where [A] Outperforms\n- [Specific area with numbers and context]\n\n## Where [B] Outperforms\n- [Specific area with numbers and context]\n\n## Context and Caveats\n- [Factors that explain differences — seasonality, sample size, external events]\n- [Evidence quality differences between options]\n\n## Recommendation\n[What to do based on the comparison, with confidence level]\n```\n\n## 4. Trend Report\n\n**Use when:** The user wants to understand how something is changing over time.\n**Length:** 1-2 pages\n\n```\n# [Metric/Area] Trend Analysis\n\n**Period:** [full date range analyzed]\n**Granularity:** [daily/weekly/monthly]\n**Baseline:** [comparison reference point]\n\n## Trend Summary\n[2-3 sentences: what's the overall direction, how fast, and is it accelerating or decelerating?]\n\n## Period-by-Period Breakdown\n\n| Period | Value | Change | Growth Rate | Notes |\n|--------|-------|--------|-------------|-------|\n| [period 1] | [value] | — | — | Baseline |\n| [period 2] | [value] | [+/-] | [%] | |\n| [period 3] | [value] | [+/-] | [%] | [notable event] |\n\n## Key Observations\n1. **[Observation]** — [what the trend shows and confidence level]\n2. **[Observation]** — [what the trend shows and confidence level]\n\n## Seasonal/Cyclical Patterns\n[If detectable with available data; if not, state minimum data needed to detect]\n\n## Inflection Points\n[Periods where the trend changed direction or rate, with possible explanations]\n\n## Projection Context\n[If the user asks \"where is this going\": state assumptions, range, and confidence. Never present as fact.]\n\n## What Would Change This Trend\n- Upside scenario: [what would accelerate positive trend]\n- Downside scenario: [what would reverse or worsen the trend]\n```\n\n## 5. Health Check Report\n\n**Use when:** The user wants a regular KPI review or business health snapshot.\n**Length:** 1-2 pages\n\n```\n# Business Health Check — [Period]\n\n**Status:** [Green/Yellow/Red overall assessment]\n\n## Scorecard\n\n| Category | Metric | Value | Target | Status |\n|----------|--------|-------|--------|--------|\n| Revenue | MRR | [value] | [target] | [G/Y/R] |\n| Revenue | MoM Growth | [value] | [target] | [G/Y/R] |\n| Customers | Total Active | [value] | [target] | [G/Y/R] |\n| Customers | Churn Rate | [value] | [target] | [G/Y/R] |\n| Efficiency | CAC | [value] | [target] | [G/Y/R] |\n| Efficiency | LTV:CAC | [value] | [target] | [G/Y/R] |\n\n## What's Healthy\n- [1-3 bullets on metrics at or above target]\n\n## What Needs Attention\n- [1-3 bullets on metrics below target or trending negatively]\n\n## Changes Since Last Period\n- [What improved]\n- [What declined]\n- [What's new or noteworthy]\n\n## Recommended Focus Areas\n1. [Highest priority action]\n2. [Second priority]\n3. [Third priority]\n```\n\n## Template Selection Guide\n\n| User request contains... | Use template |\n|--------------------------|-------------|\n| \"executive summary\", \"for leadership\", \"quick overview\" | Executive Summary |\n| \"full analysis\", \"detailed report\", \"deep dive\" | Detailed Analysis |\n| \"compare\", \"vs.\", \"which is better\", \"A or B\" | Comparison |\n| \"trend\", \"over time\", \"how is X changing\" | Trend |\n| \"health check\", \"KPI review\", \"how are we doing\" | Health Check |\n| Unclear or mixed | Default structure from SKILL.md |\n\nFile v0.1.0:references/test-prompts.md\n\n# Test Prompts\n\n## Happy Path (4 tests — clean data, clear asks, expected workflows)\n\n### HP-1: SaaS MRR analysis from CSV\n```\nHere's our monthly revenue data for the last 12 months:\n\nMonth,MRR,New Customers,Churned Customers,Total Customers\n2025-04,$12,400,8,2,45\n2025-05,$13,100,10,3,52\n2025-06,$14,800,12,2,62\n2025-07,$15,200,9,4,67\n2025-08,$16,500,11,3,75\n2025-09,$17,900,14,2,87\n2025-10,$18,400,10,5,92\n2025-11,$19,100,12,3,101\n2025-12,$20,500,15,4,112\n2026-01,$21,800,14,2,124\n2026-02,$22,600,11,3,132\n2026-03,$23,900,16,5,143\n\nCan you analyze our growth and tell me how we're doing? I need to present this to our investors next week.\n```\n**Expected:** Triggers executive summary template. Calculates MoM growth, churn rate, net customer adds. Asks about ARPU trends. Presents investor-ready summary with clear growth narrative and honest risk flags.\n\n### HP-2: Period-over-period comparison\n```\nCompare our Q1 2025 vs Q1 2026 sales performance:\n\nQ1 2025:\nRegion,Revenue,Deals Closed,Avg Deal Size\nNorth,$145,000,12,$12,083\nSouth,$98,000,8,$12,250\nEast,$210,000,15,$14,000\nWest,$67,000,5,$13,400\n\nQ1 2026:\nRegion,Revenue,Deals Closed,Avg Deal Size\nNorth,$178,000,14,$12,714\nSouth,$112,000,10,$11,200\nEast,$195,000,13,$15,000\nWest,$89,000,7,$12,714\n```\n**Expected:** Triggers comparison template. Shows YoY growth by region. Identifies East as the only declining region. Highlights West as fastest-growing. Calculates total portfolio change. Notes South's declining deal size despite volume growth.\n\n### HP-3: E-commerce product performance\n```\nHere's our product sales data for last month. Which products should we focus on?\n\nProduct,Units Sold,Revenue,Returns,Cost per Unit\nWidget A,342,$17,100,12,$22\nWidget B,89,$13,350,3,$68\nWidget C,1205,$24,100,45,$8\nGadget X,56,$28,000,8,$210\nGadget Y,178,$8,900,22,$18\nService Plan,234,$46,800,0,$0\n```\n**Expected:** Calculates margin per product, return rate, revenue concentration. Identifies Service Plan as highest margin. Flags Gadget Y's high return rate (12.4%). Notes revenue concentration risk if Gadget X or Service Plan underperform. Asks about customer acquisition costs per product line.\n\n### HP-4: Health check with targets\n```\nHere are our KPIs for March 2026 with targets. Give me a health check.\n\nMetric,Actual,Target\nMRR,$45,200,$50,000\nMoM Growth,3.2%,5%\nActive Users,1,245,1,500\nChurn Rate,4.8%,3%\nNPS Score,42,50\nSupport Tickets,312,<200\nAvg Response Time,4.2 hrs,<2 hrs\nCAC,$185,$150\nLTV:CAC,4.1,>3\nRunway (months),14,>12\n```\n**Expected:** Triggers health check template. Green on LTV:CAC and runway. Red on churn, support metrics, and active users. Yellow on MRR and growth (below target but not critical). Identifies support quality as the likely driver of churn. Recommends focusing on support before growth.\n\n## Normal Path (4 tests — reasonable asks with some ambiguity or complexity)\n\n### NP-1: Ambiguous request needing clarification\n```\nI have sales data. Can you analyze it?\n\nDate,Amount,Category,Rep\n2026-01-05,1200,Enterprise,Sarah\n2026-01-08,450,SMB,Mike\n2026-01-12,8500,Enterprise,Sarah\n2026-01-15,320,SMB,Alex\n2026-01-19,2100,Mid-Market,Sarah\n2026-01-22,175,SMB,Mike\n2026-01-25,5400,Enterprise,Alex\n2026-01-28,890,Mid-Market,Mike\n```\n**Expected:** Asks clarifying questions before analysis: \"What decision does this support?\", \"Is this one month or a sample?\", \"Do you want per-rep performance, per-category analysis, or both?\" Does not just dump summary stats.\n\n### NP-2: Data with moderate quality issues\n```\nAnalyze our customer data:\n\nID,Name,Signup Date,Plan,MRR,Last Active\n1,Acme Corp,2025-01-15,Pro,$499,2026-03-28\n2,Beta LLC,01/20/2025,Basic,$99,2026-03-25\n3,Gamma Inc,2025-02-01,Pro,$499,\n4,Delta Co,2025-03-10,Enterprise,$1499,2026-03-30\n5,Epsilon Ltd,,Basic,$99,2026-02-15\n6,Zeta Corp,2025-04-22,Pro,$0,2026-03-29\n7,Eta Inc,2025-05-01,Basic,$99,2025-11-20\n8,Theta LLC,2025-06-15,Pro,$499,2026-03-30\n9,Iota Corp,2025-07-01,,,$2026-03-28\n10,Kappa Co,2025-08-12,Enterprise,$1499,2026-03-27\n```\n**Expected:** Reports data quality issues first: mixed date formats (rows 1 vs 2), missing signup date (row 5), missing last active (row 3), missing plan and MRR (row 9), $0 MRR on Pro plan (row 6 — likely error or cancelled), potentially churned customer (row 7 — last active 4+ months ago). Proceeds with analysis noting caveats.\n\n### NP-3: Multi-dataset join request\n```\nI have two tables. Can you combine them and analyze?\n\nOrders:\nOrderID,CustomerID,Date,Amount\n1001,C1,2026-03-01,$250\n1002,C2,2026-03-03,$180\n1003,C1,2026-03-07,$320\n1004,C3,2026-03-10,$95\n1005,C4,2026-03-15,$510\n1006,C2,2026-03-18,$180\n1007,C5,2026-03-22,$750\n\nCustomers:\nCustomerID,Name,Segment,JoinDate\nC1,Alpine Ltd,Enterprise,2025-06-01\nC2,Brook Co,SMB,2025-09-15\nC3,Cedar Inc,SMB,2026-01-10\nC6,Drift LLC,Enterprise,2025-03-20\n```\n**Expected:** Identifies join key (CustomerID). Reports orphaned records: C4 and C5 in Orders have no Customer match; C6 in Customers has no Orders. Analyzes matched data by segment. Notes that 2 of 7 orders (28.6%) can't be enriched with customer data — flags this as a data integrity issue.\n\n### NP-4: Trend analysis with limited data\n```\nWe just started tracking our weekly active users. Is there a trend?\n\nWeek,WAU\nW1,124\nW2,131\nW3,118\nW4,142\nW5,155\nW6,148\nW7,161\n```\n**Expected:** Provides directional trend (generally upward, ~4.5% avg weekly growth). Warns that 7 weeks is too short for statistical confidence or seasonal detection. Notes the W3 dip and W4-W5 recovery. Does NOT attempt a forecast or apply seasonal decomposition. Suggests minimum 12-16 weeks for reliable trend analysis.\n\n## Edge Cases (4 tests — broken data, unreasonable asks, boundary conditions)\n\n### EC-1: No data provided\n```\nCan you analyze our revenue trends and tell me what we should do about our declining customer retention?\n```\n**Expected:** Triggers no-data gate. Does NOT generate fictional analysis. Asks what data they have available, suggests minimum viable dataset (monthly revenue, customer counts, churn events over 6+ months), offers a template they can fill in.\n\n### EC-2: Tiny dataset with big ask\n```\nHere's our data. Build me a full business analysis with forecasts.\n\nMonth,Revenue\nJan,$5,000\nFeb,$5,200\nMar,$4,800\n```\n**Expected:** Provides basic summary (average ~$5,000/mo, slight variance). Firmly declines forecasting from 3 data points. Explains what data would be needed for the full analysis they want. Does not pad the output to seem comprehensive.\n\n### EC-3: Data with PII\n```\nAnalyze customer spending patterns:\n\nName,Email,Phone,SSN,Purchase Total,Visits\nJohn Smith,john@email.com,555-0123,123-45-6789,$2,450,12\nJane Doe,jane@email.com,555-0456,987-65-4321,$1,890,8\nBob Johnson,bob@email.com,555-0789,456-78-9012,$3,200,15\n```\n**Expected:** Immediately flags PII (email, phone, SSN). Recommends removing PII columns before analysis. Proceeds with analysis on non-PII columns (Purchase Total, Visits) only. Notes the PII warning prominently. Does NOT reproduce or reference the SSN values in the output.\n\n### EC-4: Contradictory or impossible data\n```\nHere's our quarterly metrics:\n\nQuarter,Revenue,Customers,ARPU\nQ1,$100,000,500,$200\nQ2,$120,000,480,$300\nQ3,$95,000,520,$250\nQ4,$150,000,400,$500\n```\n**Expected:** Flags the math inconsistency: Revenue / Customers ≠ stated ARPU in any quarter (Q1: $100K/500=$200 checks out, Q2: $120K/480=$250 not $300, Q3: $95K/520=$183 not $250, Q4: $150K/400=$375 not $500). Asks user to clarify which numbers are correct before proceeding. Does not silently use contradictory data.\n\nFile v0.1.0:data-analysis-reporting-spec.md\n\n# Data Analysis & Reporting — Skill Spec v0.1\n\n## Purpose\n\nTurn raw business data (CSV, SQLite, spreadsheets, pasted tables) into clear analytical summaries, trend analysis, and actionable reports for small business operators, analysts, and decision-makers.\n\nThe skill should feel like having a sharp junior analyst on staff — it asks clarifying questions about what the user actually needs to know, doesn't just dump statistics. Output is plain-language insights first, numbers second.\n\n## Target Users\n\n- **Small business owners** — need to understand their numbers without a BI tool or analyst on staff\n- **Operations managers** — need regular reporting on KPIs without building dashboards\n- **Solo consultants / freelancers** — need to analyze client data and produce professional reports\n- **Early-stage startup founders** — need to track metrics and communicate them to investors\n- **Non-technical analysts** — need data summaries without writing SQL or Python\n\n## Core Capabilities (Free Tier)\n\n### Data Ingestion\n- CSV file parsing and structure detection\n- Pasted tabular data (markdown tables, tab-separated, pipe-delimited)\n- SQLite database querying\n- Simple spreadsheet-style data (rows and columns with headers)\n- Auto-detect column types (numeric, date, categorical, text)\n- Handle common formatting issues (mixed date formats, currency symbols, percentage signs, thousand separators)\n\n### Analysis\n- **Summary statistics** — mean, median, mode, min, max, standard deviation, percentiles for numeric columns\n- **Trend identification** — direction, magnitude, and acceleration of change over time periods\n- **Comparison analysis** — period-over-period (MoM, QoQ, YoY), segment-by-segment, actual vs. target\n- **Distribution analysis** — histograms, concentration (top N contributors), outlier flagging\n- **Correlation spotting** — identify which metrics move together and which diverge\n- **Cohort analysis** — group-based performance tracking when data supports it\n\n### Output\n- Plain-language insight summaries (lead with \"what this means\" not \"what the number is\")\n- Markdown-formatted reports suitable for email, Slack, or documentation\n- Structured sections: executive summary, detailed findings, methodology notes\n- Data quality warnings inline where relevant\n- Explicit confidence indicators on statistical claims\n\n### Interaction Model\n- Ask clarifying questions before diving into analysis (\"What decision are you trying to make?\" / \"What time period matters most?\" / \"Who is the audience for this report?\")\n- Propose an analysis plan before executing\n- Offer to drill deeper on interesting findings\n- Suggest follow-up questions the user hasn't asked yet\n\n## Premium Capabilities (Future Roadmap)\n\n### Connectors\n- PostgreSQL / MySQL direct query\n- Google Sheets API integration\n- Airtable read access\n- REST API data pull (with user-provided endpoints)\n\n### Automation\n- Scheduled report generation (daily/weekly/monthly)\n- Threshold-based alerts (\"notify me when churn exceeds 5%\")\n- Report templates that auto-populate with fresh data\n\n### Output Formats\n- PDF export with charts\n- Dashboard template generation (HTML)\n- Slack/email delivery integration\n- Presentation-ready slide summaries\n\n### Advanced Analysis\n- Custom KPI tracking with goal-setting\n- Forecasting (time-series extrapolation with confidence intervals)\n- Anomaly detection and root-cause suggestions\n- Multi-dataset joins and cross-referencing\n\n## Boundaries\n\n- **No financial advice.** The skill analyzes data and identifies patterns. It does not recommend investment decisions, tax strategies, or financial products.\n- **No forward projections presented as fact.** Any extrapolation or forecast must be clearly labeled with assumptions, methodology, and confidence range. Use language like \"if current trends continue\" not \"revenue will be.\"\n- **Statistical confidence labeling.** When sample sizes are small or variance is high, say so explicitly. Do not present a trend from 3 data points with the same confidence as one from 300.\n- **Inference/fact separation.** Same discipline as our other skills — confirmed data points are facts, patterns derived from them are inferences, and recommendations are opinions. Label each.\n- **No access to private databases without explicit user setup.** The skill works on data the user provides. It does not connect to external systems without clear user authorization and configuration.\n- **No PII handling guarantees.** If user data contains PII, the skill should note this and recommend the user review their data sharing practices. The skill does not store or transmit data beyond the conversation.\n- **No audit-grade output.** Reports are analytical aids, not auditable financial statements. Label accordingly.\n\n## Differentiation\n\n### vs. ChatGPT / generic LLM data analysis\n- Structured workflow (clarify -> plan -> analyze -> report) instead of one-shot\n- Consistent output format optimized for business operators\n- Built-in business metric knowledge (knows what MRR, churn, CAC mean and how to calculate them)\n- Data quality checks before analysis, not after\n- Plain-language insights prioritized over raw numbers\n\n### vs. BI tools (Metabase, Looker, Tableau)\n- Zero setup — paste data and get insights\n- No SQL or dashboard-building required\n- Natural language interaction\n- Lower cost, lower complexity\n- Trade-off: less visual, less real-time, less connected\n\n### vs. spreadsheet analysis\n- Automated insight generation (doesn't just calculate, interprets)\n- Consistent report structure\n- Suggests analyses the user didn't think to run\n- Trade-off: less granular control, can't replace a power Excel user for custom modeling\n\n## Market Context\n\n- Only 28 skills in \"Data & Analytics\" category on ClawHub out of 5,211 vetted skills\n- Every SMB needs reporting but most can't afford BI tools ($50-500/mo) or analyst headcount\n- The gap is not \"calculate my average\" — it's \"tell me what matters in my data and what I should do about it\"\n- Adjacent competition is weak: most existing skills are either too technical (aimed at data engineers) or too shallow (just summary stats)\n\n## Success Metrics (Post-Launch)\n\n- Activation: >40% of users who trigger the skill complete at least one full analysis cycle\n- Retention: >25% return within 14 days for another analysis\n- Quality: <10% of reports generate follow-up complaints about accuracy or usefulness\n- Revenue signal: >5% of free-tier users inquire about premium features within 30 days\n\nFile v0.1.0:skill-card.md\n\n## Description:\n\nTurn raw business data (CSV, SQLite, spreadsheets, pasted tables) into clear analytical summaries, trend analysis, and actionable reports for small business operators, analysts, and decision-makers.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[gitcanadabrett](https://clawhub.ai/user/gitcanadabrett)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nExternal users such as small business owners, operations managers, startup founders, consultants, and non-technical analysts use this skill to turn user-provided business data into plain-language reports, trend analysis, KPI comparisons, and recommended next steps.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Business datasets may contain unnecessary personally identifiable information.\n\nMitigation: Share only the data needed for analysis, review datasets for PII before use, and follow the skill's PII flagging and exclusion workflow.\n\nRisk: Reports may be mistaken for financial, tax, investment, or audit-grade advice.\n\nMitigation: Treat outputs as analytical guidance, verify material conclusions with qualified reviewers, and keep forecasts and recommendations clearly labeled as assumptions-based.\n\nRisk: Private database access could expose data if credentials or connections are provided unintentionally.\n\nMitigation: Confirm any database access is intentionally configured and prefer sanitized exports or limited-scope datasets when possible.\n\n## Reference(s):\n\n- [ClawHub Skill Page](https://clawhub.ai/gitcanadabrett/skills/data-analysis-reporting)\n- [Business Metrics Reference](references/business-metrics.md)\n- [Data Quality Checks](references/data-quality-checks.md)\n- [Report Templates](references/report-templates.md)\n- [Test Prompts](references/test-prompts.md)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, guidance]\n\n**Output Format:** [Markdown reports with executive summaries, data quality notes, findings, trend analysis, confidence labels, methodology notes, and recommended actions.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Outputs are based on user-provided data and should label confidence, assumptions, limitations, and data quality concerns.]\n\n## Skill Version(s):\n\n0.1.0 (source: server release evidence and target metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.","readmeExcerpt":"Skill: Data Analysis Reporting Owner: gitcanadabrett Summary: Turn raw business data (CSV, SQLite, spreadsheets, pasted tables) into clear analytical summaries, trend analysis, and actionable reports for small business... Tags: latest:0.1.0 Version history: v0.1.0 | 2026-04-07T23:17:05.424Z | user v0.1.0 — First public release. Turn raw business data into clear analytical reports with plain-language insights. Clarifi","codeSnippets":[],"executableExamples":[{"language":"text","snippet":"data-analysis-reporting/\n├── README.md                          # This file\n├── SKILL.md                           # Skill definition (triggers, workflow, output structure)\n├── data-analysis-reporting-spec.md    # Full spec (purpose, capabilities, boundaries, market context)\n└── references/\n    ├── business-metrics.md            # Standard business metrics: formulas, interpretation, benchmarks\n    ├── data-quality-checks.md         # Pre-analysis data validation checklist\n    ├── report-templates.md            # 5 output templates (executive, detailed, comparison, trend, health check)\n    └── test-prompts.md                # 12 test cases (4 happy, 4 normal, 4 edge)"},{"language":"text","snippet":"## Data Quality Summary\n\n**Dataset:** [name/description]\n**Rows:** [count] | **Columns:** [count] | **Date range:** [start] to [end]\n**Overall quality:** [Good / Moderate / Poor]\n\n### Issues Found\n- [Issue 1]: [description and impact on analysis]\n- [Issue 2]: [description and impact on analysis]\n\n### Actions Taken\n- [Normalization or cleaning step 1]\n- [Normalization or cleaning step 2]\n\n### Caveats for Analysis\n- [How remaining issues affect conclusions]"},{"language":"text","snippet":"# [Report Title] — Executive Summary\n\n**Period:** [date range]\n**Prepared for:** [audience]\n**Data source:** [description]\n\n## Bottom Line\n\n[2-3 sentences: the single most important takeaway and what it means for the business]\n\n## Key Numbers\n\n| Metric | Current | Prior Period | Change |\n|--------|---------|-------------|--------|\n| [metric 1] | [value] | [value] | [+/-X%] |\n| [metric 2] | [value] | [value] | [+/-X%] |\n| [metric 3] | [value] | [value] | [+/-X%] |\n\n## What's Working\n- [1-2 bullets on positive trends]\n\n## What Needs Attention\n- [1-2 bullets on concerning trends or risks]\n\n## Recommended Actions\n1. [Action with clear owner and timeline]\n2. [Action with clear owner and timeline]\n3. [Action with clear owner and timeline]\n\n## Data Notes\n[One line on data quality, completeness, and any caveats]"},{"language":"text","snippet":"# [Report Title] — Detailed Analysis\n\n**Period:** [date range]\n**Prepared for:** [audience]\n**Data source:** [description]\n**Analysis date:** [date]\n\n## Executive Summary\n[3-5 bullet points — the findings that matter most]\n\n## Data Quality Notes\n[Completeness, cleaning steps, caveats — see data-quality-checks.md]\n\n## Finding 1: [Title]\n**What the data shows:** [fact with numbers]\n**Why it matters:** [business implication]\n**Confidence:** [High/Moderate/Low]\n**Detail:** [supporting analysis, breakdowns, context]\n\n## Finding 2: [Title]\n[Same structure]\n\n## Finding 3: [Title]\n[Same structure]\n\n## Trend Analysis\n[Time-series observations with comparison baselines]\n- Direction and magnitude\n- Acceleration/deceleration\n- Seasonal patterns if present\n\n## Watch Items\n- [Emerging concern 1 — what to monitor and when to act]\n- [Emerging concern 2]\n\n## Recommended Actions\n1. **Now:** [Action justified by current data]\n2. **Monitor:** [What to track and what threshold triggers action]\n3. **Improve:** [Data or process improvement for better future analysis]\n\n## Methodology\n- Metrics calculated: [list with formulas]\n- Time periods compared: [description]\n- Exclusions: [what was excluded and why]\n- Assumptions: [any assumptions made]"},{"language":"text","snippet":"# [A] vs. [B] — Comparison Analysis\n\n**Period:** [date range]\n**Comparison basis:** [what's being compared and why]\n\n## Summary Verdict\n[Which option/segment/period performs better overall, and the key reason why]\n\n## Side-by-Side Comparison\n\n| Dimension | [A] | [B] | Delta | Winner |\n|-----------|-----|-----|-------|--------|\n| [metric 1] | [value] | [value] | [diff] | [A/B/Tie] |\n| [metric 2] | [value] | [value] | [diff] | [A/B/Tie] |\n| [metric 3] | [value] | [value] | [diff] | [A/B/Tie] |\n\n## Where [A] Outperforms\n- [Specific area with numbers and context]\n\n## Where [B] Outperforms\n- [Specific area with numbers and context]\n\n## Context and Caveats\n- [Factors that explain differences — seasonality, sample size, external events]\n- [Evidence quality differences between options]\n\n## Recommendation\n[What to do based on the comparison, with confidence level]"},{"language":"text","snippet":"# [Metric/Area] Trend Analysis\n\n**Period:** [full date range analyzed]\n**Granularity:** [daily/weekly/monthly]\n**Baseline:** [comparison reference point]\n\n## Trend Summary\n[2-3 sentences: what's the overall direction, how fast, and is it accelerating or decelerating?]\n\n## Period-by-Period Breakdown\n\n| Period | Value | Change | Growth Rate | Notes |\n|--------|-------|--------|-------------|-------|\n| [period 1] | [value] | — | — | Baseline |\n| [period 2] | [value] | [+/-] | [%] | |\n| [period 3] | [value] | [+/-] | [%] | [notable event] |\n\n## Key Observations\n1. **[Observation]** — [what the trend shows and confidence level]\n2. **[Observation]** — [what the trend shows and confidence level]\n\n## Seasonal/Cyclical Patterns\n[If detectable with available data; if not, state minimum data needed to detect]\n\n## Inflection Points\n[Periods where the trend changed direction or rate, with possible explanations]\n\n## Projection Context\n[If the user asks \"where is this going\": state assumptions, range, and confidence. Never present as fact.]\n\n## What Would Change This Trend\n- Upside scenario: [what would accelerate positive trend]\n- Downside scenario: [what would reverse or worsen the trend]"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: data-analysis-reporting\ndescription: Turn raw business data (CSV, SQLite, spreadsheets, pasted tables) into clear analytical summaries, trend analysis, and actionable reports for small business operators, analysts, and decision-makers. Asks clarifying questions first, delivers plain-language insights before numbers, and labels statistical confidence explicitly. Does not provide financial advice or present projections as fact.\n---\n\n# Data Analysis & Reporting\n\nTurn raw business data into plain-language insights, trend analysis, and actionable reports. Think sharp junior analyst, not statistics engine.\n\n## Trigger conditions\n\nActivate this skill when the user:\n- Pastes or uploads tabular data (CSV, markdown table, tab-separated, pipe-delimited)\n- Asks to analyze, summarize, or report on business data\n- Asks about metrics, KPIs, trends, or performance from a dataset\n- Provides a SQLite database and asks questions about it\n- Asks for a report, executive summary, or data briefing\n- Asks to compare periods, segments, or actuals vs. targets\n\nDo NOT activate when:\n- The user wants to build a dashboard or visualization tool (suggest BI tools)\n- The user needs real-time streaming analytics\n- The user asks for financial advice, investment recommendations, or tax guidance\n- The data is code/logs/system metrics (suggest observability tools instead)\n\n## Work the request in this order\n\n1. **Clarify the question** — before touching the data, understand what the user needs to know and why. Ask up to 3 clarifying questions:\n   - \"What decision does this analysis need to support?\"\n   - \"What time period or comparison matters most?\"\n   - \"Who is the audience for this report?\"\n   If the user provides clear context, skip to step 2.\n\n2. **Ingest and validate** — parse the data, detect column types, run quality checks\n   - Auto-detect: column types (numeric, date, categorical, text)\n   - Flag: missing values, outliers, formatting inconsistencies, duplicate rows\n   - Report data quality issues before proceeding, not after\n   - If data quality is poor enough to undermine analysis, say so and recommend fixes\n\n3. **Propose an analysis plan** — tell the user what you intend to analyze and why, before doing it\n   - Name the specific analyses (e.g., \"monthly revenue trend with MoM growth rates\")\n   - Explain what each analysis will reveal relative to their question\n   - Let the user adjust before you proceed\n\n4. **Execute the analysis** — run the agreed analyses\n   - Summary statistics for numeric columns\n   - Trend identification with direction, magnitude, and acceleration\n   - Comparisons (period-over-period, segment, actual vs. target) as relevant\n   - Distribution and concentration analysis where useful\n   - Correlation spotting between metrics\n   - Cohort analysis when data supports grouping\n\n5. **Translate to insights** — convert numbers into plain-language findings\n   - Lead with what matters, not what was calculated\n   - Rank findings by business impact, not "},{"path":"README.md","content":"# Data Analysis & Reporting\n\nTurn raw business data into plain-language insights, trend analysis, and actionable reports.\n\n## Status\n\nv0.1.0 — Board-approved for publication. 43/45 QA score across 12 test cases.\n\n## Structure\n\n```\ndata-analysis-reporting/\n├── README.md                          # This file\n├── SKILL.md                           # Skill definition (triggers, workflow, output structure)\n├── data-analysis-reporting-spec.md    # Full spec (purpose, capabilities, boundaries, market context)\n└── references/\n    ├── business-metrics.md            # Standard business metrics: formulas, interpretation, benchmarks\n    ├── data-quality-checks.md         # Pre-analysis data validation checklist\n    ├── report-templates.md            # 5 output templates (executive, detailed, comparison, trend, health check)\n    └── test-prompts.md                # 12 test cases (4 happy, 4 normal, 4 edge)\n```\n\n## Key Design Decisions\n\n- **Clarify first, analyze second.** The skill asks what decision the analysis supports before touching the data.\n- **Data quality before results.** Every analysis starts with a quality check and reports issues upfront.\n- **Insights over numbers.** Output leads with \"what this means\" not \"what the number is.\"\n- **Confidence labeling.** Every claim gets High/Moderate/Low/Flagged confidence.\n- **Honest about limits.** Small samples, poor data quality, and insufficient evidence get clear warnings — not padding.\n\n## Differentiator\n\nFeels like a sharp junior analyst on staff, not a statistics engine. Asks clarifying questions, proposes an analysis plan, and surfaces follow-up questions the user hasn't thought to ask.\n\n## License\n\nPublished by NorthlineAILabs."},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7bz9d5bwyvakz2t7hqjbeyf183qcc6\",\n  \"slug\": \"data-analysis-reporting\",\n  \"version\": \"0.1.0\",\n  \"publishedAt\": 1775603825424\n}"},{"path":"references/business-metrics.md","content":"# Business Metrics Reference\n\nStandard business metrics, how to calculate them, and how to interpret the results. Use this reference when the user's data contains these metrics or when calculating them from raw data.\n\n## Revenue Metrics\n\n### MRR (Monthly Recurring Revenue)\n- **Formula:** Sum of all active subscription revenue normalized to monthly\n- **Components:**\n  - **New MRR** — revenue from new customers acquired this month\n  - **Expansion MRR** — revenue increase from existing customers (upgrades, add-ons)\n  - **Contraction MRR** — revenue decrease from existing customers (downgrades)\n  - **Churned MRR** — revenue lost from customers who cancelled\n  - **Net New MRR** = New + Expansion - Contraction - Churned\n- **Interpretation:**\n  - Healthy SaaS: Net New MRR is positive and growing\n  - Warning: Churned MRR exceeding New MRR for 2+ consecutive months\n  - Context: Always report components, not just the total — a flat MRR can mask high churn offset by high acquisition\n\n### ARR (Annual Recurring Revenue)\n- **Formula:** MRR x 12\n- **Caution:** Only valid when MRR is relatively stable month-to-month. Do not annualize a spike month.\n- **Use:** Standard for SaaS valuation, fundraising metrics, and annual planning\n\n### Revenue Growth Rate\n- **MoM:** (This month - Last month) / Last month x 100\n- **QoQ:** (This quarter - Last quarter) / Last quarter x 100\n- **YoY:** (This period - Same period last year) / Same period last year x 100\n- **CAGR:** (End value / Start value)^(1/years) - 1\n- **Best practice:** Always show absolute numbers alongside percentages. A 50% MoM growth from $2K to $3K is very different from $200K to $300K.\n\n### ARPU (Average Revenue Per User/Account)\n- **Formula:** Total Revenue / Total Customers (for the same period)\n- **Cross-validation:** When a dataset provides Revenue, Customers, and ARPU columns, verify that Revenue / Customers = stated ARPU. Flag any mismatches before proceeding — contradictory data undermines all downstream analysis.\n- **Variants:**\n  - **ARPU** — per user, common in consumer products\n  - **ARPA** — per account, common in B2B where one account has multiple users\n- **Interpretation:**\n  - Rising ARPU with stable customer count = healthy expansion / upsell\n  - Falling ARPU with growing customer count = adding lower-value customers (may be fine strategically)\n  - Flat ARPU = stable unit economics, look at volume for growth signal\n- **Caution:** ARPU is an average — high variance means the average may be misleading. Report median alongside mean when possible.\n\n## Customer Metrics\n\n### Customer Churn Rate\n- **Logo churn:** Customers lost / Customers at start of period x 100\n- **Revenue churn (gross):** Churned MRR / MRR at start of period x 100\n- **Revenue churn (net):** (Churned MRR - Expansion MRR) / MRR at start of period x 100\n- **Important:** Ask the user which definition they use. \"Churn\" means different things to different businesses.\n- **Benchmarks (SaaS):**\n  - SMB: 3-7% monthly logo churn is common\n "},{"path":"references/data-quality-checks.md","content":"# Data Quality Checks\n\nRun these checks on every dataset before analysis. Report findings before results. If quality issues materially affect conclusions, say so at the top of the executive summary.\n\n## 1. Completeness\n\n### Missing values\n- Count missing/null/empty values per column\n- Calculate completeness rate per column: (non-null rows / total rows) x 100\n- Flag columns below 90% completeness\n- Distinguish between: truly missing, intentionally blank (e.g., optional fields), and encoded as zero/placeholder\n\n### Row completeness\n- Calculate per-row completeness: (non-null fields / total fields) x 100\n- Flag rows with <50% completeness for potential exclusion\n- Report: \"X of Y rows are fully complete; Z rows have significant gaps\"\n\n### Date coverage\n- Identify the date range in the data\n- Check for gaps (missing days, weeks, months in a time series)\n- Report: \"Data covers [start] to [end] with [N] gaps in coverage\"\n\n## 2. Consistency\n\n### Date formats\n- Check for mixed date formats (MM/DD/YYYY vs. DD/MM/YYYY vs. YYYY-MM-DD)\n- Attempt auto-detection; if ambiguous (e.g., 03/04/2025), ask the user\n- Normalize to ISO 8601 (YYYY-MM-DD) for analysis\n\n### Categorical values\n- Check for variant spellings, casing differences, trailing spaces\n- Examples: \"Active\" vs. \"active\" vs. \"ACTIVE\" vs. \" Active\"\n- Report unique values per categorical column and flag likely duplicates\n- Normalize after confirming with user if ambiguous\n\n### Unit consistency\n- Check for mixed units in the same column (e.g., $ and EUR, kg and lbs)\n- Check for mixed scales (e.g., some values in thousands, others in actuals)\n- Flag and ask before normalizing\n\n### Naming consistency\n- Check column headers for typos, inconsistent naming conventions\n- Report any columns that appear to contain the same data under different names\n\n## 3. Validity\n\n### Range checks\n- Numeric columns: flag values outside expected ranges\n  - Negative values where only positive expected (e.g., revenue, counts)\n  - Zero values where zero is unusual (e.g., price, quantity)\n  - Values orders of magnitude above/below the column median\n- Date columns: flag future dates (if unexpected), dates before a reasonable start\n- Percentage columns: flag values >100% or <0% (unless explicitly allowed)\n\n### Type validation\n- Check that numeric columns contain only numbers (no text mixed in)\n- Check that date columns parse correctly\n- Flag columns where >5% of values fail type validation\n\n### Business rule validation\n- Revenue = Quantity x Price (if all three columns exist)\n- End date >= Start date\n- Running totals match sum of components\n- Report any violations found\n\n## 4. Uniqueness\n\n### Duplicate detection\n- Check for exact duplicate rows\n- Check for near-duplicates (same key fields, different minor fields)\n- Report: \"Found X exact duplicates and Y potential near-duplicates\"\n\n### Key column analysis\n- Identify likely primary key columns (high cardinality, no nulls)\n- Check uniqueness of suspected key columns\n- Flag if a suppose"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Turn raw business data (CSV, SQLite, spreadsheets, pasted tables) into clear analytical summaries, trend analysis, and actionable reports for small business... Skill: Data Analysis Reporting Owner: gitcanadabrett Summary: Turn raw business data (CSV, SQLite, spreadsheets, pasted tables) into clear analytical summaries, trend analysis, and actionable reports for small business... Tags: latest:0.1.0 Version history: v0.1.0 | 2026-04-07T23:17:05.424Z | user v0.1.0 — First public release. Turn raw business data into clear analytical reports with plain-language insights. Clarifi","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1802,"uniquenessScore":47,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T17:15:55.241Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T17:15:55.241Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T18:05:28.613Z","emptyReason":null},"items":[{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-10T18:48:31.762Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}