{"id":"a69ca8a9-75ae-432b-9da0-0f6dd00425ab","entityType":"agent","slug":"clawhub-zhumorris-agent-causal","name":"Agent Causal","canonicalUrl":"https://www.xpersona.co/agent/clawhub-zhumorris-agent-causal","canonicalPath":"/agent/clawhub-zhumorris-agent-causal","generatedAt":"2026-10-11T03:55:52.203Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-11T00:10:40.806Z","emptyReason":null},"description":"Agent Causal Decision Tool helps you and your AI agents answer one question from experiment data: should we ship this change, keep running the test, or roll... Skill: Agent Causal Owner: zhumorris Summary: Agent Causal Decision Tool helps you and your AI agents answer one question from experiment data: should we ship this change, keep running the test, or roll... Tags: latest:0.10.3 Version history: v0.10.3 | 2026-05-07T11:46:56.976Z | user Skill version 0.10.3 aligned with v0.10.2 git tag — SKILL.md tarball URL now uses v0.10.2 (matching the published skill version). Also","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.2K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s17977p9bp8aqeg4qvcakzky5n85syh7:agent-causal","sourceUrl":"https://clawhub.ai/zhumorris/agent-causal","homepage":"https://clawhub.ai/zhumorris/skills/agent-causal","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/zhumorris/agent-causal","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/zhumorris/skills/agent-causal","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":62,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Agent Causal Decision Tool helps you and your AI agents answer one question from experiment data: should we ship this change, keep running the test, or roll... "},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T00:10:40.806Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T00:10:40.806Z","emptyReason":null},"stars":null,"forks":null,"downloads":1223,"packageName":null,"latestVersion":"0.10.3","tractionLabel":"1.2K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T00:10:40.728Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T00:10:40.806Z","lastCrawledAt":"2026-10-11T00:10:40.728Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T00:10:40.728Z","lastVerifiedAt":null,"highlights":[{"version":"0.10.3","createdAt":"2026-05-07T11:46:56.976Z","changelog":"Skill version 0.10.3 aligned with v0.10.2 git tag — SKILL.md tarball URL now uses v0.10.2 (matching the published skill version). Also creates git tag v0.10.2 on GitHub for reproducibility.","fileCount":3,"zipByteSize":12014},{"version":"0.10.2","createdAt":"2026-05-07T11:41:04.487Z","changelog":"ClawScan fixes: tarball URL version (0.10.0→0.10.1), explicit PostHog network call gate, least-privilege token guidance, fix runtime network contradiction","fileCount":2,"zipByteSize":10502},{"version":"0.10.1","createdAt":"2026-05-07T11:16:54.334Z","changelog":"Security hardening: primary install now uses release tarball download (curl+tar) instead of git clone; added credential handling docs; clarified hardcoded exec commands","fileCount":2,"zipByteSize":10420},{"version":"0.10.0","createdAt":"2026-05-07T09:44:51.468Z","changelog":"Fix B: stdio invalid-params error code -32600→-32602. Fix C: docstring '9 actions'→'10 actions'. Fix D: stdio buffer guard (max 1000 lines / 100KB to prevent memory exhaustion).","fileCount":2,"zipByteSize":10224},{"version":"0.9.9","createdAt":"2026-05-07T08:47:45.312Z","changelog":"Fix all 12 Cursor review findings: Bayesian RNG seeding, webhook payload copy, httpx import guard, ValueError in plan, Any import, bayesian is-True check, lift=inf guard, cohort prior_decision=None, magic 'unknown' sentinel, dead code removal, Optional annotation.","fileCount":2,"zipByteSize":10225},{"version":"0.9.8","createdAt":"2026-05-07T08:17:24.123Z","changelog":"Fix HTTP startup doc (remove broken uvicorn src.api:app). Add ge=0 validation to DIDInput — negative values now return ValidationError instead of crashing in Poisson bootstrap.","fileCount":2,"zipByteSize":10224},{"version":"0.9.7","createdAt":"2026-05-07T08:04:49.803Z","changelog":"Move httpx import to module level (not per-call). Add mock-based tests verifying webhook fires on ship/escalate but not on keep_running.","fileCount":2,"zipByteSize":10224},{"version":"0.9.6","createdAt":"2026-05-07T08:00:16.013Z","changelog":"Phase 4 (run_workflow orchestrator) + Phase 5 (webhook notifications). Fix --save/--notify CLI flag polarity.","fileCount":2,"zipByteSize":10224}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17977p9bp8aqeg4qvcakzky5n85syh7:agent-causal","setupComplexity":"low","setupSteps":["Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zhumorris-agent-causal/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zhumorris-agent-causal/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zhumorris-agent-causal/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-zhumorris-agent-causal/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-zhumorris-agent-causal/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-zhumorris-agent-causal/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T03:55:52.200Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zhumorris-agent-causal/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zhumorris-agent-causal/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zhumorris-agent-causal/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zhumorris-agent-causal/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-11T00:10:40.806Z","emptyReason":null},"readme":"Skill: Agent Causal\n\nOwner: zhumorris\n\nSummary: Agent Causal Decision Tool helps you and your AI agents answer one question from experiment data: should we ship this change, keep running the test, or roll...\n\nTags: latest:0.10.3\n\nVersion history:\n\nv0.10.3 | 2026-05-07T11:46:56.976Z | user\n\nSkill version 0.10.3 aligned with v0.10.2 git tag — SKILL.md tarball URL now uses v0.10.2 (matching the published skill version). Also creates git tag v0.10.2 on GitHub for reproducibility.\n\nv0.10.2 | 2026-05-07T11:41:04.487Z | user\n\nClawScan fixes: tarball URL version (0.10.0→0.10.1), explicit PostHog network call gate, least-privilege token guidance, fix runtime network contradiction\n\nv0.10.1 | 2026-05-07T11:16:54.334Z | user\n\nSecurity hardening: primary install now uses release tarball download (curl+tar) instead of git clone; added credential handling docs; clarified hardcoded exec commands\n\nv0.10.0 | 2026-05-07T09:44:51.468Z | user\n\nFix B: stdio invalid-params error code -32600→-32602. Fix C: docstring '9 actions'→'10 actions'. Fix D: stdio buffer guard (max 1000 lines / 100KB to prevent memory exhaustion).\n\nv0.9.9 | 2026-05-07T08:47:45.312Z | user\n\nFix all 12 Cursor review findings: Bayesian RNG seeding, webhook payload copy, httpx import guard, ValueError in plan, Any import, bayesian is-True check, lift=inf guard, cohort prior_decision=None, magic 'unknown' sentinel, dead code removal, Optional annotation.\n\nv0.9.8 | 2026-05-07T08:17:24.123Z | user\n\nFix HTTP startup doc (remove broken uvicorn src.api:app). Add ge=0 validation to DIDInput — negative values now return ValidationError instead of crashing in Poisson bootstrap.\n\nv0.9.7 | 2026-05-07T08:04:49.803Z | user\n\nMove httpx import to module level (not per-call). Add mock-based tests verifying webhook fires on ship/escalate but not on keep_running.\n\nv0.9.6 | 2026-05-07T08:00:16.013Z | user\n\nPhase 4 (run_workflow orchestrator) + Phase 5 (webhook notifications). Fix --save/--notify CLI flag polarity.\n\nv0.9.5 | 2026-05-07T07:36:34.485Z | user\n\nAdd cohort/segment breakdown to the auto-detection table in README.\n\nv0.9.4 | 2026-05-07T07:29:29.810Z | user\n\nRemove duplicate ### A/B Test Analysis (Frequentist) section from both SKILL.md files. All sections now unique.\n\nv0.9.3 | 2026-05-07T07:25:47.146Z | user\n\nRemove duplicate ## When to Use It section. Only one instance remains (inside the What is this / Why it exists / When to use it block).\n\nv0.9.2 | 2026-05-07T07:22:50.375Z | user\n\nUpdate Why it exists / When to use it sections to include easy-mode dispatcher (decide) and PostHog connector (connect). No code changes.\n\nv0.9.1 | 2026-05-07T07:15:14.729Z | user\n\nAdd What is this / Why it exists / When to use it section to ClawHub SKILL.md to match GitHub README structure\n\nv0.9.0 | 2026-05-07T07:08:40.129Z | user\n\nPhase 2: Easy-mode dispatcher + Phase 3: PostHog connector. 9 actions (added decide, connect). 475 tests passing.\n\nv0.8.2 | 2026-05-03T09:46:17.642Z | user\n\nSecurity hardening: explicit Security Model section explaining no runtime remote fetch, correct publishedAt timestamp, removed pip install git+ options\n\nv0.8.1 | 2026-05-03T09:40:22.443Z | user\n\nv0.8.1: correct SKILL.md content, lift_ci_95, mde_ci_95, --n-bootstrap, 30 warning codes\n\nv0.8.0 | 2026-05-03T05:31:17.626Z | auto\n\nNo user-visible changes in this version.\n\n- Version bumped to 0.8.0 with no detected changes to files or documentation.\n- Functionality and usage remain identical to the previous release.\n\nv0.7.7 | 2026-05-02T04:01:50.005Z | user\n\nSecurity: add threat model to SECURITY.md addressing exec/RCE, git install, and SQLite risk; SKILL.md: local install as recommended option (no remote fetch at runtime)\n\nv0.7.6 | 2026-05-01T14:43:43.194Z | auto\n\nagent-causal 0.7.6 is a documentation version update.\n\n- Updated SKILL.md to reflect version 0.7.6 in metadata.\n- No functional or code changes; this release updates documentation only.\n- Ensures documentation is consistent with the current version.\n\nv0.7.5 | 2026-05-01T14:31:42.649Z | auto\n\n- Version updated from 0.7.3 to 0.7.5 in SKILL.md metadata.\n- No changes to code or functionality; this update reflects documentation/metadata only.\n\nv0.7.4 | 2026-05-01T14:26:58.493Z | auto\n\nagent-causal 0.7.4 changelog:\n\n- Metadata formatting in SKILL.md changed from JSON-style to YAML-style for better compatibility.\n- No changes to functionality or user-facing documentation.\n\nv0.7.3 | 2026-05-01T14:25:36.111Z | auto\n\n- Update version to 0.7.3 in skill metadata for consistency.\n- No functional or user-facing changes; SKILL.md documentation version synced with codebase.\n\nv0.7.2 | 2026-05-01T14:20:52.445Z | auto\n\nagent-causal 0.7.2\n\n- Updated SKILL.md metadata to reflect version 0.7.2.\n- No functional or interface changes; documentation and metadata update only.\n\nv0.7.1 | 2026-05-01T14:06:26.304Z | user\n\nAdd cohort_breakdown (Mode 6) with Benjamini-Hochberg correction. Cohort decision override flag. next_analysis_suggestion in ab output. CSV input support. Python API for cohort_breakdown.\n\nv0.7.0 | 2026-05-01T03:08:30.759Z | auto\n\nVersion 0.7.0 adds sequential/early stopping support and improves documentation.\n\n- Added docs for sequential/early stopping: opt-in flag for A/B tests, allows early stopping when evidence is strong, with conservative thresholds and audit trail.\n- Updated \"Why it exists\" section to mention sequential testing capability.\n- Added a \"Quickstart\" section for faster onboarding.\n- No changes to core commands or requiring additional dependencies.\n\nv0.6.7 | 2026-05-01T02:55:36.907Z | auto\n\n- Documentation formatting improved for clarity and consistency in the SKILL.md.\n- Minor text edits to bullet points, headings, and list formatting.\n- No changes to tool functionality, commands, parameters, or outputs.\n\nv0.6.6 | 2026-05-01T02:53:40.385Z | auto\n\nNo functional or logic changes; this version updates documentation only.\n\n- Revised and reformatted SKILL.md for clarity and consistency.\n- Minor capitalization and style adjustments.\n- No code or user-facing command changes.\n- No updates to features or dependencies.\n\nv0.6.5 | 2026-05-01T02:46:32.943Z | auto\n\n- Major documentation update: SKILL.md rewritten for clarity, brevity, and step-by-step usage.\n- Concise descriptions added for tool purpose, key capabilities, and when to use each analysis.\n- Example commands and typical output for planning, frequentist A/B, Bayesian A/B, and DiD highlighted in short, readable blocks.\n- Guidance improved on setup, parameters, and output interpretation.\n- Technical details and setup examples separated for easier onboarding.\n\nv0.6.4 | 2026-05-01T02:39:50.660Z | auto\n\n- SKILL.md description rewritten for conciseness and clarity; now provides a succinct summary of the tool and its purpose.\n- Full documentation and detailed usage remain available further in the file.\n- No changes to code, features, or commands in this version.\n\nv0.6.3 | 2026-05-01T02:39:17.187Z | auto\n\nVersion 0.6.3 Changelog\n\n- Expanded and clarified skill documentation for greater context and usability.\n- Added a detailed introductory explanation, including when and why to use the decision tool.\n- Explained typical team and agent workflows, highlighting limitations of ad hoc experiment analysis.\n- Emphasized audit trails, transparency, and structured decisions for agent/automation integration.\n- No functional or code changes—documentation update only.\n\nv0.6.2 | 2026-05-01T02:09:40.661Z | auto\n\nNo user-facing changes detected in version 0.6.2.\n\n- No changes were found between this version and the previous release.\n- Documentation, features, and metadata remain unchanged.\n\nv0.6.1 | 2026-04-30T10:28:54.249Z | user\n\nForce rescan after security hold\n\nv0.6.0 | 2026-04-30T09:49:04.961Z | user\n\nPhase II.A/B: sequential early stopping + DiD diagnostics; 6 static-scan fixes\n\nv0.5.5 | 2026-04-30T06:23:51.804Z | auto\n\n## agent-causal 0.5.5\n\n- Updated documentation; no functional changes to the skill's logic or features.\n- Project description, usage instructions, and examples remain unchanged.\n- Internal metadata or packaging information may have been updated.\n\nv0.5.4 | 2026-04-30T06:16:13.588Z | auto\n\n- Adds detailed documentation to SKILL.md, covering setup, commands, parameters, and example outputs.\n- Documents support for experiment planning, frequentist and Bayesian A/B tests, Difference-in-Differences analysis, audit/reproducibility, and SQLite-backed experiment history.\n- Outlines new audit features including experiment maturity scoring and detailed decision paths.\n- Includes updated usage instructions for all major tool commands.\n\nv0.5.3 | 2026-04-30T06:08:13.580Z | auto\n\nagent-causal v0.5.3\n\n- Updated skill description to better explain the tool’s purpose and clarify outputs.\n- Version bump to 0.5.3 in metadata.\n- No functional or interface changes to commands or features.\n\nv0.5.2 | 2026-04-30T05:01:51.869Z | auto\n\n- Version bump to 0.5.2.\n- Updated metadata in SKILL.md to reflect new version (0.5.2).\n- No changes to features, usage, or commands.\n\nv0.5.1 | 2026-04-30T04:39:37.904Z | auto\n\n- Added license field (\"Apache-2.0\") to skill metadata.\n- Updated metadata version from 0.5.0 to 0.5.1.\n\nv0.5.0 | 2026-04-30T04:22:54.370Z | auto\n\n**Big update: adds experiment planning, Bayesian A/B tests, audits with maturity scoring, and history persistence.**\n\n- Added experiment planning (`ab_plan`) to estimate required sample size, test duration, and feasibility.\n- Introduced Bayesian A/B test analysis with probability-based decision thresholds.\n- Decision audits now support maturity scoring, grading experiment readiness.\n- All results can be saved to (and listed from) local SQLite experiment history.\n- Metadata version updated to 0.5.0, and documentation extensively updated.\n\nv0.2.0 | 2026-04-29T09:57:53.834Z | user\n\nFix security review issues: add GitHub source link, explicit install steps, proper setup instructions\n\nv0.1.0 | 2026-04-29T09:50:34.180Z | user\n\nInitial release: A/B testing, DiD analysis, and Decision Audit for AI agents\n\nArchive index:\n\nArchive v0.10.3: 3 files, 12014 bytes\n\nFiles: _meta.json (132b), skill-card.md (2777b), SKILL.md (28118b)\n\nFile v0.10.3:SKILL.md\n\n---\nname: agent-causal\ndescription: \"Agent Causal Decision Tool helps you and your AI agents answer one question from experiment data: should we ship this change, keep running the test, or roll it back? Returns structured JSON decisions, key statistics, and audit trails from A/B tests (frequentist + Bayesian), DiD, cohort/segment analysis, and sequential early stopping.\"\nmetadata:\n  openclaw:\n    category: data-science\n    version: \"0.10.3\"\n    license: Apache-2.0\n    tools: [exec]\n    requires:\n      bins: [python3, git, pip]\n      python_packages: [click, scipy, numpy, pydantic]\n    source: https://github.com/ZhuMorris/agent-causal-decision-tool\n---\n\n# Agent Causal Decision Tool\n\nA causal decision and audit tool for AI agents. Evaluate product changes using A/B testing, Difference-in-Differences, and sequential early stopping.\n\n**Source:** https://github.com/ZhuMorris/agent-causal-decision-tool\n\n## What is this?\n\nAgent Causal Decision Tool helps you and your AI agents answer one question from experiment data: \"should we ship this change, keep running the test, or roll it back?\" It takes in simple A/B or rollout summaries and returns a structured JSON decision, key statistics, and an audit record you can store or review later.\n\nRather than being a full experimentation platform, it is a **decision engine**. You bring the data (from your logs, BI tool, or CSV); it handles the stats, decision logic, and audit trail.\n\n### Why it exists\n\nIn many teams, experiment decisions happen in ad hoc spreadsheets or dashboards. People glance at lift, argue about whether the sample size is enough, and sometimes ship features based on noisy or biased results. Agents make this worse if they are wired to react to any small uplift they see.\n\nThis tool wraps a few standard methods into one consistent, agent‑friendly interface:\n\n- **Easy-mode dispatcher (`decide`)** — no need to know which statistical method to use. Paste your numbers and it auto-selects A/B, Bayesian, DiD, or planning from your input fields.\n- **Frequentist A/B testing** for classic \"control vs variant\" questions.\n- **Bayesian A/B testing** when you want answers like \"there is a 93% chance B is better than A\" instead of only p‑values.\n- **Difference‑in‑differences (DiD)** for quasi‑experiments like staged rollouts or region‑based launches where you cannot randomize perfectly.\n- **Cohort / segment breakdown** when an aggregate result is inconclusive — you can slice by user segment to find hidden signals, with Benjamini-Hochberg correction for 4+ segments.\n- **Planning and power checks** so you can see if a test is realistic before you start it.\n- **Decision audit** so humans can see what the agent did, why it did it, and how strong the evidence really was.\n- **External connectors** — pull experiment data directly from PostHog, normalize it, and run a decision in one step. No manual export needed.\n\nThe goal is not to replace your analytics stack, but to give agents a small, reliable decision block they can call inside workflows.\n\n### When to use it\n\nUse this tool whenever you or your agents have experiment or rollout results and need a decision you can defend:\n\n- You ran an A/B test and want to know whether to ship, keep running, or reject the variant.\n- You're not sure which method to use — let `decide` auto-detect from your numbers.\n- You ran an A/B test and it was inconclusive — you want to know if a specific user segment is driving (or diluting) the effect.\n- You rolled out a feature to one region or cohort first and want a DiD estimate of impact compared to a similar control group.\n- You prefer a Bayesian summary (\"95% chance B is better; expected lift 3–5%\") to drive thresholds in automated workflows.\n- You need an audit trail with experiment period, traffic size, assumptions, thresholds, and warnings so product, data, or risk teams can review agent decisions later.\n- You want to plan an experiment (sample size, minimum detectable effect, expected duration) or compare current results to previous experiments to see which wins are robust.\n- Your experiment data lives in PostHog — you want to fetch, normalize, and decide without any manual CSV export.\n\n---\n\n## Security Model\n\nThis skill runs as a **local CLI tool only**. After one-time setup, it requires no network access.\n\n**Setup (one-time, before first use):**\n```bash\n# Download the release tarball — no git clone needed\ncurl -sL https://github.com/ZhuMorris/agent-causal-decision-tool/archive/refs/tags/v0.10.2.tar.gz -o agent-causal.tar.gz\ntar -xzf agent-causal.tar.gz\npip install agent-causal-decision-tool-0.10.2/ -q\n```\n\n\nAfter installation, the `agent-causal` command is available locally. The tool performs only local statistical calculations.\n\n**When network access occurs:** The PostHog connector (`agent-causal connect posthog`) makes outbound HTTPS requests to your PostHog instance only when explicitly invoked — never automatically. If you do not use the connector, no outbound network access is needed at any point.\n\n**No runtime network access during analysis.** The decision engine, audit, and cohort analysis do not make outbound requests.\n**Tools used:** `exec` (for running the `agent-causal` CLI commands you specify). Commands are fully hardcoded with no user-supplied strings interpolated into shell execution.\n**PostHog token scope:** Use a read-only API token with minimal scopes. Do not use tokens with write or admin permissions.\n**Credential handling:** PostHog API credentials are read from env vars (`POSTHOG_API_KEY`/`POSTHOG_PROJECT_ID`) or a local `~/.posthogrc` file — never hardcoded or logged.\n\n---\n\n## Setup\n\nBefore using this skill, install the tool (one-time):\n\n```bash\ngit clone https://github.com/ZhuMorris/agent-causal-decision-tool.git ~/clawd/agent-causal-decision-tool\npip install ~/clawd/agent-causal-decision-tool -q\n```\n\nAfter this, `agent-causal` is available as a local command. No further network access is required.\n\n---\n\n## Agent-Native Actions (JSON-RPC 2.0)\n\nFor AI agent integrations, Agent Causal exposes a JSON-RPC 2.0 API over both stdio and HTTP.\n\n### Stdio mode (for OpenClaw, Codex, Claude Code)\n\n```bash\npython -m src.api stdio\n```\n\n### HTTP mode\n\n```bash\npython -m src.api http --port 8000\n```\n\n### Actions\n\n| Action | Description |\n|--------|-------------|\n| `decide` | **Easy-mode dispatcher** — auto-selects A/B, Bayesian, DiD, or planning from your input fields |\n| `decide_ab` | Frequentist A/B test (`mode: frequentist`) or Bayesian (`mode: bayesian`) |\n| `decide_rollout` | DiD for staged rollouts / quasi-experiments |\n| `plan_test` | Experiment planning (sample size, MDE, feasibility) |\n| `audit_result` | Full audit of a stored result by ID |\n| `save_result` | Persist a decision result to SQLite history |\n| `get_result` | Retrieve a stored result by ID |\n| `compare_results` | Compare multiple stored experiments |\n| `connect` | Fetch experiment data from an external connector (e.g. PostHog) |\n\n### Request format\n\n```json\n{\n  \"jsonrpc\": \"2.0\",\n  \"method\": \"decide_ab\",\n  \"params\": {\n    \"mode\": \"frequentist\",\n    \"input\": {\n      \"control_conversions\": 100,\n      \"control_total\": 5000,\n      \"variant_conversions\": 130,\n      \"variant_total\": 5000\n    }\n  },\n  \"id\": 1\n}\n```\n\n### Unified response schema\n\n```json\n{\n  \"decision\": \"ship|keep_running|reject|escalate\",\n  \"recommended_next_action\": \"Deploy variant — statistical significance achieved with positive lift.\",\n  \"selected_method\": \"ab_test|bayesian_ab|did|planning\",\n  \"selection_reason\": \"Why this method was chosen\",\n  \"confidence\": \"high|medium|low\",\n  \"effect_summary\": \"Estimated lift: +30.00% (positive)\",\n  \"warnings\": [{\"code\": \"LOW_TRAFFIC\", \"message\": \"...\", \"severity\": \"info\"}],\n  \"limitations\": [\"No multiple testing correction applied\"],\n  \"audit_summary\": \"ab_test: Decision\",\n  \"source_metadata\": {\"connector\": \"langsmith\", \"dataset_id\": \"ds-001\"},\n  \"internal_result\": { ... }\n}\n```\n\n### Error format\n\n```json\n{\n  \"code\": \"VALIDATION_ERROR\",\n  \"message\": \"Invalid A/B test inputs\",\n  \"data\": {\n    \"details\": [{\"field\": \"control_total\", \"issue\": \"must be >= 1\"}],\n    \"request_id\": null\n  }\n}\n```\n\n---\n\n## Commands\n\n### Easy-mode Dispatcher (`decide`)\n\n**Don't know which method you need?** `decide` auto-detects from your input fields — no need to pick the right command:\n\n```bash\n# A/B test (auto-detected)\nPYTHONPATH=. python3 -m src.cli decide --control 100/5000 --variant 130/5000\nPYTHONPATH=. python3 -m src.cli decide --control 100/5000 --variant 130/5000 --format text\n\n# Bayesian A/B (--bayesian flag)\nPYTHONPATH=. python3 -m src.cli decide --control 100/5000 --variant 130/5000 --bayesian\n\n# DiD / staged rollout (auto-detected from pre/post treated fields)\nPYTHONPATH=. python3 -m src.cli decide --pre-control 1000 --post-control 1200 --pre-treated 200 --post-treated 280\n\n# Experiment planning (auto-detected from --baseline + --mde)\nPYTHONPATH=. python3 -m src.cli decide --baseline 0.05 --mde 10 --traffic 10000\n```\n\n\n**Auto-detection matrix:**\n| You provide... | It runs... |\n|---|---|\n| `--control` + `--variant` | Frequentist A/B |\n| `--control` + `--variant` + `--bayesian` | Bayesian A/B |\n| `--pre-control` + `--post-control` + `--pre-treated` + `--post-treated` | DiD (Difference-in-Differences) |\n| `--baseline` + `--mde` | Experiment planning |\n\n**JSON-RPC:** `{\"jsonrpc\":\"2.0\",\"method\":\"decide\",\"params\":{...fields...},\"id\":\"1\"}` — same auto-detection over stdio or HTTP.\n\n---\n\n### Experiment Planning (ab_plan)\n\nEstimate required sample size, duration, and feasibility before running an experiment:\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli plan --baseline 0.02 --mde 5 --traffic 5000\n```\n\n**Parameters:**\n- `--baseline` (required): Baseline conversion rate (e.g., `0.02` for 2%)\n- `--mde` (required): Minimum detectable effect as % lift (e.g., `5` for 5% lift)\n- `--traffic` (required): Daily traffic per arm\n- `--confidence` (default `0.95`): Confidence level\n- `--power` (default `0.8`): Statistical power\n- `--allocation`: `equal` (default) or `custom`\n- `--allocation-ratio`: Custom ratio when allocation=custom (e.g., `0.3/0.7`)\n- `--format`: `json` (default) or `text`\n\n**Planning output:**\n```json\n{\n  \"mode\": \"planning\",\n  \"recommendation\": {\n    \"decision\": \"feasible|slow|not_recommended\",\n    \"confidence\": \"high|medium|low\",\n    \"summary\": \"...\"\n  },\n  \"planning\": {\n    \"required_sample_per_arm\": 182934,\n    \"total_required\": 365868,\n    \"estimated_days\": 36.6,\n    \"feasibility\": \"slow\",\n    \"allocation_used\": {\"control\": 0.5, \"variant\": 0.5}\n  },\n  \"warnings\": [...]\n}\n```\n\n**Feasibility thresholds:**\n- `feasible`: ≤14 days\n- `slow`: 15–60 days\n- `not_recommended`: >60 days\n\n### A/B Test Analysis (Frequentist)\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli ab --control 100/5000 --variant 130/5000\n```\n\n**Parameters:**\n- `--control`: Control group conversions/total (e.g., `100/5000`)\n- `--variant`: Variant group conversions/total (e.g., `130/5000`)\n- `--name`: Variant name (optional, default: `variant_1`)\n- `--format`: Output format `json` (default) or `text`\n\n**Sequential / Early Stopping (optional):**\n- `--sequential/--no-sequential`: Enable sequential early stopping evaluation\n- `--experiment-start`, `--experiment-end`: ISO 8601 timestamps for runtime calculation\n- `--min-runtime-days` (default 7): Minimum days before early stop is considered\n- `--min-sample-per-arm` (default 2000): Minimum sample per arm before early stop\n- `--early-stop-p` (default 0.01): p-value threshold for early stop\n- `--max-runtime-days`: Hard cap; escalates if exceeded without strong result\n\n**Trigger logic:** Both min-runtime AND min-sample-per-arm must be met, AND p-value below `--early-stop-p`. Max runtime exceeded always escalates.\n```bash\nPYTHONPATH=. python3 -m src.cli ab --control 100/5000 --variant 130/5000\n```\n\n**Example Output:**\n```json\n{\n  \"schema_version\": \"0.8.0\",\n  \"mode\": \"ab_test\",\n  \"recommendation\": {\n    \"decision\": \"ship\",\n    \"confidence\": \"medium\",\n    \"summary\": \"Variant performs 30.00% better (p=0.0454). Ship it.\",\n    \"primary_metricLift\": 30.0,\n    \"p_value\": 0.045361\n  },\n  \"statistics\": {\n    \"control_rate\": 0.02,\n    \"variant_rate\": 0.026,\n    \"relative_lift_pct\": 30.0,\n    \"z_score\": 2.0013,\n    \"p_value\": 0.045361,\n    \"lift_ci_95\": [0.000124, 0.011876],\n    \"relative_lift_ci_95\": [0.619, 59.381]\n  },\n  \"traffic_stats\": {\n    \"control_size\": 5000,\n    \"variant_size\": 5000,\n    \"total_size\": 10000\n  },\n  \"warnings\": [],\n  \"next_steps\": [\"Deploy variant\", \"Monitor over time for regression\"],\n  \"audit\": {\n    \"decision_path\": [\n      {\"step\": \"Input validation\", \"passed\": true},\n      {\"step\": \"Traffic check\", \"passed\": true},\n      {\"step\": \"Conversion rate calculation\", \"passed\": true},\n      {\"step\": \"Statistical significance test\", \"passed\": true},\n      {\"step\": \"Effect size check\", \"passed\": true},\n      {\"step\": \"Decision\", \"passed\": true}\n    ]\n  }\n}\n```\n\n### Bayesian A/B Test\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli bayes --control 100/5000 --variant 130/5000\n```\n\n**Uses Beta-Binomial conjugate model with Jeffreys prior.**\n- Prior: Beta(0.5, 0.5) — uninformative\n- Posterior: Beta(α + successes, β + failures)\n- Decision via Monte Carlo simulation (20k samples)\n- Thresholds: P(variant wins) ≥ 0.95 → ship, ≤ 0.05 → reject\n\n**Parameters:**\n- `--control`, `--variant`: Conversions/total (same as `ab`)\n- `--name`: Variant name\n- `--format`: `json` (default) or `text`\n- `--samples`: Monte Carlo samples (default: 20000)\n\n**Example output:**\n```json\n{\n  \"schema_version\": \"0.8.0\",\n  \"timestamp\": \"2026-05-06T13:00:00.000Z\",\n  \"mode\": \"bayesian_ab\",\n  \"recommendation\": {\n    \"decision\": \"ship\",\n    \"confidence\": \"medium\",\n    \"summary\": \"Variant wins with P(better)=0.976. Median lift=30.10%. Ship.\",\n    \"primary_metricLift\": 30.10,\n    \"p_value\": 0.9758\n  },\n  \"statistics\": {\n    \"control_rate_observed\": 0.0200,\n    \"variant_rate_observed\": 0.0260,\n    \"relative_lift_pct\": 30.00,\n    \"posterior_control\": {\"alpha\": 100.5, \"beta\": 4900.5, \"mean\": 0.0201},\n    \"posterior_variant\": {\"alpha\": 130.5, \"beta\": 4870.5, \"mean\": 0.0261},\n    \"p_variant_wins\": 0.9758,\n    \"p_control_wins\": 0.0239,\n    \"p_tie\": 0.0003,\n    \"lift_median_pct\": 30.10,\n    \"lift_95ci_pct\": [0.20, 69.15],\n    \"expected_lift_hdi_95\": [0.0004, 0.0014],\n    \"relative_lift_hdi_95\": [2.00, 70.00],\n    \"monte_carlo_samples\": 20000,\n    \"prior_used\": {\"alpha\": 0.5, \"beta\": 0.5, \"type\": \"Jeffreys\"}\n  },\n  \"traffic_stats\": {\n    \"control_size\": 5000,\n    \"variant_size\": 5000,\n    \"total_size\": 10000\n  },\n  \"warnings\": [],\n  \"next_steps\": [\"Deploy variant\", \"Monitor for regression\"],\n  \"audit\": {\n    \"experiment_type\": \"bayesian_ab\",\n    \"thresholds_applied\": {\"ship\": 0.95, \"reject\": 0.05},\n    \"assumptions\": [\"Independent observations between groups\", \"No selection bias in group assignment\", \"Jeffrey's prior is appropriate for conversion rates\"],\n    \"limitations\": [\"Monte Carlo simulation has finite sampling error\", \"No multiple testing correction applied\"]\n  },\n  \"inputs\": {\n    \"control_conversions\": 100,\n    \"control_total\": 5000,\n    \"variant_conversions\": 130,\n    \"variant_total\": 5000\n  }\n}\n```\n\nAccess fields via `.` attribute (e.g. `result.recommendation.decision`) or `.model_dump()` for dict. Serialize with `.model_dump_json()`.\n\n**When to use Bayesian vs Frequentist:**\n\n### DiD Analysis\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli did --pre-control 1000 --post-control 1100 --pre-treated 900 --post-treated 1150\n```\n\n**Parameters:**\n- `--pre-control`: Control group metric before treatment\n- `--post-control`: Control group metric after treatment\n- `--pre-treated`: Treated group metric before treatment\n- `--post-treated`: Treated group metric after treatment\n- `--n-bootstrap` (default 2000, range 500–10000): Number of bootstrap resamples for DiD CI\n\n### Cohort / Segment Breakdown\n\nWhen an aggregate A/B or DiD result is inconclusive, break down results by user segment to find hidden signals:\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli cohort-breakdown --file segments.json\n```\n\n**Input format (JSON):**\n```json\n{\n  \"experiment_id\": \"checkout-v3\",\n  \"metric\": \"conversion_rate\",\n  \"prior_result_id\": \"dec_20260501_001\",\n  \"prior_decision\": \"wait\",\n  \"segments\": [\n    {\n      \"segment_name\": \"new_users\",\n      \"segment_definition_note\": \"Users registered within last 30 days\",\n      \"control_conversions\": 21,\n      \"control_total\": 1000,\n      \"variant_conversions\": 67,\n      \"variant_total\": 1000\n    },\n    {\n      \"segment_name\": \"returning_users\",\n      \"segment_definition_note\": \"Users registered more than 30 days ago\",\n      \"control_conversions\": 220,\n      \"control_total\": 4000,\n      \"variant_conversions\": 228,\n      \"variant_total\": 4000\n    }\n  ]\n}\n```\n\n**Input format (CSV alternative):**\n```\nsegment_name,segment_definition_note,arm,conversions,total\nnew_users,Users registered within last 30 days,control,21,1000\nnew_users,Users registered within last 30 days,variant,67,1000\nreturning_users,Users registered more than 30 days ago,control,220,4000\nreturning_users,Users registered more than 30 days ago,variant,228,4000\n```\n\n**Parameters:**\n- `--file`: Path to JSON or CSV segment file\n- `--json`: JSON input string (alternative to `--file`)\n- `--format`: Output format `json` (default) or `text`\n- `--save`: Save result to experiment history\n\n**Multiple comparison correction:**\n- 4+ segments: Benjamini-Hochberg FDR correction applied automatically\n- 5+ segments: Also offers Bonferroni as alternative via `--method bonferroni`\n\n**Example output:**\n```json\n{\n  \"method\": \"experiment_cohort_breakdown\",\n  \"cohort_decision_override\": true,\n  \"cohort_override_reason\": \"Strong positive signal in 'new_users' (lift=219.0%, adj-p=0.0000) contradicts aggregate decision 'wait'\",\n  \"interaction_flag\": false,\n  \"segments\": [\n    {\n      \"segment_name\": \"new_users\",\n      \"control_rate\": 0.021,\n      \"variant_rate\": 0.067,\n      \"relative_lift_pct\": 219.05,\n      \"p_value_raw\": 0.0000,\n      \"p_value_adjusted\": 0.0000,\n      \"decision\": \"strongly_positive\",\n      \"priority_rank\": 1\n    }\n  ],\n  \"priority_ranking\": [\n    {\"rank\": 1, \"segment\": \"new_users\", \"rationale\": \"Strong positive effect (lift=219.1%, adj-p=0.0000). Highest priority.\"}\n  ],\n  \"summary\": \"new_users drives the effect. 1 segment(s) positive.\",\n  \"recommended_next_action\": \"targeted_rollout\"\n}\n```\n\n**When to use cohort breakdown:**\n- Aggregate A/B result is `keep_running` or `escalate` — segment analysis may reveal a hidden signal\n- One segment is strongly positive while another is strongly negative (interaction flag)\n- You want to ship only to specific segments rather than all users\n\n**Key features:**\n- Per-segment two-proportion z-test with 95% confidence\n- Benjamini-Hochberg FDR correction for 4+ segments (controls false-discovery rate)\n- Priority ranking by absolute lift magnitude\n- Cohort decision override: fires when a segment contradicts the aggregate decision\n- Interaction flag: triggered when segments show opposing strongly-significant directions\n\n### Decision Audit\n\nReconstruct and explain a previous decision:\n\n```bash\n# Save result to file\nPYTHONPATH=. python3 -m src.cli ab --control 100/5000 --variant 130/5000 > /tmp/result.json\n\n# Audit it (human-readable)\nPYTHONPATH=. python3 -m src.cli audit /tmp/result.json --format text\n\n# Audit with experiment maturity assessment\nPYTHONPATH=. python3 -m src.cli audit /tmp/result.json --maturity\n```\n\n**Maturity assessment** (with `--maturity` flag):\n- Scores experiments 0–100 across 8 checks\n- Labels: `mature` (≥90), `adequate` (≥70), `immature` (≥50), `inadequate` (<50)\n- Checks: decision path completeness, critical warnings, limitations documented, traffic sufficiency, confidence level, step documentation\n\n**Example audit output:**\n```\n-- DECISION PATH --\n1. Input validation [✓]\n   control_total: 5000, variant_total: 5000\n2. Traffic check [✓]\n   control_size: 5000, min_required: 1000\n3. Conversion rate calculation [✓]\n   control_rate: 0.02, variant_rate: 0.026\n4. Statistical significance test [✓]\n   p_value: 0.045361, alpha: 0.05\n5. Effect size check [✓]\n   lift_pct: 30.0, threshold: 1\n6. Decision [✓]\n   decision: ship, confidence: medium\n\n-- FINAL DECISION --\n  Decision: SHIP\n```\n\n### Experiment History & Persistence\n\nAll commands support `--save` to persist results to local SQLite history:\n\n```bash\n# Run and save in one step\nPYTHONPATH=. python3 -m src.cli ab --control 100/5000 --variant 130/5000 --save\nPYTHONPATH=. python3 -m src.cli did --pre-control 1000 --post-control 1100 --pre-treated 900 --post-treated 1150 --save\nPYTHONPATH=. python3 -m src.cli plan --baseline 0.02 --mde 5 --traffic 5000 --save\n```\n\n**History commands:**\n\n```bash\n# List recent experiments\nPYTHONPATH=. python3 -m src.cli history\nPYTHONPATH=. python3 -m src.cli history --mode ab_test --limit 10\n\n# Compare multiple experiments by ID\nPYTHONPATH=. python3 -m src.cli compare 1 2 3\n\n# Save a prior JSON result file to history\nPYTHONPATH=. python3 -m src.cli save /tmp/result.json --name \"checkout-v3-test\"\n```\n\n**History output example:**\n```\nID    Date       Mode       Decision   Lift     P-value  Summary\n--------------------------------------------------------------------\n3     2026-04-30 did        ship       16.67    -        Treatment effect is 150.00...\n2     2026-04-30 ab_test    escalate   6.25     0.6947   Results not conclusive...\n1     2026-04-30 ab_test    ship       30.00    0.0454   Variant performs 30.00%...\n```\n\n**Compare output example:**\n```\nEXPERIMENT COMPARISON\n==================================================\nExperiments compared: 3\n\nSummary by decision:\n  SHIP: 2 experiment(s)\n  ESCALATE: 1 experiment(s)\n\nSummary by mode:\n  ab_test: 2 experiment(s)\n  did: 1 experiment(s)\n\nLift summary: max=30.00%, min=6.25%, avg=17.64% (3 experiments)\n\nAttention needed: 2 experiments recommend ship. Review if they test the same metric.\n```\n\n**Persistence:**\n- SQLite DB stored at `~/.agent-causal/history.db`\n- All experiment modes supported: `ab_test`, `did`, `planning`\n- Full raw JSON preserved for audit reconstruction\n- Filter by mode, limit results, name experiments for later reference\n\n## Decision Reference\n\n| Decision | Meaning | When |\n|----------|---------|------|\n| `ship` | Deploy variant | p < 0.05 AND positive lift |\n| `keep_running` | Continue experiment | p < 0.3, trending positive |\n| `reject` | Do not deploy | p < 0.05 AND negative lift |\n| `escalate` | Needs human review | Not conclusive or critical warnings |\n| `targeted_rollout` | Ship to specific segment only | Strong signal in one segment, aggregate inconclusive |\n| `full_rollout` | Ship to all users | All segments positive |\n| `abandon_segment` | Do not ship to specific segment | Strong negative in one segment despite aggregate ship |\n| `confirm_rejection` | Confirm abandonment | All segments negative |\n\n## Python API\n\n```python\nimport sys\nsys.path.insert(0, '~/clawd/agent-causal-decision-tool')\n\nfrom src.ab_test import calculate_ab\n\nresult = calculate_ab({\n    \"control_conversions\": 100,\n    \"control_total\": 5000,\n    \"variant_conversions\": 130,\n    \"variant_total\": 5000\n})\n\nif result.recommendation.decision == \"ship\":\n    # Deploy variant\n    pass\n```\n\n## Warnings & Limitations\n\n- **LOW_TRAFFIC**: Sample size below 1000 per group\n- **SMALL_EFFECT**: Lift < 1%, may not be practically significant\n- **INCONCLUSIVE**: Result not statistically significant or strong enough to act on\n- **NOT_SIGNIFICANT**: Far from significant; consider stopping the experiment\n- **BORDERLINE_P_VALUE**: p-value between 0.05 and 0.10 — weak evidence, not conclusive\n- **CORRECTION_CONSERVATIVE**: Multiple comparison correction applied; may increase false negatives\n- **SEQUENTIAL_EARLY_STOP**: Experiment stopped early via sequential testing; interpret with caution\n- **SEQUENTIAL_CONDITIONS_NOT_MET**: Early stopping conditions not met; normal decision applied\n- **MAX_RUNTIME_EXCEEDED**: Hard runtime cap exceeded without strong result; escalating\n- **ZERO_BASELINE**: Pre-period values cannot be zero for reliable DiD\n- **PARALLEL_TRENDS_VIOLATED**: Control and treated groups show very different pre-to-post ratios (critical)\n- **PARALLEL_TRENDS_WEAK**: Ratios diverge somewhat; monitor closely\n- **BOTH_GROUPS_GREW**: Both groups grew; cannot separate treatment effect from time trend\n- **AGGREGATE_DATA**: Analysis on aggregated data; use individual-level data for robust inference\n- **AGGREGATE_DATA_DID**: DiD result with high caution; not equivalent to a randomized experiment\n- **SINGLE_PRE_PERIOD**: Only one pre-period observation; parallel trends cannot be assessed\n- **SMALL_SAMPLE**: Sample size small; estimates unreliable\n- **IMBALANCED_GROUPS**: Treatment/control group sizes very different; may bias DiD estimate\n- **LARGE_EFFECT_SMALL_SAMPLE**: Large effect estimate from small sample; prioritize replication\n- **PARALLEL_TRENDS_NO_DATA**: No pre-period count provided; parallel trends cannot be assessed\n- **BOOTSTRAP_CI_UNRELIABLE**: Bootstrap CI not computed — count < 100 or zero baseline\n- **BOOTSTRAP_CI_WIDE**: Bootstrap CI range > 2×|DiD estimate|; point estimate uncertain\n- **DID_CI_CROSSES_ZERO**: Bootstrap CI crosses zero; effect direction uncertain\n- **SLOW_EXPERIMENT**: Estimated duration > 30 days; seasonal effects may confound results\n- **INFEASIBLE_EXPERIMENT**: Duration too long; consider DiD instead\n- **SMALL_MDE**: MDE very small; may require impossibly large sample\n- **BASELINE_VERY_LOW**: Baseline rate < 0.5%; estimations may be unreliable\n- **BASELINE_NEAR_ZERO**: Baseline rate < 0.1%; do not run experiment without careful review\n- **PRIOR_DOMINATES**: Very low total traffic; Jeffreys prior dominates posterior; interpret with caution\n- **CREDIBLE_INTERVAL_WIDE**: Bayesian credible interval is very wide; estimate uncertain\n\n## Schema Contract\n\nThe tool exposes a versioned schema contract for agent consumption:\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli schema\n```\n\nThis prints `schema.json` — a wrapper containing `schema_version`, `schema_coverage` (`ab`, `did`, `plan`, `bayes`), `schema_coverage_pending` (`cohort`), `severity_contract`, and `definitions` (JSON Schema from Pydantic models).\n\nAll output models include `schema_version` field injected from package metadata — never hardcoded.\n\n## Location\n\n- **GitHub:** https://github.com/ZhuMorris/agent-causal-decision-tool\n- **Local:** `~/clawd/agent-causal-decision-tool/`\n\n## Dependencies\n\n- Python 3.9+\n- click >= 8.1.0\n- scipy >= 1.11.0\n- numpy >= 1.26.0\n- pydantic >= 2.10.0\n\n## External Connectors\n\nFetch experiment data directly from external sources. The `connect` action normalizes external data into the internal experiment schema before running a decision.\n\n### PostHog\n\n```bash\n# Health check (validates credentials, no data fetched)\nPYTHONPATH=. python3 -m src.cli connect posthog --dry-run\n\n# Fetch experiment and print normalized data\nPYTHONPATH=. python3 -m src.cli connect posthog --experiment-id <id>\n\n# Fetch and run through decision workflow automatically\nPYTHONPATH=. python3 -m src.cli connect posthog --experiment-id <id> --decide\n\n# JSON-RPC call\n{\"jsonrpc\":\"2.0\",\"method\":\"connect\",\"params\":{\"source\":\"posthog\",\"experiment_id\":\"<id>\"},\"id\":\"1\"}\n```\n\n**Environment / config:**\n- `POSTHOG_API_KEY` + `POSTHOG_PROJECT_ID` env vars, OR\n- `~/.posthogrc` with `api_key`, `project_id`, `instance_url` fields\n\n**Connector result schema:**\n```json\n{\n  \"data\": { \"control_conversions\": 120, \"control_total\": 5000, \"variant_conversions\": 145, \"variant_total\": 5000 },\n  \"source_metadata\": { \"connector\": \"posthog\", \"experiment_id\": \"...\", \"fetch_timestamp\": \"...\" },\n  \"warnings\": []\n}\n```\n\n**Errors:**\n- `INSUFFICIENT_DATA` — experiment found but missing required fields\n- `ConnectorAuthError` — invalid/missing API key\n- `ConnectorNotFoundError` — experiment not found\n\nFile v0.10.3:_meta.json\n\n{\n  \"ownerId\": \"kn73d2hycfhnph00er7p4gsn8h85r7cy\",\n  \"slug\": \"agent-causal\",\n  \"version\": \"0.10.3\",\n  \"publishedAt\": 1778154416976\n}\n\nFile v0.10.3:skill-card.md\n\n## Description:\n\nAgent Causal Decision Tool returns structured JSON decisions, key statistics, and audit trails that help agents decide whether to ship, continue, reject, or escalate experiment changes across A/B tests, Bayesian A/B tests, Difference-in-Differences, cohort analysis, planning, and sequential early stopping.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[zhumorris](https://clawhub.ai/user/zhumorris)\n\n### License/Terms of Use:\n\nApache-2.0\n\n## Use Case:\n\nDevelopers, product analysts, and AI agents use this skill to evaluate experiment or rollout summaries and produce defensible ship, continue, reject, targeted rollout, or review decisions with statistics and audit records.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Installation can fetch remote code, and the security evidence flags the mutable git clone install path as a review concern.\n\nMitigation: Prefer a pinned, checksummed release or isolated environment and avoid the mutable git clone install path.\n\nRisk: HTTP JSON-RPC mode can expose a network-facing service if bound or routed beyond local development use.\n\nMitigation: Keep HTTP mode on localhost unless independent authentication and transport controls are added.\n\nRisk: The PostHog connector makes outbound HTTPS requests and uses API credentials when explicitly invoked.\n\nMitigation: Use read-only PostHog tokens with minimal scopes and keep credentials in environment variables or local config rather than prompts or logs.\n\nRisk: Experiment recommendations can be misleading when inputs are underpowered, aggregate-only, or violate method assumptions.\n\nMitigation: Review emitted warnings, confidence, limitations, and audit records before acting on ship, reject, or targeted rollout recommendations.\n\n## Reference(s):\n\n- [Agent Causal source repository](https://github.com/ZhuMorris/agent-causal-decision-tool)\n- [Agent Causal ClawHub skill page](https://clawhub.ai/zhumorris/skills/agent-causal)\n\n## Skill Output:\n\n**Output Type(s):** [JSON, Text, Shell commands, Configuration, Guidance]\n\n**Output Format:** [Structured JSON decision objects, text summaries, and Markdown guidance with command examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May persist local SQLite audit and experiment history records when save or history commands are used.]\n\n## Skill Version(s):\n\n0.10.3 (source: artifact/_meta.json, SKILL.md metadata.openclaw.version, evidence.release.version)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v0.10.2: 2 files, 10502 bytes\n\nFiles: _meta.json (132b), SKILL.md (28118b)\n\nFile v0.10.2:SKILL.md\n\n---\nname: agent-causal\ndescription: \"Agent Causal Decision Tool helps you and your AI agents answer one question from experiment data: should we ship this change, keep running the test, or roll it back? Returns structured JSON decisions, key statistics, and audit trails from A/B tests (frequentist + Bayesian), DiD, cohort/segment analysis, and sequential early stopping.\"\nmetadata:\n  openclaw:\n    category: data-science\n    version: \"0.10.2\"\n    license: Apache-2.0\n    tools: [exec]\n    requires:\n      bins: [python3, git, pip]\n      python_packages: [click, scipy, numpy, pydantic]\n    source: https://github.com/ZhuMorris/agent-causal-decision-tool\n---\n\n# Agent Causal Decision Tool\n\nA causal decision and audit tool for AI agents. Evaluate product changes using A/B testing, Difference-in-Differences, and sequential early stopping.\n\n**Source:** https://github.com/ZhuMorris/agent-causal-decision-tool\n\n## What is this?\n\nAgent Causal Decision Tool helps you and your AI agents answer one question from experiment data: \"should we ship this change, keep running the test, or roll it back?\" It takes in simple A/B or rollout summaries and returns a structured JSON decision, key statistics, and an audit record you can store or review later.\n\nRather than being a full experimentation platform, it is a **decision engine**. You bring the data (from your logs, BI tool, or CSV); it handles the stats, decision logic, and audit trail.\n\n### Why it exists\n\nIn many teams, experiment decisions happen in ad hoc spreadsheets or dashboards. People glance at lift, argue about whether the sample size is enough, and sometimes ship features based on noisy or biased results. Agents make this worse if they are wired to react to any small uplift they see.\n\nThis tool wraps a few standard methods into one consistent, agent‑friendly interface:\n\n- **Easy-mode dispatcher (`decide`)** — no need to know which statistical method to use. Paste your numbers and it auto-selects A/B, Bayesian, DiD, or planning from your input fields.\n- **Frequentist A/B testing** for classic \"control vs variant\" questions.\n- **Bayesian A/B testing** when you want answers like \"there is a 93% chance B is better than A\" instead of only p‑values.\n- **Difference‑in‑differences (DiD)** for quasi‑experiments like staged rollouts or region‑based launches where you cannot randomize perfectly.\n- **Cohort / segment breakdown** when an aggregate result is inconclusive — you can slice by user segment to find hidden signals, with Benjamini-Hochberg correction for 4+ segments.\n- **Planning and power checks** so you can see if a test is realistic before you start it.\n- **Decision audit** so humans can see what the agent did, why it did it, and how strong the evidence really was.\n- **External connectors** — pull experiment data directly from PostHog, normalize it, and run a decision in one step. No manual export needed.\n\nThe goal is not to replace your analytics stack, but to give agents a small, reliable decision block they can call inside workflows.\n\n### When to use it\n\nUse this tool whenever you or your agents have experiment or rollout results and need a decision you can defend:\n\n- You ran an A/B test and want to know whether to ship, keep running, or reject the variant.\n- You're not sure which method to use — let `decide` auto-detect from your numbers.\n- You ran an A/B test and it was inconclusive — you want to know if a specific user segment is driving (or diluting) the effect.\n- You rolled out a feature to one region or cohort first and want a DiD estimate of impact compared to a similar control group.\n- You prefer a Bayesian summary (\"95% chance B is better; expected lift 3–5%\") to drive thresholds in automated workflows.\n- You need an audit trail with experiment period, traffic size, assumptions, thresholds, and warnings so product, data, or risk teams can review agent decisions later.\n- You want to plan an experiment (sample size, minimum detectable effect, expected duration) or compare current results to previous experiments to see which wins are robust.\n- Your experiment data lives in PostHog — you want to fetch, normalize, and decide without any manual CSV export.\n\n---\n\n## Security Model\n\nThis skill runs as a **local CLI tool only**. After one-time setup, it requires no network access.\n\n**Setup (one-time, before first use):**\n```bash\n# Download the release tarball — no git clone needed\ncurl -sL https://github.com/ZhuMorris/agent-causal-decision-tool/archive/refs/tags/v0.10.1.tar.gz -o agent-causal.tar.gz\ntar -xzf agent-causal.tar.gz\npip install agent-causal-decision-tool-0.10.1/ -q\n```\n\n\nAfter installation, the `agent-causal` command is available locally. The tool performs only local statistical calculations.\n\n**When network access occurs:** The PostHog connector (`agent-causal connect posthog`) makes outbound HTTPS requests to your PostHog instance only when explicitly invoked — never automatically. If you do not use the connector, no outbound network access is needed at any point.\n\n**No runtime network access during analysis.** The decision engine, audit, and cohort analysis do not make outbound requests.\n**Tools used:** `exec` (for running the `agent-causal` CLI commands you specify). Commands are fully hardcoded with no user-supplied strings interpolated into shell execution.\n**PostHog token scope:** Use a read-only API token with minimal scopes. Do not use tokens with write or admin permissions.\n**Credential handling:** PostHog API credentials are read from env vars (`POSTHOG_API_KEY`/`POSTHOG_PROJECT_ID`) or a local `~/.posthogrc` file — never hardcoded or logged.\n\n---\n\n## Setup\n\nBefore using this skill, install the tool (one-time):\n\n```bash\ngit clone https://github.com/ZhuMorris/agent-causal-decision-tool.git ~/clawd/agent-causal-decision-tool\npip install ~/clawd/agent-causal-decision-tool -q\n```\n\nAfter this, `agent-causal` is available as a local command. No further network access is required.\n\n---\n\n## Agent-Native Actions (JSON-RPC 2.0)\n\nFor AI agent integrations, Agent Causal exposes a JSON-RPC 2.0 API over both stdio and HTTP.\n\n### Stdio mode (for OpenClaw, Codex, Claude Code)\n\n```bash\npython -m src.api stdio\n```\n\n### HTTP mode\n\n```bash\npython -m src.api http --port 8000\n```\n\n### Actions\n\n| Action | Description |\n|--------|-------------|\n| `decide` | **Easy-mode dispatcher** — auto-selects A/B, Bayesian, DiD, or planning from your input fields |\n| `decide_ab` | Frequentist A/B test (`mode: frequentist`) or Bayesian (`mode: bayesian`) |\n| `decide_rollout` | DiD for staged rollouts / quasi-experiments |\n| `plan_test` | Experiment planning (sample size, MDE, feasibility) |\n| `audit_result` | Full audit of a stored result by ID |\n| `save_result` | Persist a decision result to SQLite history |\n| `get_result` | Retrieve a stored result by ID |\n| `compare_results` | Compare multiple stored experiments |\n| `connect` | Fetch experiment data from an external connector (e.g. PostHog) |\n\n### Request format\n\n```json\n{\n  \"jsonrpc\": \"2.0\",\n  \"method\": \"decide_ab\",\n  \"params\": {\n    \"mode\": \"frequentist\",\n    \"input\": {\n      \"control_conversions\": 100,\n      \"control_total\": 5000,\n      \"variant_conversions\": 130,\n      \"variant_total\": 5000\n    }\n  },\n  \"id\": 1\n}\n```\n\n### Unified response schema\n\n```json\n{\n  \"decision\": \"ship|keep_running|reject|escalate\",\n  \"recommended_next_action\": \"Deploy variant — statistical significance achieved with positive lift.\",\n  \"selected_method\": \"ab_test|bayesian_ab|did|planning\",\n  \"selection_reason\": \"Why this method was chosen\",\n  \"confidence\": \"high|medium|low\",\n  \"effect_summary\": \"Estimated lift: +30.00% (positive)\",\n  \"warnings\": [{\"code\": \"LOW_TRAFFIC\", \"message\": \"...\", \"severity\": \"info\"}],\n  \"limitations\": [\"No multiple testing correction applied\"],\n  \"audit_summary\": \"ab_test: Decision\",\n  \"source_metadata\": {\"connector\": \"langsmith\", \"dataset_id\": \"ds-001\"},\n  \"internal_result\": { ... }\n}\n```\n\n### Error format\n\n```json\n{\n  \"code\": \"VALIDATION_ERROR\",\n  \"message\": \"Invalid A/B test inputs\",\n  \"data\": {\n    \"details\": [{\"field\": \"control_total\", \"issue\": \"must be >= 1\"}],\n    \"request_id\": null\n  }\n}\n```\n\n---\n\n## Commands\n\n### Easy-mode Dispatcher (`decide`)\n\n**Don't know which method you need?** `decide` auto-detects from your input fields — no need to pick the right command:\n\n```bash\n# A/B test (auto-detected)\nPYTHONPATH=. python3 -m src.cli decide --control 100/5000 --variant 130/5000\nPYTHONPATH=. python3 -m src.cli decide --control 100/5000 --variant 130/5000 --format text\n\n# Bayesian A/B (--bayesian flag)\nPYTHONPATH=. python3 -m src.cli decide --control 100/5000 --variant 130/5000 --bayesian\n\n# DiD / staged rollout (auto-detected from pre/post treated fields)\nPYTHONPATH=. python3 -m src.cli decide --pre-control 1000 --post-control 1200 --pre-treated 200 --post-treated 280\n\n# Experiment planning (auto-detected from --baseline + --mde)\nPYTHONPATH=. python3 -m src.cli decide --baseline 0.05 --mde 10 --traffic 10000\n```\n\n\n**Auto-detection matrix:**\n| You provide... | It runs... |\n|---|---|\n| `--control` + `--variant` | Frequentist A/B |\n| `--control` + `--variant` + `--bayesian` | Bayesian A/B |\n| `--pre-control` + `--post-control` + `--pre-treated` + `--post-treated` | DiD (Difference-in-Differences) |\n| `--baseline` + `--mde` | Experiment planning |\n\n**JSON-RPC:** `{\"jsonrpc\":\"2.0\",\"method\":\"decide\",\"params\":{...fields...},\"id\":\"1\"}` — same auto-detection over stdio or HTTP.\n\n---\n\n### Experiment Planning (ab_plan)\n\nEstimate required sample size, duration, and feasibility before running an experiment:\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli plan --baseline 0.02 --mde 5 --traffic 5000\n```\n\n**Parameters:**\n- `--baseline` (required): Baseline conversion rate (e.g., `0.02` for 2%)\n- `--mde` (required): Minimum detectable effect as % lift (e.g., `5` for 5% lift)\n- `--traffic` (required): Daily traffic per arm\n- `--confidence` (default `0.95`): Confidence level\n- `--power` (default `0.8`): Statistical power\n- `--allocation`: `equal` (default) or `custom`\n- `--allocation-ratio`: Custom ratio when allocation=custom (e.g., `0.3/0.7`)\n- `--format`: `json` (default) or `text`\n\n**Planning output:**\n```json\n{\n  \"mode\": \"planning\",\n  \"recommendation\": {\n    \"decision\": \"feasible|slow|not_recommended\",\n    \"confidence\": \"high|medium|low\",\n    \"summary\": \"...\"\n  },\n  \"planning\": {\n    \"required_sample_per_arm\": 182934,\n    \"total_required\": 365868,\n    \"estimated_days\": 36.6,\n    \"feasibility\": \"slow\",\n    \"allocation_used\": {\"control\": 0.5, \"variant\": 0.5}\n  },\n  \"warnings\": [...]\n}\n```\n\n**Feasibility thresholds:**\n- `feasible`: ≤14 days\n- `slow`: 15–60 days\n- `not_recommended`: >60 days\n\n### A/B Test Analysis (Frequentist)\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli ab --control 100/5000 --variant 130/5000\n```\n\n**Parameters:**\n- `--control`: Control group conversions/total (e.g., `100/5000`)\n- `--variant`: Variant group conversions/total (e.g., `130/5000`)\n- `--name`: Variant name (optional, default: `variant_1`)\n- `--format`: Output format `json` (default) or `text`\n\n**Sequential / Early Stopping (optional):**\n- `--sequential/--no-sequential`: Enable sequential early stopping evaluation\n- `--experiment-start`, `--experiment-end`: ISO 8601 timestamps for runtime calculation\n- `--min-runtime-days` (default 7): Minimum days before early stop is considered\n- `--min-sample-per-arm` (default 2000): Minimum sample per arm before early stop\n- `--early-stop-p` (default 0.01): p-value threshold for early stop\n- `--max-runtime-days`: Hard cap; escalates if exceeded without strong result\n\n**Trigger logic:** Both min-runtime AND min-sample-per-arm must be met, AND p-value below `--early-stop-p`. Max runtime exceeded always escalates.\n```bash\nPYTHONPATH=. python3 -m src.cli ab --control 100/5000 --variant 130/5000\n```\n\n**Example Output:**\n```json\n{\n  \"schema_version\": \"0.8.0\",\n  \"mode\": \"ab_test\",\n  \"recommendation\": {\n    \"decision\": \"ship\",\n    \"confidence\": \"medium\",\n    \"summary\": \"Variant performs 30.00% better (p=0.0454). Ship it.\",\n    \"primary_metricLift\": 30.0,\n    \"p_value\": 0.045361\n  },\n  \"statistics\": {\n    \"control_rate\": 0.02,\n    \"variant_rate\": 0.026,\n    \"relative_lift_pct\": 30.0,\n    \"z_score\": 2.0013,\n    \"p_value\": 0.045361,\n    \"lift_ci_95\": [0.000124, 0.011876],\n    \"relative_lift_ci_95\": [0.619, 59.381]\n  },\n  \"traffic_stats\": {\n    \"control_size\": 5000,\n    \"variant_size\": 5000,\n    \"total_size\": 10000\n  },\n  \"warnings\": [],\n  \"next_steps\": [\"Deploy variant\", \"Monitor over time for regression\"],\n  \"audit\": {\n    \"decision_path\": [\n      {\"step\": \"Input validation\", \"passed\": true},\n      {\"step\": \"Traffic check\", \"passed\": true},\n      {\"step\": \"Conversion rate calculation\", \"passed\": true},\n      {\"step\": \"Statistical significance test\", \"passed\": true},\n      {\"step\": \"Effect size check\", \"passed\": true},\n      {\"step\": \"Decision\", \"passed\": true}\n    ]\n  }\n}\n```\n\n### Bayesian A/B Test\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli bayes --control 100/5000 --variant 130/5000\n```\n\n**Uses Beta-Binomial conjugate model with Jeffreys prior.**\n- Prior: Beta(0.5, 0.5) — uninformative\n- Posterior: Beta(α + successes, β + failures)\n- Decision via Monte Carlo simulation (20k samples)\n- Thresholds: P(variant wins) ≥ 0.95 → ship, ≤ 0.05 → reject\n\n**Parameters:**\n- `--control`, `--variant`: Conversions/total (same as `ab`)\n- `--name`: Variant name\n- `--format`: `json` (default) or `text`\n- `--samples`: Monte Carlo samples (default: 20000)\n\n**Example output:**\n```json\n{\n  \"schema_version\": \"0.8.0\",\n  \"timestamp\": \"2026-05-06T13:00:00.000Z\",\n  \"mode\": \"bayesian_ab\",\n  \"recommendation\": {\n    \"decision\": \"ship\",\n    \"confidence\": \"medium\",\n    \"summary\": \"Variant wins with P(better)=0.976. Median lift=30.10%. Ship.\",\n    \"primary_metricLift\": 30.10,\n    \"p_value\": 0.9758\n  },\n  \"statistics\": {\n    \"control_rate_observed\": 0.0200,\n    \"variant_rate_observed\": 0.0260,\n    \"relative_lift_pct\": 30.00,\n    \"posterior_control\": {\"alpha\": 100.5, \"beta\": 4900.5, \"mean\": 0.0201},\n    \"posterior_variant\": {\"alpha\": 130.5, \"beta\": 4870.5, \"mean\": 0.0261},\n    \"p_variant_wins\": 0.9758,\n    \"p_control_wins\": 0.0239,\n    \"p_tie\": 0.0003,\n    \"lift_median_pct\": 30.10,\n    \"lift_95ci_pct\": [0.20, 69.15],\n    \"expected_lift_hdi_95\": [0.0004, 0.0014],\n    \"relative_lift_hdi_95\": [2.00, 70.00],\n    \"monte_carlo_samples\": 20000,\n    \"prior_used\": {\"alpha\": 0.5, \"beta\": 0.5, \"type\": \"Jeffreys\"}\n  },\n  \"traffic_stats\": {\n    \"control_size\": 5000,\n    \"variant_size\": 5000,\n    \"total_size\": 10000\n  },\n  \"warnings\": [],\n  \"next_steps\": [\"Deploy variant\", \"Monitor for regression\"],\n  \"audit\": {\n    \"experiment_type\": \"bayesian_ab\",\n    \"thresholds_applied\": {\"ship\": 0.95, \"reject\": 0.05},\n    \"assumptions\": [\"Independent observations between groups\", \"No selection bias in group assignment\", \"Jeffrey's prior is appropriate for conversion rates\"],\n    \"limitations\": [\"Monte Carlo simulation has finite sampling error\", \"No multiple testing correction applied\"]\n  },\n  \"inputs\": {\n    \"control_conversions\": 100,\n    \"control_total\": 5000,\n    \"variant_conversions\": 130,\n    \"variant_total\": 5000\n  }\n}\n```\n\nAccess fields via `.` attribute (e.g. `result.recommendation.decision`) or `.model_dump()` for dict. Serialize with `.model_dump_json()`.\n\n**When to use Bayesian vs Frequentist:**\n\n### DiD Analysis\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli did --pre-control 1000 --post-control 1100 --pre-treated 900 --post-treated 1150\n```\n\n**Parameters:**\n- `--pre-control`: Control group metric before treatment\n- `--post-control`: Control group metric after treatment\n- `--pre-treated`: Treated group metric before treatment\n- `--post-treated`: Treated group metric after treatment\n- `--n-bootstrap` (default 2000, range 500–10000): Number of bootstrap resamples for DiD CI\n\n### Cohort / Segment Breakdown\n\nWhen an aggregate A/B or DiD result is inconclusive, break down results by user segment to find hidden signals:\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli cohort-breakdown --file segments.json\n```\n\n**Input format (JSON):**\n```json\n{\n  \"experiment_id\": \"checkout-v3\",\n  \"metric\": \"conversion_rate\",\n  \"prior_result_id\": \"dec_20260501_001\",\n  \"prior_decision\": \"wait\",\n  \"segments\": [\n    {\n      \"segment_name\": \"new_users\",\n      \"segment_definition_note\": \"Users registered within last 30 days\",\n      \"control_conversions\": 21,\n      \"control_total\": 1000,\n      \"variant_conversions\": 67,\n      \"variant_total\": 1000\n    },\n    {\n      \"segment_name\": \"returning_users\",\n      \"segment_definition_note\": \"Users registered more than 30 days ago\",\n      \"control_conversions\": 220,\n      \"control_total\": 4000,\n      \"variant_conversions\": 228,\n      \"variant_total\": 4000\n    }\n  ]\n}\n```\n\n**Input format (CSV alternative):**\n```\nsegment_name,segment_definition_note,arm,conversions,total\nnew_users,Users registered within last 30 days,control,21,1000\nnew_users,Users registered within last 30 days,variant,67,1000\nreturning_users,Users registered more than 30 days ago,control,220,4000\nreturning_users,Users registered more than 30 days ago,variant,228,4000\n```\n\n**Parameters:**\n- `--file`: Path to JSON or CSV segment file\n- `--json`: JSON input string (alternative to `--file`)\n- `--format`: Output format `json` (default) or `text`\n- `--save`: Save result to experiment history\n\n**Multiple comparison correction:**\n- 4+ segments: Benjamini-Hochberg FDR correction applied automatically\n- 5+ segments: Also offers Bonferroni as alternative via `--method bonferroni`\n\n**Example output:**\n```json\n{\n  \"method\": \"experiment_cohort_breakdown\",\n  \"cohort_decision_override\": true,\n  \"cohort_override_reason\": \"Strong positive signal in 'new_users' (lift=219.0%, adj-p=0.0000) contradicts aggregate decision 'wait'\",\n  \"interaction_flag\": false,\n  \"segments\": [\n    {\n      \"segment_name\": \"new_users\",\n      \"control_rate\": 0.021,\n      \"variant_rate\": 0.067,\n      \"relative_lift_pct\": 219.05,\n      \"p_value_raw\": 0.0000,\n      \"p_value_adjusted\": 0.0000,\n      \"decision\": \"strongly_positive\",\n      \"priority_rank\": 1\n    }\n  ],\n  \"priority_ranking\": [\n    {\"rank\": 1, \"segment\": \"new_users\", \"rationale\": \"Strong positive effect (lift=219.1%, adj-p=0.0000). Highest priority.\"}\n  ],\n  \"summary\": \"new_users drives the effect. 1 segment(s) positive.\",\n  \"recommended_next_action\": \"targeted_rollout\"\n}\n```\n\n**When to use cohort breakdown:**\n- Aggregate A/B result is `keep_running` or `escalate` — segment analysis may reveal a hidden signal\n- One segment is strongly positive while another is strongly negative (interaction flag)\n- You want to ship only to specific segments rather than all users\n\n**Key features:**\n- Per-segment two-proportion z-test with 95% confidence\n- Benjamini-Hochberg FDR correction for 4+ segments (controls false-discovery rate)\n- Priority ranking by absolute lift magnitude\n- Cohort decision override: fires when a segment contradicts the aggregate decision\n- Interaction flag: triggered when segments show opposing strongly-significant directions\n\n### Decision Audit\n\nReconstruct and explain a previous decision:\n\n```bash\n# Save result to file\nPYTHONPATH=. python3 -m src.cli ab --control 100/5000 --variant 130/5000 > /tmp/result.json\n\n# Audit it (human-readable)\nPYTHONPATH=. python3 -m src.cli audit /tmp/result.json --format text\n\n# Audit with experiment maturity assessment\nPYTHONPATH=. python3 -m src.cli audit /tmp/result.json --maturity\n```\n\n**Maturity assessment** (with `--maturity` flag):\n- Scores experiments 0–100 across 8 checks\n- Labels: `mature` (≥90), `adequate` (≥70), `immature` (≥50), `inadequate` (<50)\n- Checks: decision path completeness, critical warnings, limitations documented, traffic sufficiency, confidence level, step documentation\n\n**Example audit output:**\n```\n-- DECISION PATH --\n1. Input validation [✓]\n   control_total: 5000, variant_total: 5000\n2. Traffic check [✓]\n   control_size: 5000, min_required: 1000\n3. Conversion rate calculation [✓]\n   control_rate: 0.02, variant_rate: 0.026\n4. Statistical significance test [✓]\n   p_value: 0.045361, alpha: 0.05\n5. Effect size check [✓]\n   lift_pct: 30.0, threshold: 1\n6. Decision [✓]\n   decision: ship, confidence: medium\n\n-- FINAL DECISION --\n  Decision: SHIP\n```\n\n### Experiment History & Persistence\n\nAll commands support `--save` to persist results to local SQLite history:\n\n```bash\n# Run and save in one step\nPYTHONPATH=. python3 -m src.cli ab --control 100/5000 --variant 130/5000 --save\nPYTHONPATH=. python3 -m src.cli did --pre-control 1000 --post-control 1100 --pre-treated 900 --post-treated 1150 --save\nPYTHONPATH=. python3 -m src.cli plan --baseline 0.02 --mde 5 --traffic 5000 --save\n```\n\n**History commands:**\n\n```bash\n# List recent experiments\nPYTHONPATH=. python3 -m src.cli history\nPYTHONPATH=. python3 -m src.cli history --mode ab_test --limit 10\n\n# Compare multiple experiments by ID\nPYTHONPATH=. python3 -m src.cli compare 1 2 3\n\n# Save a prior JSON result file to history\nPYTHONPATH=. python3 -m src.cli save /tmp/result.json --name \"checkout-v3-test\"\n```\n\n**History output example:**\n```\nID    Date       Mode       Decision   Lift     P-value  Summary\n--------------------------------------------------------------------\n3     2026-04-30 did        ship       16.67    -        Treatment effect is 150.00...\n2     2026-04-30 ab_test    escalate   6.25     0.6947   Results not conclusive...\n1     2026-04-30 ab_test    ship       30.00    0.0454   Variant performs 30.00%...\n```\n\n**Compare output example:**\n```\nEXPERIMENT COMPARISON\n==================================================\nExperiments compared: 3\n\nSummary by decision:\n  SHIP: 2 experiment(s)\n  ESCALATE: 1 experiment(s)\n\nSummary by mode:\n  ab_test: 2 experiment(s)\n  did: 1 experiment(s)\n\nLift summary: max=30.00%, min=6.25%, avg=17.64% (3 experiments)\n\nAttention needed: 2 experiments recommend ship. Review if they test the same metric.\n```\n\n**Persistence:**\n- SQLite DB stored at `~/.agent-causal/history.db`\n- All experiment modes supported: `ab_test`, `did`, `planning`\n- Full raw JSON preserved for audit reconstruction\n- Filter by mode, limit results, name experiments for later reference\n\n## Decision Reference\n\n| Decision | Meaning | When |\n|----------|---------|------|\n| `ship` | Deploy variant | p < 0.05 AND positive lift |\n| `keep_running` | Continue experiment | p < 0.3, trending positive |\n| `reject` | Do not deploy | p < 0.05 AND negative lift |\n| `escalate` | Needs human review | Not conclusive or critical warnings |\n| `targeted_rollout` | Ship to specific segment only | Strong signal in one segment, aggregate inconclusive |\n| `full_rollout` | Ship to all users | All segments positive |\n| `abandon_segment` | Do not ship to specific segment | Strong negative in one segment despite aggregate ship |\n| `confirm_rejection` | Confirm abandonment | All segments negative |\n\n## Python API\n\n```python\nimport sys\nsys.path.insert(0, '~/clawd/agent-causal-decision-tool')\n\nfrom src.ab_test import calculate_ab\n\nresult = calculate_ab({\n    \"control_conversions\": 100,\n    \"control_total\": 5000,\n    \"variant_conversions\": 130,\n    \"variant_total\": 5000\n})\n\nif result.recommendation.decision == \"ship\":\n    # Deploy variant\n    pass\n```\n\n## Warnings & Limitations\n\n- **LOW_TRAFFIC**: Sample size below 1000 per group\n- **SMALL_EFFECT**: Lift < 1%, may not be practically significant\n- **INCONCLUSIVE**: Result not statistically significant or strong enough to act on\n- **NOT_SIGNIFICANT**: Far from significant; consider stopping the experiment\n- **BORDERLINE_P_VALUE**: p-value between 0.05 and 0.10 — weak evidence, not conclusive\n- **CORRECTION_CONSERVATIVE**: Multiple comparison correction applied; may increase false negatives\n- **SEQUENTIAL_EARLY_STOP**: Experiment stopped early via sequential testing; interpret with caution\n- **SEQUENTIAL_CONDITIONS_NOT_MET**: Early stopping conditions not met; normal decision applied\n- **MAX_RUNTIME_EXCEEDED**: Hard runtime cap exceeded without strong result; escalating\n- **ZERO_BASELINE**: Pre-period values cannot be zero for reliable DiD\n- **PARALLEL_TRENDS_VIOLATED**: Control and treated groups show very different pre-to-post ratios (critical)\n- **PARALLEL_TRENDS_WEAK**: Ratios diverge somewhat; monitor closely\n- **BOTH_GROUPS_GREW**: Both groups grew; cannot separate treatment effect from time trend\n- **AGGREGATE_DATA**: Analysis on aggregated data; use individual-level data for robust inference\n- **AGGREGATE_DATA_DID**: DiD result with high caution; not equivalent to a randomized experiment\n- **SINGLE_PRE_PERIOD**: Only one pre-period observation; parallel trends cannot be assessed\n- **SMALL_SAMPLE**: Sample size small; estimates unreliable\n- **IMBALANCED_GROUPS**: Treatment/control group sizes very different; may bias DiD estimate\n- **LARGE_EFFECT_SMALL_SAMPLE**: Large effect estimate from small sample; prioritize replication\n- **PARALLEL_TRENDS_NO_DATA**: No pre-period count provided; parallel trends cannot be assessed\n- **BOOTSTRAP_CI_UNRELIABLE**: Bootstrap CI not computed — count < 100 or zero baseline\n- **BOOTSTRAP_CI_WIDE**: Bootstrap CI range > 2×|DiD estimate|; point estimate uncertain\n- **DID_CI_CROSSES_ZERO**: Bootstrap CI crosses zero; effect direction uncertain\n- **SLOW_EXPERIMENT**: Estimated duration > 30 days; seasonal effects may confound results\n- **INFEASIBLE_EXPERIMENT**: Duration too long; consider DiD instead\n- **SMALL_MDE**: MDE very small; may require impossibly large sample\n- **BASELINE_VERY_LOW**: Baseline rate < 0.5%; estimations may be unreliable\n- **BASELINE_NEAR_ZERO**: Baseline rate < 0.1%; do not run experiment without careful review\n- **PRIOR_DOMINATES**: Very low total traffic; Jeffreys prior dominates posterior; interpret with caution\n- **CREDIBLE_INTERVAL_WIDE**: Bayesian credible interval is very wide; estimate uncertain\n\n## Schema Contract\n\nThe tool exposes a versioned schema contract for agent consumption:\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli schema\n```\n\nThis prints `schema.json` — a wrapper containing `schema_version`, `schema_coverage` (`ab`, `did`, `plan`, `bayes`), `schema_coverage_pending` (`cohort`), `severity_contract`, and `definitions` (JSON Schema from Pydantic models).\n\nAll output models include `schema_version` field injected from package metadata — never hardcoded.\n\n## Location\n\n- **GitHub:** https://github.com/ZhuMorris/agent-causal-decision-tool\n- **Local:** `~/clawd/agent-causal-decision-tool/`\n\n## Dependencies\n\n- Python 3.9+\n- click >= 8.1.0\n- scipy >= 1.11.0\n- numpy >= 1.26.0\n- pydantic >= 2.10.0\n\n## External Connectors\n\nFetch experiment data directly from external sources. The `connect` action normalizes external data into the internal experiment schema before running a decision.\n\n### PostHog\n\n```bash\n# Health check (validates credentials, no data fetched)\nPYTHONPATH=. python3 -m src.cli connect posthog --dry-run\n\n# Fetch experiment and print normalized data\nPYTHONPATH=. python3 -m src.cli connect posthog --experiment-id <id>\n\n# Fetch and run through decision workflow automatically\nPYTHONPATH=. python3 -m src.cli connect posthog --experiment-id <id> --decide\n\n# JSON-RPC call\n{\"jsonrpc\":\"2.0\",\"method\":\"connect\",\"params\":{\"source\":\"posthog\",\"experiment_id\":\"<id>\"},\"id\":\"1\"}\n```\n\n**Environment / config:**\n- `POSTHOG_API_KEY` + `POSTHOG_PROJECT_ID` env vars, OR\n- `~/.posthogrc` with `api_key`, `project_id`, `instance_url` fields\n\n**Connector result schema:**\n```json\n{\n  \"data\": { \"control_conversions\": 120, \"control_total\": 5000, \"variant_conversions\": 145, \"variant_total\": 5000 },\n  \"source_metadata\": { \"connector\": \"posthog\", \"experiment_id\": \"...\", \"fetch_timestamp\": \"...\" },\n  \"warnings\": []\n}\n```\n\n**Errors:**\n- `INSUFFICIENT_DATA` — experiment found but missing required fields\n- `ConnectorAuthError` — invalid/missing API key\n- `ConnectorNotFoundError` — experiment not found\n\nFile v0.10.2:_meta.json\n\n{\n  \"ownerId\": \"kn73d2hycfhnph00er7p4gsn8h85r7cy\",\n  \"slug\": \"agent-causal\",\n  \"version\": \"0.10.2\",\n  \"publishedAt\": 1778154064487\n}\n\nArchive v0.10.1: 2 files, 10420 bytes\n\nFiles: _meta.json (132b), SKILL.md (28009b)\n\nFile v0.10.1:SKILL.md\n\n---\nname: agent-causal\ndescription: \"Agent Causal Decision Tool helps you and your AI agents answer one question from experiment data: should we ship this change, keep running the test, or roll it back? Returns structured JSON decisions, key statistics, and audit trails from A/B tests (frequentist + Bayesian), DiD, cohort/segment analysis, and sequential early stopping.\"\nmetadata:\n  openclaw:\n    category: data-science\n    version: \"0.10.1\"\n    license: Apache-2.0\n    tools: [exec]\n    requires:\n      bins: [python3, git, pip]\n      python_packages: [click, scipy, numpy, pydantic]\n    source: https://github.com/ZhuMorris/agent-causal-decision-tool\n---\n\n# Agent Causal Decision Tool\n\nA causal decision and audit tool for AI agents. Evaluate product changes using A/B testing, Difference-in-Differences, and sequential early stopping.\n\n**Source:** https://github.com/ZhuMorris/agent-causal-decision-tool\n\n## What is this?\n\nAgent Causal Decision Tool helps you and your AI agents answer one question from experiment data: \"should we ship this change, keep running the test, or roll it back?\" It takes in simple A/B or rollout summaries and returns a structured JSON decision, key statistics, and an audit record you can store or review later.\n\nRather than being a full experimentation platform, it is a **decision engine**. You bring the data (from your logs, BI tool, or CSV); it handles the stats, decision logic, and audit trail.\n\n### Why it exists\n\nIn many teams, experiment decisions happen in ad hoc spreadsheets or dashboards. People glance at lift, argue about whether the sample size is enough, and sometimes ship features based on noisy or biased results. Agents make this worse if they are wired to react to any small uplift they see.\n\nThis tool wraps a few standard methods into one consistent, agent‑friendly interface:\n\n- **Easy-mode dispatcher (`decide`)** — no need to know which statistical method to use. Paste your numbers and it auto-selects A/B, Bayesian, DiD, or planning from your input fields.\n- **Frequentist A/B testing** for classic \"control vs variant\" questions.\n- **Bayesian A/B testing** when you want answers like \"there is a 93% chance B is better than A\" instead of only p‑values.\n- **Difference‑in‑differences (DiD)** for quasi‑experiments like staged rollouts or region‑based launches where you cannot randomize perfectly.\n- **Cohort / segment breakdown** when an aggregate result is inconclusive — you can slice by user segment to find hidden signals, with Benjamini-Hochberg correction for 4+ segments.\n- **Planning and power checks** so you can see if a test is realistic before you start it.\n- **Decision audit** so humans can see what the agent did, why it did it, and how strong the evidence really was.\n- **External connectors** — pull experiment data directly from PostHog, normalize it, and run a decision in one step. No manual export needed.\n\nThe goal is not to replace your analytics stack, but to give agents a small, reliable decision block they can call inside workflows.\n\n### When to use it\n\nUse this tool whenever you or your agents have experiment or rollout results and need a decision you can defend:\n\n- You ran an A/B test and want to know whether to ship, keep running, or reject the variant.\n- You're not sure which method to use — let `decide` auto-detect from your numbers.\n- You ran an A/B test and it was inconclusive — you want to know if a specific user segment is driving (or diluting) the effect.\n- You rolled out a feature to one region or cohort first and want a DiD estimate of impact compared to a similar control group.\n- You prefer a Bayesian summary (\"95% chance B is better; expected lift 3–5%\") to drive thresholds in automated workflows.\n- You need an audit trail with experiment period, traffic size, assumptions, thresholds, and warnings so product, data, or risk teams can review agent decisions later.\n- You want to plan an experiment (sample size, minimum detectable effect, expected duration) or compare current results to previous experiments to see which wins are robust.\n- Your experiment data lives in PostHog — you want to fetch, normalize, and decide without any manual CSV export.\n\n---\n\n## Security Model\n\nThis skill runs as a **local CLI tool only**. After one-time setup, it requires no network access.\n\n**Setup (one-time, before first use):**\n```bash\n# Download the release tarball — no git clone needed\ncurl -sL https://github.com/ZhuMorris/agent-causal-decision-tool/archive/refs/tags/v0.10.0.tar.gz -o agent-causal.tar.gz\ntar -xzf agent-causal.tar.gz\npip install agent-causal-decision-tool-0.10.0/ -q\n\n# Or clone once and install from local path (if you already have the repo):\ngit clone https://github.com/ZhuMorris/agent-causal-decision-tool.git ~/clawd/agent-causal-decision-tool\npip install ~/clawd/agent-causal-decision-tool -q\n```\n\n\nAfter installation, the `agent-causal` command is available locally. The skill itself only reads your experiment data and runs local statistical calculations.\n\n**No runtime network access.** The tool does not fetch code, pull external dependencies at runtime, or make outbound network requests during analysis.\n\n**Tools used:** `exec` (for running the `agent-causal` CLI commands you specify). Commands are fully hardcoded with no user-supplied strings interpolated into shell execution.\n\n\n**Credential handling:** PostHog API credentials are read from env vars (`POSTHOG_API_KEY`/`POSTHOG_PROJECT_ID`) or a local `~/.posthogrc` file — never hardcoded or logged.\n\n---\n\n## Setup\n\nBefore using this skill, install the tool (one-time):\n\n```bash\ngit clone https://github.com/ZhuMorris/agent-causal-decision-tool.git ~/clawd/agent-causal-decision-tool\npip install ~/clawd/agent-causal-decision-tool -q\n```\n\nAfter this, `agent-causal` is available as a local command. No further network access is required.\n\n---\n\n## Agent-Native Actions (JSON-RPC 2.0)\n\nFor AI agent integrations, Agent Causal exposes a JSON-RPC 2.0 API over both stdio and HTTP.\n\n### Stdio mode (for OpenClaw, Codex, Claude Code)\n\n```bash\npython -m src.api stdio\n```\n\n### HTTP mode\n\n```bash\npython -m src.api http --port 8000\n```\n\n### Actions\n\n| Action | Description |\n|--------|-------------|\n| `decide` | **Easy-mode dispatcher** — auto-selects A/B, Bayesian, DiD, or planning from your input fields |\n| `decide_ab` | Frequentist A/B test (`mode: frequentist`) or Bayesian (`mode: bayesian`) |\n| `decide_rollout` | DiD for staged rollouts / quasi-experiments |\n| `plan_test` | Experiment planning (sample size, MDE, feasibility) |\n| `audit_result` | Full audit of a stored result by ID |\n| `save_result` | Persist a decision result to SQLite history |\n| `get_result` | Retrieve a stored result by ID |\n| `compare_results` | Compare multiple stored experiments |\n| `connect` | Fetch experiment data from an external connector (e.g. PostHog) |\n\n### Request format\n\n```json\n{\n  \"jsonrpc\": \"2.0\",\n  \"method\": \"decide_ab\",\n  \"params\": {\n    \"mode\": \"frequentist\",\n    \"input\": {\n      \"control_conversions\": 100,\n      \"control_total\": 5000,\n      \"variant_conversions\": 130,\n      \"variant_total\": 5000\n    }\n  },\n  \"id\": 1\n}\n```\n\n### Unified response schema\n\n```json\n{\n  \"decision\": \"ship|keep_running|reject|escalate\",\n  \"recommended_next_action\": \"Deploy variant — statistical significance achieved with positive lift.\",\n  \"selected_method\": \"ab_test|bayesian_ab|did|planning\",\n  \"selection_reason\": \"Why this method was chosen\",\n  \"confidence\": \"high|medium|low\",\n  \"effect_summary\": \"Estimated lift: +30.00% (positive)\",\n  \"warnings\": [{\"code\": \"LOW_TRAFFIC\", \"message\": \"...\", \"severity\": \"info\"}],\n  \"limitations\": [\"No multiple testing correction applied\"],\n  \"audit_summary\": \"ab_test: Decision\",\n  \"source_metadata\": {\"connector\": \"langsmith\", \"dataset_id\": \"ds-001\"},\n  \"internal_result\": { ... }\n}\n```\n\n### Error format\n\n```json\n{\n  \"code\": \"VALIDATION_ERROR\",\n  \"message\": \"Invalid A/B test inputs\",\n  \"data\": {\n    \"details\": [{\"field\": \"control_total\", \"issue\": \"must be >= 1\"}],\n    \"request_id\": null\n  }\n}\n```\n\n---\n\n## Commands\n\n### Easy-mode Dispatcher (`decide`)\n\n**Don't know which method you need?** `decide` auto-detects from your input fields — no need to pick the right command:\n\n```bash\n# A/B test (auto-detected)\nPYTHONPATH=. python3 -m src.cli decide --control 100/5000 --variant 130/5000\nPYTHONPATH=. python3 -m src.cli decide --control 100/5000 --variant 130/5000 --format text\n\n# Bayesian A/B (--bayesian flag)\nPYTHONPATH=. python3 -m src.cli decide --control 100/5000 --variant 130/5000 --bayesian\n\n# DiD / staged rollout (auto-detected from pre/post treated fields)\nPYTHONPATH=. python3 -m src.cli decide --pre-control 1000 --post-control 1200 --pre-treated 200 --post-treated 280\n\n# Experiment planning (auto-detected from --baseline + --mde)\nPYTHONPATH=. python3 -m src.cli decide --baseline 0.05 --mde 10 --traffic 10000\n```\n\n\n**Auto-detection matrix:**\n| You provide... | It runs... |\n|---|---|\n| `--control` + `--variant` | Frequentist A/B |\n| `--control` + `--variant` + `--bayesian` | Bayesian A/B |\n| `--pre-control` + `--post-control` + `--pre-treated` + `--post-treated` | DiD (Difference-in-Differences) |\n| `--baseline` + `--mde` | Experiment planning |\n\n**JSON-RPC:** `{\"jsonrpc\":\"2.0\",\"method\":\"decide\",\"params\":{...fields...},\"id\":\"1\"}` — same auto-detection over stdio or HTTP.\n\n---\n\n### Experiment Planning (ab_plan)\n\nEstimate required sample size, duration, and feasibility before running an experiment:\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli plan --baseline 0.02 --mde 5 --traffic 5000\n```\n\n**Parameters:**\n- `--baseline` (required): Baseline conversion rate (e.g., `0.02` for 2%)\n- `--mde` (required): Minimum detectable effect as % lift (e.g., `5` for 5% lift)\n- `--traffic` (required): Daily traffic per arm\n- `--confidence` (default `0.95`): Confidence level\n- `--power` (default `0.8`): Statistical power\n- `--allocation`: `equal` (default) or `custom`\n- `--allocation-ratio`: Custom ratio when allocation=custom (e.g., `0.3/0.7`)\n- `--format`: `json` (default) or `text`\n\n**Planning output:**\n```json\n{\n  \"mode\": \"planning\",\n  \"recommendation\": {\n    \"decision\": \"feasible|slow|not_recommended\",\n    \"confidence\": \"high|medium|low\",\n    \"summary\": \"...\"\n  },\n  \"planning\": {\n    \"required_sample_per_arm\": 182934,\n    \"total_required\": 365868,\n    \"estimated_days\": 36.6,\n    \"feasibility\": \"slow\",\n    \"allocation_used\": {\"control\": 0.5, \"variant\": 0.5}\n  },\n  \"warnings\": [...]\n}\n```\n\n**Feasibility thresholds:**\n- `feasible`: ≤14 days\n- `slow`: 15–60 days\n- `not_recommended`: >60 days\n\n### A/B Test Analysis (Frequentist)\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli ab --control 100/5000 --variant 130/5000\n```\n\n**Parameters:**\n- `--control`: Control group conversions/total (e.g., `100/5000`)\n- `--variant`: Variant group conversions/total (e.g., `130/5000`)\n- `--name`: Variant name (optional, default: `variant_1`)\n- `--format`: Output format `json` (default) or `text`\n\n**Sequential / Early Stopping (optional):**\n- `--sequential/--no-sequential`: Enable sequential early stopping evaluation\n- `--experiment-start`, `--experiment-end`: ISO 8601 timestamps for runtime calculation\n- `--min-runtime-days` (default 7): Minimum days before early stop is considered\n- `--min-sample-per-arm` (default 2000): Minimum sample per arm before early stop\n- `--early-stop-p` (default 0.01): p-value threshold for early stop\n- `--max-runtime-days`: Hard cap; escalates if exceeded without strong result\n\n**Trigger logic:** Both min-runtime AND min-sample-per-arm must be met, AND p-value below `--early-stop-p`. Max runtime exceeded always escalates.\n```bash\nPYTHONPATH=. python3 -m src.cli ab --control 100/5000 --variant 130/5000\n```\n\n**Example Output:**\n```json\n{\n  \"schema_version\": \"0.8.0\",\n  \"mode\": \"ab_test\",\n  \"recommendation\": {\n    \"decision\": \"ship\",\n    \"confidence\": \"medium\",\n    \"summary\": \"Variant performs 30.00% better (p=0.0454). Ship it.\",\n    \"primary_metricLift\": 30.0,\n    \"p_value\": 0.045361\n  },\n  \"statistics\": {\n    \"control_rate\": 0.02,\n    \"variant_rate\": 0.026,\n    \"relative_lift_pct\": 30.0,\n    \"z_score\": 2.0013,\n    \"p_value\": 0.045361,\n    \"lift_ci_95\": [0.000124, 0.011876],\n    \"relative_lift_ci_95\": [0.619, 59.381]\n  },\n  \"traffic_stats\": {\n    \"control_size\": 5000,\n    \"variant_size\": 5000,\n    \"total_size\": 10000\n  },\n  \"warnings\": [],\n  \"next_steps\": [\"Deploy variant\", \"Monitor over time for regression\"],\n  \"audit\": {\n    \"decision_path\": [\n      {\"step\": \"Input validation\", \"passed\": true},\n      {\"step\": \"Traffic check\", \"passed\": true},\n      {\"step\": \"Conversion rate calculation\", \"passed\": true},\n      {\"step\": \"Statistical significance test\", \"passed\": true},\n      {\"step\": \"Effect size check\", \"passed\": true},\n      {\"step\": \"Decision\", \"passed\": true}\n    ]\n  }\n}\n```\n\n### Bayesian A/B Test\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli bayes --control 100/5000 --variant 130/5000\n```\n\n**Uses Beta-Binomial conjugate model with Jeffreys prior.**\n- Prior: Beta(0.5, 0.5) — uninformative\n- Posterior: Beta(α + successes, β + failures)\n- Decision via Monte Carlo simulation (20k samples)\n- Thresholds: P(variant wins) ≥ 0.95 → ship, ≤ 0.05 → reject\n\n**Parameters:**\n- `--control`, `--variant`: Conversions/total (same as `ab`)\n- `--name`: Variant name\n- `--format`: `json` (default) or `text`\n- `--samples`: Monte Carlo samples (default: 20000)\n\n**Example output:**\n```json\n{\n  \"schema_version\": \"0.8.0\",\n  \"timestamp\": \"2026-05-06T13:00:00.000Z\",\n  \"mode\": \"bayesian_ab\",\n  \"recommendation\": {\n    \"decision\": \"ship\",\n    \"confidence\": \"medium\",\n    \"summary\": \"Variant wins with P(better)=0.976. Median lift=30.10%. Ship.\",\n    \"primary_metricLift\": 30.10,\n    \"p_value\": 0.9758\n  },\n  \"statistics\": {\n    \"control_rate_observed\": 0.0200,\n    \"variant_rate_observed\": 0.0260,\n    \"relative_lift_pct\": 30.00,\n    \"posterior_control\": {\"alpha\": 100.5, \"beta\": 4900.5, \"mean\": 0.0201},\n    \"posterior_variant\": {\"alpha\": 130.5, \"beta\": 4870.5, \"mean\": 0.0261},\n    \"p_variant_wins\": 0.9758,\n    \"p_control_wins\": 0.0239,\n    \"p_tie\": 0.0003,\n    \"lift_median_pct\": 30.10,\n    \"lift_95ci_pct\": [0.20, 69.15],\n    \"expected_lift_hdi_95\": [0.0004, 0.0014],\n    \"relative_lift_hdi_95\": [2.00, 70.00],\n    \"monte_carlo_samples\": 20000,\n    \"prior_used\": {\"alpha\": 0.5, \"beta\": 0.5, \"type\": \"Jeffreys\"}\n  },\n  \"traffic_stats\": {\n    \"control_size\": 5000,\n    \"variant_size\": 5000,\n    \"total_size\": 10000\n  },\n  \"warnings\": [],\n  \"next_steps\": [\"Deploy variant\", \"Monitor for regression\"],\n  \"audit\": {\n    \"experiment_type\": \"bayesian_ab\",\n    \"thresholds_applied\": {\"ship\": 0.95, \"reject\": 0.05},\n    \"assumptions\": [\"Independent observations between groups\", \"No selection bias in group assignment\", \"Jeffrey's prior is appropriate for conversion rates\"],\n    \"limitations\": [\"Monte Carlo simulation has finite sampling error\", \"No multiple testing correction applied\"]\n  },\n  \"inputs\": {\n    \"control_conversions\": 100,\n    \"control_total\": 5000,\n    \"variant_conversions\": 130,\n    \"variant_total\": 5000\n  }\n}\n```\n\nAccess fields via `.` attribute (e.g. `result.recommendation.decision`) or `.model_dump()` for dict. Serialize with `.model_dump_json()`.\n\n**When to use Bayesian vs Frequentist:**\n\n### DiD Analysis\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli did --pre-control 1000 --post-control 1100 --pre-treated 900 --post-treated 1150\n```\n\n**Parameters:**\n- `--pre-control`: Control group metric before treatment\n- `--post-control`: Control group metric after treatment\n- `--pre-treated`: Treated group metric before treatment\n- `--post-treated`: Treated group metric after treatment\n- `--n-bootstrap` (default 2000, range 500–10000): Number of bootstrap resamples for DiD CI\n\n### Cohort / Segment Breakdown\n\nWhen an aggregate A/B or DiD result is inconclusive, break down results by user segment to find hidden signals:\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli cohort-breakdown --file segments.json\n```\n\n**Input format (JSON):**\n```json\n{\n  \"experiment_id\": \"checkout-v3\",\n  \"metric\": \"conversion_rate\",\n  \"prior_result_id\": \"dec_20260501_001\",\n  \"prior_decision\": \"wait\",\n  \"segments\": [\n    {\n      \"segment_name\": \"new_users\",\n      \"segment_definition_note\": \"Users registered within last 30 days\",\n      \"control_conversions\": 21,\n      \"control_total\": 1000,\n      \"variant_conversions\": 67,\n      \"variant_total\": 1000\n    },\n    {\n      \"segment_name\": \"returning_users\",\n      \"segment_definition_note\": \"Users registered more than 30 days ago\",\n      \"control_conversions\": 220,\n      \"control_total\": 4000,\n      \"variant_conversions\": 228,\n      \"variant_total\": 4000\n    }\n  ]\n}\n```\n\n**Input format (CSV alternative):**\n```\nsegment_name,segment_definition_note,arm,conversions,total\nnew_users,Users registered within last 30 days,control,21,1000\nnew_users,Users registered within last 30 days,variant,67,1000\nreturning_users,Users registered more than 30 days ago,control,220,4000\nreturning_users,Users registered more than 30 days ago,variant,228,4000\n```\n\n**Parameters:**\n- `--file`: Path to JSON or CSV segment file\n- `--json`: JSON input string (alternative to `--file`)\n- `--format`: Output format `json` (default) or `text`\n- `--save`: Save result to experiment history\n\n**Multiple comparison correction:**\n- 4+ segments: Benjamini-Hochberg FDR correction applied automatically\n- 5+ segments: Also offers Bonferroni as alternative via `--method bonferroni`\n\n**Example output:**\n```json\n{\n  \"method\": \"experiment_cohort_breakdown\",\n  \"cohort_decision_override\": true,\n  \"cohort_override_reason\": \"Strong positive signal in 'new_users' (lift=219.0%, adj-p=0.0000) contradicts aggregate decision 'wait'\",\n  \"interaction_flag\": false,\n  \"segments\": [\n    {\n      \"segment_name\": \"new_users\",\n      \"control_rate\": 0.021,\n      \"variant_rate\": 0.067,\n      \"relative_lift_pct\": 219.05,\n      \"p_value_raw\": 0.0000,\n      \"p_value_adjusted\": 0.0000,\n      \"decision\": \"strongly_positive\",\n      \"priority_rank\": 1\n    }\n  ],\n  \"priority_ranking\": [\n    {\"rank\": 1, \"segment\": \"new_users\", \"rationale\": \"Strong positive effect (lift=219.1%, adj-p=0.0000). Highest priority.\"}\n  ],\n  \"summary\": \"new_users drives the effect. 1 segment(s) positive.\",\n  \"recommended_next_action\": \"targeted_rollout\"\n}\n```\n\n**When to use cohort breakdown:**\n- Aggregate A/B result is `keep_running` or `escalate` — segment analysis may reveal a hidden signal\n- One segment is strongly positive while another is strongly negative (interaction flag)\n- You want to ship only to specific segments rather than all users\n\n**Key features:**\n- Per-segment two-proportion z-test with 95% confidence\n- Benjamini-Hochberg FDR correction for 4+ segments (controls false-discovery rate)\n- Priority ranking by absolute lift magnitude\n- Cohort decision override: fires when a segment contradicts the aggregate decision\n- Interaction flag: triggered when segments show opposing strongly-significant directions\n\n### Decision Audit\n\nReconstruct and explain a previous decision:\n\n```bash\n# Save result to file\nPYTHONPATH=. python3 -m src.cli ab --control 100/5000 --variant 130/5000 > /tmp/result.json\n\n# Audit it (human-readable)\nPYTHONPATH=. python3 -m src.cli audit /tmp/result.json --format text\n\n# Audit with experiment maturity assessment\nPYTHONPATH=. python3 -m src.cli audit /tmp/result.json --maturity\n```\n\n**Maturity assessment** (with `--maturity` flag):\n- Scores experiments 0–100 across 8 checks\n- Labels: `mature` (≥90), `adequate` (≥70), `immature` (≥50), `inadequate` (<50)\n- Checks: decision path completeness, critical warnings, limitations documented, traffic sufficiency, confidence level, step documentation\n\n**Example audit output:**\n```\n-- DECISION PATH --\n1. Input validation [✓]\n   control_total: 5000, variant_total: 5000\n2. Traffic check [✓]\n   control_size: 5000, min_required: 1000\n3. Conversion rate calculation [✓]\n   control_rate: 0.02, variant_rate: 0.026\n4. Statistical significance test [✓]\n   p_value: 0.045361, alpha: 0.05\n5. Effect size check [✓]\n   lift_pct: 30.0, threshold: 1\n6. Decision [✓]\n   decision: ship, confidence: medium\n\n-- FINAL DECISION --\n  Decision: SHIP\n```\n\n### Experiment History & Persistence\n\nAll commands support `--save` to persist results to local SQLite history:\n\n```bash\n# Run and save in one step\nPYTHONPATH=. python3 -m src.cli ab --control 100/5000 --variant 130/5000 --save\nPYTHONPATH=. python3 -m src.cli did --pre-control 1000 --post-control 1100 --pre-treated 900 --post-treated 1150 --save\nPYTHONPATH=. python3 -m src.cli plan --baseline 0.02 --mde 5 --traffic 5000 --save\n```\n\n**History commands:**\n\n```bash\n# List recent experiments\nPYTHONPATH=. python3 -m src.cli history\nPYTHONPATH=. python3 -m src.cli history --mode ab_test --limit 10\n\n# Compare multiple experiments by ID\nPYTHONPATH=. python3 -m src.cli compare 1 2 3\n\n# Save a prior JSON result file to history\nPYTHONPATH=. python3 -m src.cli save /tmp/result.json --name \"checkout-v3-test\"\n```\n\n**History output example:**\n```\nID    Date       Mode       Decision   Lift     P-value  Summary\n--------------------------------------------------------------------\n3     2026-04-30 did        ship       16.67    -        Treatment effect is 150.00...\n2     2026-04-30 ab_test    escalate   6.25     0.6947   Results not conclusive...\n1     2026-04-30 ab_test    ship       30.00    0.0454   Variant performs 30.00%...\n```\n\n**Compare output example:**\n```\nEXPERIMENT COMPARISON\n==================================================\nExperiments compared: 3\n\nSummary by decision:\n  SHIP: 2 experiment(s)\n  ESCALATE: 1 experiment(s)\n\nSummary by mode:\n  ab_test: 2 experiment(s)\n  did: 1 experiment(s)\n\nLift summary: max=30.00%, min=6.25%, avg=17.64% (3 experiments)\n\nAttention needed: 2 experiments recommend ship. Review if they test the same metric.\n```\n\n**Persistence:**\n- SQLite DB stored at `~/.agent-causal/history.db`\n- All experiment modes supported: `ab_test`, `did`, `planning`\n- Full raw JSON preserved for audit reconstruction\n- Filter by mode, limit results, name experiments for later reference\n\n## Decision Reference\n\n| Decision | Meaning | When |\n|----------|---------|------|\n| `ship` | Deploy variant | p < 0.05 AND positive lift |\n| `keep_running` | Continue experiment | p < 0.3, trending positive |\n| `reject` | Do not deploy | p < 0.05 AND negative lift |\n| `escalate` | Needs human review | Not conclusive or critical warnings |\n| `targeted_rollout` | Ship to specific segment only | Strong signal in one segment, aggregate inconclusive |\n| `full_rollout` | Ship to all users | All segments positive |\n| `abandon_segment` | Do not ship to specific segment | Strong negative in one segment despite aggregate ship |\n| `confirm_rejection` | Confirm abandonment | All segments negative |\n\n## Python API\n\n```python\nimport sys\nsys.path.insert(0, '~/clawd/agent-causal-decision-tool')\n\nfrom src.ab_test import calculate_ab\n\nresult = calculate_ab({\n    \"control_conversions\": 100,\n    \"control_total\": 5000,\n    \"variant_conversions\": 130,\n    \"variant_total\": 5000\n})\n\nif result.recommendation.decision == \"ship\":\n    # Deploy variant\n    pass\n```\n\n## Warnings & Limitations\n\n- **LOW_TRAFFIC**: Sample size below 1000 per group\n- **SMALL_EFFECT**: Lift < 1%, may not be practically significant\n- **INCONCLUSIVE**: Result not statistically significant or strong enough to act on\n- **NOT_SIGNIFICANT**: Far from significant; consider stopping the experiment\n- **BORDERLINE_P_VALUE**: p-value between 0.05 and 0.10 — weak evidence, not conclusive\n- **CORRECTION_CONSERVATIVE**: Multiple comparison correction applied; may increase false negatives\n- **SEQUENTIAL_EARLY_STOP**: Experiment stopped early via sequential testing; interpret with caution\n- **SEQUENTIAL_CONDITIONS_NOT_MET**: Early stopping conditions not met; normal decision applied\n- **MAX_RUNTIME_EXCEEDED**: Hard runtime cap exceeded without strong result; escalating\n- **ZERO_BASELINE**: Pre-period values cannot be zero for reliable DiD\n- **PARALLEL_TRENDS_VIOLATED**: Control and treated groups show very different pre-to-post ratios (critical)\n- **PARALLEL_TRENDS_WEAK**: Ratios diverge somewhat; monitor closely\n- **BOTH_GROUPS_GREW**: Both groups grew; cannot separate treatment effect from time trend\n- **AGGREGATE_DATA**: Analysis on aggregated data; use individual-level data for robust inference\n- **AGGREGATE_DATA_DID**: DiD result with high caution; not equivalent to a randomized experiment\n- **SINGLE_PRE_PERIOD**: Only one pre-period observation; parallel trends cannot be assessed\n- **SMALL_SAMPLE**: Sample size small; estimates unreliable\n- **IMBALANCED_GROUPS**: Treatment/control group sizes very different; may bias DiD estimate\n- **LARGE_EFFECT_SMALL_SAMPLE**: Large effect estimate from small sample; prioritize replication\n- **PARALLEL_TRENDS_NO_DATA**: No pre-period count provided; parallel trends cannot be assessed\n- **BOOTSTRAP_CI_UNRELIABLE**: Bootstrap CI not computed — count < 100 or zero baseline\n- **BOOTSTRAP_CI_WIDE**: Bootstrap CI range > 2×|DiD estimate|; point estimate uncertain\n- **DID_CI_CROSSES_ZERO**: Bootstrap CI crosses zero; effect direction uncertain\n- **SLOW_EXPERIMENT**: Estimated duration > 30 days; seasonal effects may confound results\n- **INFEASIBLE_EXPERIMENT**: Duration too long; consider DiD instead\n- **SMALL_MDE**: MDE very small; may require impossibly large sample\n- **BASELINE_VERY_LOW**: Baseline rate < 0.5%; estimations may be unreliable\n- **BASELINE_NEAR_ZERO**: Baseline rate < 0.1%; do not run experiment without careful review\n- **PRIOR_DOMINATES**: Very low total traffic; Jeffreys prior dominates posterior; interpret with caution\n- **CREDIBLE_INTERVAL_WIDE**: Bayesian credible interval is very wide; estimate uncertain\n\n## Schema Contract\n\nThe tool exposes a versioned schema contract for agent consumption:\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli schema\n```\n\nThis prints `schema.json` — a wrapper containing `schema_version`, `schema_coverage` (`ab`, `did`, `plan`, `bayes`), `schema_coverage_pending` (`cohort`), `severity_contract`, and `definitions` (JSON Schema from Pydantic models).\n\nAll output models include `schema_version` field injected from package metadata — never hardcoded.\n\n## Location\n\n- **GitHub:** https://github.com/ZhuMorris/agent-causal-decision-tool\n- **Local:** `~/clawd/agent-causal-decision-tool/`\n\n## Dependencies\n\n- Python 3.9+\n- click >= 8.1.0\n- scipy >= 1.11.0\n- numpy >= 1.26.0\n- pydantic >= 2.10.0\n\n## External Connectors\n\nFetch experiment data directly from external sources. The `connect` action normalizes external data into the internal experiment schema before running a decision.\n\n### PostHog\n\n```bash\n# Health check (validates credentials, no data fetched)\nPYTHONPATH=. python3 -m src.cli connect posthog --dry-run\n\n# Fetch experiment and print normalized data\nPYTHONPATH=. python3 -m src.cli connect posthog --experiment-id <id>\n\n# Fetch and run through decision workflow automatically\nPYTHONPATH=. python3 -m src.cli connect posthog --experiment-id <id> --decide\n\n# JSON-RPC call\n{\"jsonrpc\":\"2.0\",\"method\":\"connect\",\"params\":{\"source\":\"posthog\",\"experiment_id\":\"<id>\"},\"id\":\"1\"}\n```\n\n**Environment / config:**\n- `POSTHOG_API_KEY` + `POSTHOG_PROJECT_ID` env vars, OR\n- `~/.posthogrc` with `api_key`, `project_id`, `instance_url` fields\n\n**Connector result schema:**\n```json\n{\n  \"data\": { \"control_conversions\": 120, \"control_total\": 5000, \"variant_conversions\": 145, \"variant_total\": 5000 },\n  \"source_metadata\": { \"connector\": \"posthog\", \"experiment_id\": \"...\", \"fetch_timestamp\": \"...\" },\n  \"warnings\": []\n}\n```\n\n**Errors:**\n- `INSUFFICIENT_DATA` — experiment found but missing required fields\n- `ConnectorAuthError` — invalid/missing API key\n- `ConnectorNotFoundError` — experiment not found\n\nFile v0.10.1:_meta.json\n\n{\n  \"ownerId\": \"kn73d2hycfhnph00er7p4gsn8h85r7cy\",\n  \"slug\": \"agent-causal\",\n  \"version\": \"0.10.1\",\n  \"publishedAt\": 1778152614334\n}\n\nArchive v0.10.0: 2 files, 10224 bytes\n\nFiles: _meta.json (132b), SKILL.md (27470b)\n\nFile v0.10.0:SKILL.md\n\n---\nname: agent-causal\ndescription: \"Agent Causal Decision Tool helps you and your AI agents answer one question from experiment data: should we ship this change, keep running the test, or roll it back? Returns structured JSON decisions, key statistics, and audit trails from A/B tests (frequentist + Bayesian), DiD, cohort/segment analysis, and sequential early stopping.\"\nmetadata:\n  openclaw:\n    category: data-science\n    version: \"0.10.0\"\n    license: Apache-2.0\n    tools: [exec]\n    requires:\n      bins: [python3, git, pip]\n      python_packages: [click, scipy, numpy, pydantic]\n    source: https://github.com/ZhuMorris/agent-causal-decision-tool\n---\n\n# Agent Causal Decision Tool\n\nA causal decision and audit tool for AI agents. Evaluate product changes using A/B testing, Difference-in-Differences, and sequential early stopping.\n\n**Source:** https://github.com/ZhuMorris/agent-causal-decision-tool\n\n## What is this?\n\nAgent Causal Decision Tool helps you and your AI agents answer one question from experiment data: \"should we ship this change, keep running the test, or roll it back?\" It takes in simple A/B or rollout summaries and returns a structured JSON decision, key statistics, and an audit record you can store or review later.\n\nRather than being a full experimentation platform, it is a **decision engine**. You bring the data (from your logs, BI tool, or CSV); it handles the stats, decision logic, and audit trail.\n\n### Why it exists\n\nIn many teams, experiment decisions happen in ad hoc spreadsheets or dashboards. People glance at lift, argue about whether the sample size is enough, and sometimes ship features based on noisy or biased results. Agents make this worse if they are wired to react to any small uplift they see.\n\nThis tool wraps a few standard methods into one consistent, agent‑friendly interface:\n\n- **Easy-mode dispatcher (`decide`)** — no need to know which statistical method to use. Paste your numbers and it auto-selects A/B, Bayesian, DiD, or planning from your input fields.\n- **Frequentist A/B testing** for classic \"control vs variant\" questions.\n- **Bayesian A/B testing** when you want answers like \"there is a 93% chance B is better than A\" instead of only p‑values.\n- **Difference‑in‑differences (DiD)** for quasi‑experiments like staged rollouts or region‑based launches where you cannot randomize perfectly.\n- **Cohort / segment breakdown** when an aggregate result is inconclusive — you can slice by user segment to find hidden signals, with Benjamini-Hochberg correction for 4+ segments.\n- **Planning and power checks** so you can see if a test is realistic before you start it.\n- **Decision audit** so humans can see what the agent did, why it did it, and how strong the evidence really was.\n- **External connectors** — pull experiment data directly from PostHog, normalize it, and run a decision in one step. No manual export needed.\n\nThe goal is not to replace your analytics stack, but to give agents a small, reliable decision block they can call inside workflows.\n\n### When to use it\n\nUse this tool whenever you or your agents have experiment or rollout results and need a decision you can defend:\n\n- You ran an A/B test and want to know whether to ship, keep running, or reject the variant.\n- You're not sure which method to use — let `decide` auto-detect from your numbers.\n- You ran an A/B test and it was inconclusive — you want to know if a specific user segment is driving (or diluting) the effect.\n- You rolled out a feature to one region or cohort first and want a DiD estimate of impact compared to a similar control group.\n- You prefer a Bayesian summary (\"95% chance B is better; expected lift 3–5%\") to drive thresholds in automated workflows.\n- You need an audit trail with experiment period, traffic size, assumptions, thresholds, and warnings so product, data, or risk teams can review agent decisions later.\n- You want to plan an experiment (sample size, minimum detectable effect, expected duration) or compare current results to previous experiments to see which wins are robust.\n- Your experiment data lives in PostHog — you want to fetch, normalize, and decide without any manual CSV export.\n\n---\n\n## Security Model\n\nThis skill runs as a local CLI tool only — no code is fetched from remote at runtime.\n\n**Setup (one-time, before first use):**\n```bash\n# Clone once to local disk — no runtime network access needed\ngit clone https://github.com/ZhuMorris/agent-causal-decision-tool.git ~/clawd/agent-causal-decision-tool\npip install ~/clawd/agent-causal-decision-tool -q\n```\n\nAfter installation, the `agent-causal` command is available locally. The skill itself only reads your experiment data and runs local statistical calculations — it does not fetch code, pull external dependencies at runtime, or make outbound network requests during analysis.\n\n**Tools used:** `exec` (for running the `agent-causal` CLI commands you specify). No subprocess spawning with unsanitized input.\n\n---\n\n## Setup\n\nBefore using this skill, install the tool (one-time):\n\n```bash\ngit clone https://github.com/ZhuMorris/agent-causal-decision-tool.git ~/clawd/agent-causal-decision-tool\npip install ~/clawd/agent-causal-decision-tool -q\n```\n\nAfter this, `agent-causal` is available as a local command. No further network access is required.\n\n---\n\n## Agent-Native Actions (JSON-RPC 2.0)\n\nFor AI agent integrations, Agent Causal exposes a JSON-RPC 2.0 API over both stdio and HTTP.\n\n### Stdio mode (for OpenClaw, Codex, Claude Code)\n\n```bash\npython -m src.api stdio\n```\n\n### HTTP mode\n\n```bash\npython -m src.api http --port 8000\n```\n\n### Actions\n\n| Action | Description |\n|--------|-------------|\n| `decide` | **Easy-mode dispatcher** — auto-selects A/B, Bayesian, DiD, or planning from your input fields |\n| `decide_ab` | Frequentist A/B test (`mode: frequentist`) or Bayesian (`mode: bayesian`) |\n| `decide_rollout` | DiD for staged rollouts / quasi-experiments |\n| `plan_test` | Experiment planning (sample size, MDE, feasibility) |\n| `audit_result` | Full audit of a stored result by ID |\n| `save_result` | Persist a decision result to SQLite history |\n| `get_result` | Retrieve a stored result by ID |\n| `compare_results` | Compare multiple stored experiments |\n| `connect` | Fetch experiment data from an external connector (e.g. PostHog) |\n\n### Request format\n\n```json\n{\n  \"jsonrpc\": \"2.0\",\n  \"method\": \"decide_ab\",\n  \"params\": {\n    \"mode\": \"frequentist\",\n    \"input\": {\n      \"control_conversions\": 100,\n      \"control_total\": 5000,\n      \"variant_conversions\": 130,\n      \"variant_total\": 5000\n    }\n  },\n  \"id\": 1\n}\n```\n\n### Unified response schema\n\n```json\n{\n  \"decision\": \"ship|keep_running|reject|escalate\",\n  \"recommended_next_action\": \"Deploy variant — statistical significance achieved with positive lift.\",\n  \"selected_method\": \"ab_test|bayesian_ab|did|planning\",\n  \"selection_reason\": \"Why this method was chosen\",\n  \"confidence\": \"high|medium|low\",\n  \"effect_summary\": \"Estimated lift: +30.00% (positive)\",\n  \"warnings\": [{\"code\": \"LOW_TRAFFIC\", \"message\": \"...\", \"severity\": \"info\"}],\n  \"limitations\": [\"No multiple testing correction applied\"],\n  \"audit_summary\": \"ab_test: Decision\",\n  \"source_metadata\": {\"connector\": \"langsmith\", \"dataset_id\": \"ds-001\"},\n  \"internal_result\": { ... }\n}\n```\n\n### Error format\n\n```json\n{\n  \"code\": \"VALIDATION_ERROR\",\n  \"message\": \"Invalid A/B test inputs\",\n  \"data\": {\n    \"details\": [{\"field\": \"control_total\", \"issue\": \"must be >= 1\"}],\n    \"request_id\": null\n  }\n}\n```\n\n---\n\n## Commands\n\n### Easy-mode Dispatcher (`decide`)\n\n**Don't know which method you need?** `decide` auto-detects from your input fields — no need to pick the right command:\n\n```bash\n# A/B test (auto-detected)\nPYTHONPATH=. python3 -m src.cli decide --control 100/5000 --variant 130/5000\nPYTHONPATH=. python3 -m src.cli decide --control 100/5000 --variant 130/5000 --format text\n\n# Bayesian A/B (--bayesian flag)\nPYTHONPATH=. python3 -m src.cli decide --control 100/5000 --variant 130/5000 --bayesian\n\n# DiD / staged rollout (auto-detected from pre/post treated fields)\nPYTHONPATH=. python3 -m src.cli decide --pre-control 1000 --post-control 1200 --pre-treated 200 --post-treated 280\n\n# Experiment planning (auto-detected from --baseline + --mde)\nPYTHONPATH=. python3 -m src.cli decide --baseline 0.05 --mde 10 --traffic 10000\n```\n\n\n**Auto-detection matrix:**\n| You provide... | It runs... |\n|---|---|\n| `--control` + `--variant` | Frequentist A/B |\n| `--control` + `--variant` + `--bayesian` | Bayesian A/B |\n| `--pre-control` + `--post-control` + `--pre-treated` + `--post-treated` | DiD (Difference-in-Differences) |\n| `--baseline` + `--mde` | Experiment planning |\n\n**JSON-RPC:** `{\"jsonrpc\":\"2.0\",\"method\":\"decide\",\"params\":{...fields...},\"id\":\"1\"}` — same auto-detection over stdio or HTTP.\n\n---\n\n### Experiment Planning (ab_plan)\n\nEstimate required sample size, duration, and feasibility before running an experiment:\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli plan --baseline 0.02 --mde 5 --traffic 5000\n```\n\n**Parameters:**\n- `--baseline` (required): Baseline conversion rate (e.g., `0.02` for 2%)\n- `--mde` (required): Minimum detectable effect as % lift (e.g., `5` for 5% lift)\n- `--traffic` (required): Daily traffic per arm\n- `--confidence` (default `0.95`): Confidence level\n- `--power` (default `0.8`): Statistical power\n- `--allocation`: `equal` (default) or `custom`\n- `--allocation-ratio`: Custom ratio when allocation=custom (e.g., `0.3/0.7`)\n- `--format`: `json` (default) or `text`\n\n**Planning output:**\n```json\n{\n  \"mode\": \"planning\",\n  \"recommendation\": {\n    \"decision\": \"feasible|slow|not_recommended\",\n    \"confidence\": \"high|medium|low\",\n    \"summary\": \"...\"\n  },\n  \"planning\": {\n    \"required_sample_per_arm\": 182934,\n    \"total_required\": 365868,\n    \"estimated_days\": 36.6,\n    \"feasibility\": \"slow\",\n    \"allocation_used\": {\"control\": 0.5, \"variant\": 0.5}\n  },\n  \"warnings\": [...]\n}\n```\n\n**Feasibility thresholds:**\n- `feasible`: ≤14 days\n- `slow`: 15–60 days\n- `not_recommended`: >60 days\n\n### A/B Test Analysis (Frequentist)\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli ab --control 100/5000 --variant 130/5000\n```\n\n**Parameters:**\n- `--control`: Control group conversions/total (e.g., `100/5000`)\n- `--variant`: Variant group conversions/total (e.g., `130/5000`)\n- `--name`: Variant name (optional, default: `variant_1`)\n- `--format`: Output format `json` (default) or `text`\n\n**Sequential / Early Stopping (optional):**\n- `--sequential/--no-sequential`: Enable sequential early stopping evaluation\n- `--experiment-start`, `--experiment-end`: ISO 8601 timestamps for runtime calculation\n- `--min-runtime-days` (default 7): Minimum days before early stop is considered\n- `--min-sample-per-arm` (default 2000): Minimum sample per arm before early stop\n- `--early-stop-p` (default 0.01): p-value threshold for early stop\n- `--max-runtime-days`: Hard cap; escalates if exceeded without strong result\n\n**Trigger logic:** Both min-runtime AND min-sample-per-arm must be met, AND p-value below `--early-stop-p`. Max runtime exceeded always escalates.\n```bash\nPYTHONPATH=. python3 -m src.cli ab --control 100/5000 --variant 130/5000\n```\n\n**Example Output:**\n```json\n{\n  \"schema_version\": \"0.8.0\",\n  \"mode\": \"ab_test\",\n  \"recommendation\": {\n    \"decision\": \"ship\",\n    \"confidence\": \"medium\",\n    \"summary\": \"Variant performs 30.00% better (p=0.0454). Ship it.\",\n    \"primary_metricLift\": 30.0,\n    \"p_value\": 0.045361\n  },\n  \"statistics\": {\n    \"control_rate\": 0.02,\n    \"variant_rate\": 0.026,\n    \"relative_lift_pct\": 30.0,\n    \"z_score\": 2.0013,\n    \"p_value\": 0.045361,\n    \"lift_ci_95\": [0.000124, 0.011876],\n    \"relative_lift_ci_95\": [0.619, 59.381]\n  },\n  \"traffic_stats\": {\n    \"control_size\": 5000,\n    \"variant_size\": 5000,\n    \"total_size\": 10000\n  },\n  \"warnings\": [],\n  \"next_steps\": [\"Deploy variant\", \"Monitor over time for regression\"],\n  \"audit\": {\n    \"decision_path\": [\n      {\"step\": \"Input validation\", \"passed\": true},\n      {\"step\": \"Traffic check\", \"passed\": true},\n      {\"step\": \"Conversion rate calculation\", \"passed\": true},\n      {\"step\": \"Statistical significance test\", \"passed\": true},\n      {\"step\": \"Effect size check\", \"passed\": true},\n      {\"step\": \"Decision\", \"passed\": true}\n    ]\n  }\n}\n```\n\n### Bayesian A/B Test\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli bayes --control 100/5000 --variant 130/5000\n```\n\n**Uses Beta-Binomial conjugate model with Jeffreys prior.**\n- Prior: Beta(0.5, 0.5) — uninformative\n- Posterior: Beta(α + successes, β + failures)\n- Decision via Monte Carlo simulation (20k samples)\n- Thresholds: P(variant wins) ≥ 0.95 → ship, ≤ 0.05 → reject\n\n**Parameters:**\n- `--control`, `--variant`: Conversions/total (same as `ab`)\n- `--name`: Variant name\n- `--format`: `json` (default) or `text`\n- `--samples`: Monte Carlo samples (default: 20000)\n\n**Example output:**\n```json\n{\n  \"schema_version\": \"0.8.0\",\n  \"timestamp\": \"2026-05-06T13:00:00.000Z\",\n  \"mode\": \"bayesian_ab\",\n  \"recommendation\": {\n    \"decision\": \"ship\",\n    \"confidence\": \"medium\",\n    \"summary\": \"Variant wins with P(better)=0.976. Median lift=30.10%. Ship.\",\n    \"primary_metricLift\": 30.10,\n    \"p_value\": 0.9758\n  },\n  \"statistics\": {\n    \"control_rate_observed\": 0.0200,\n    \"variant_rate_observed\": 0.0260,\n    \"relative_lift_pct\": 30.00,\n    \"posterior_control\": {\"alpha\": 100.5, \"beta\": 4900.5, \"mean\": 0.0201},\n    \"posterior_variant\": {\"alpha\": 130.5, \"beta\": 4870.5, \"mean\": 0.0261},\n    \"p_variant_wins\": 0.9758,\n    \"p_control_wins\": 0.0239,\n    \"p_tie\": 0.0003,\n    \"lift_median_pct\": 30.10,\n    \"lift_95ci_pct\": [0.20, 69.15],\n    \"expected_lift_hdi_95\": [0.0004, 0.0014],\n    \"relative_lift_hdi_95\": [2.00, 70.00],\n    \"monte_carlo_samples\": 20000,\n    \"prior_used\": {\"alpha\": 0.5, \"beta\": 0.5, \"type\": \"Jeffreys\"}\n  },\n  \"traffic_stats\": {\n    \"control_size\": 5000,\n    \"variant_size\": 5000,\n    \"total_size\": 10000\n  },\n  \"warnings\": [],\n  \"next_steps\": [\"Deploy variant\", \"Monitor for regression\"],\n  \"audit\": {\n    \"experiment_type\": \"bayesian_ab\",\n    \"thresholds_applied\": {\"ship\": 0.95, \"reject\": 0.05},\n    \"assumptions\": [\"Independent observations between groups\", \"No selection bias in group assignment\", \"Jeffrey's prior is appropriate for conversion rates\"],\n    \"limitations\": [\"Monte Carlo simulation has finite sampling error\", \"No multiple testing correction applied\"]\n  },\n  \"inputs\": {\n    \"control_conversions\": 100,\n    \"control_total\": 5000,\n    \"variant_conversions\": 130,\n    \"variant_total\": 5000\n  }\n}\n```\n\nAccess fields via `.` attribute (e.g. `result.recommendation.decision`) or `.model_dump()` for dict. Serialize with `.model_dump_json()`.\n\n**When to use Bayesian vs Frequentist:**\n\n### DiD Analysis\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli did --pre-control 1000 --post-control 1100 --pre-treated 900 --post-treated 1150\n```\n\n**Parameters:**\n- `--pre-control`: Control group metric before treatment\n- `--post-control`: Control group metric after treatment\n- `--pre-treated`: Treated group metric before treatment\n- `--post-treated`: Treated group metric after treatment\n- `--n-bootstrap` (default 2000, range 500–10000): Number of bootstrap resamples for DiD CI\n\n### Cohort / Segment Breakdown\n\nWhen an aggregate A/B or DiD result is inconclusive, break down results by user segment to find hidden signals:\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli cohort-breakdown --file segments.json\n```\n\n**Input format (JSON):**\n```json\n{\n  \"experiment_id\": \"checkout-v3\",\n  \"metric\": \"conversion_rate\",\n  \"prior_result_id\": \"dec_20260501_001\",\n  \"prior_decision\": \"wait\",\n  \"segments\": [\n    {\n      \"segment_name\": \"new_users\",\n      \"segment_definition_note\": \"Users registered within last 30 days\",\n      \"control_conversions\": 21,\n      \"control_total\": 1000,\n      \"variant_conversions\": 67,\n      \"variant_total\": 1000\n    },\n    {\n      \"segment_name\": \"returning_users\",\n      \"segment_definition_note\": \"Users registered more than 30 days ago\",\n      \"control_conversions\": 220,\n      \"control_total\": 4000,\n      \"variant_conversions\": 228,\n      \"variant_total\": 4000\n    }\n  ]\n}\n```\n\n**Input format (CSV alternative):**\n```\nsegment_name,segment_definition_note,arm,conversions,total\nnew_users,Users registered within last 30 days,control,21,1000\nnew_users,Users registered within last 30 days,variant,67,1000\nreturning_users,Users registered more than 30 days ago,control,220,4000\nreturning_users,Users registered more than 30 days ago,variant,228,4000\n```\n\n**Parameters:**\n- `--file`: Path to JSON or CSV segment file\n- `--json`: JSON input string (alternative to `--file`)\n- `--format`: Output format `json` (default) or `text`\n- `--save`: Save result to experiment history\n\n**Multiple comparison correction:**\n- 4+ segments: Benjamini-Hochberg FDR correction applied automatically\n- 5+ segments: Also offers Bonferroni as alternative via `--method bonferroni`\n\n**Example output:**\n```json\n{\n  \"method\": \"experiment_cohort_breakdown\",\n  \"cohort_decision_override\": true,\n  \"cohort_override_reason\": \"Strong positive signal in 'new_users' (lift=219.0%, adj-p=0.0000) contradicts aggregate decision 'wait'\",\n  \"interaction_flag\": false,\n  \"segments\": [\n    {\n      \"segment_name\": \"new_users\",\n      \"control_rate\": 0.021,\n      \"variant_rate\": 0.067,\n      \"relative_lift_pct\": 219.05,\n      \"p_value_raw\": 0.0000,\n      \"p_value_adjusted\": 0.0000,\n      \"decision\": \"strongly_positive\",\n      \"priority_rank\": 1\n    }\n  ],\n  \"priority_ranking\": [\n    {\"rank\": 1, \"segment\": \"new_users\", \"rationale\": \"Strong positive effect (lift=219.1%, adj-p=0.0000). Highest priority.\"}\n  ],\n  \"summary\": \"new_users drives the effect. 1 segment(s) positive.\",\n  \"recommended_next_action\": \"targeted_rollout\"\n}\n```\n\n**When to use cohort breakdown:**\n- Aggregate A/B result is `keep_running` or `escalate` — segment analysis may reveal a hidden signal\n- One segment is strongly positive while another is strongly negative (interaction flag)\n- You want to ship only to specific segments rather than all users\n\n**Key features:**\n- Per-segment two-proportion z-test with 95% confidence\n- Benjamini-Hochberg FDR correction for 4+ segments (controls false-discovery rate)\n- Priority ranking by absolute lift magnitude\n- Cohort decision override: fires when a segment contradicts the aggregate decision\n- Interaction flag: triggered when segments show opposing strongly-significant directions\n\n### Decision Audit\n\nReconstruct and explain a previous decision:\n\n```bash\n# Save result to file\nPYTHONPATH=. python3 -m src.cli ab --control 100/5000 --variant 130/5000 > /tmp/result.json\n\n# Audit it (human-readable)\nPYTHONPATH=. python3 -m src.cli audit /tmp/result.json --format text\n\n# Audit with experiment maturity assessment\nPYTHONPATH=. python3 -m src.cli audit /tmp/result.json --maturity\n```\n\n**Maturity assessment** (with `--maturity` flag):\n- Scores experiments 0–100 across 8 checks\n- Labels: `mature` (≥90), `adequate` (≥70), `immature` (≥50), `inadequate` (<50)\n- Checks: decision path completeness, critical warnings, limitations documented, traffic sufficiency, confidence level, step documentation\n\n**Example audit output:**\n```\n-- DECISION PATH --\n1. Input validation [✓]\n   control_total: 5000, variant_total: 5000\n2. Traffic check [✓]\n   control_size: 5000, min_required: 1000\n3. Conversion rate calculation [✓]\n   control_rate: 0.02, variant_rate: 0.026\n4. Statistical significance test [✓]\n   p_value: 0.045361, alpha: 0.05\n5. Effect size check [✓]\n   lift_pct: 30.0, threshold: 1\n6. Decision [✓]\n   decision: ship, confidence: medium\n\n-- FINAL DECISION --\n  Decision: SHIP\n```\n\n### Experiment History & Persistence\n\nAll commands support `--save` to persist results to local SQLite history:\n\n```bash\n# Run and save in one step\nPYTHONPATH=. python3 -m src.cli ab --control 100/5000 --variant 130/5000 --save\nPYTHONPATH=. python3 -m src.cli did --pre-control 1000 --post-control 1100 --pre-treated 900 --post-treated 1150 --save\nPYTHONPATH=. python3 -m src.cli plan --baseline 0.02 --mde 5 --traffic 5000 --save\n```\n\n**History commands:**\n\n```bash\n# List recent experiments\nPYTHONPATH=. python3 -m src.cli history\nPYTHONPATH=. python3 -m src.cli history --mode ab_test --limit 10\n\n# Compare multiple experiments by ID\nPYTHONPATH=. python3 -m src.cli compare 1 2 3\n\n# Save a prior JSON result file to history\nPYTHONPATH=. python3 -m src.cli save /tmp/result.json --name \"checkout-v3-test\"\n```\n\n**History output example:**\n```\nID    Date       Mode       Decision   Lift     P-value  Summary\n--------------------------------------------------------------------\n3     2026-04-30 did        ship       16.67    -        Treatment effect is 150.00...\n2     2026-04-30 ab_test    escalate   6.25     0.6947   Results not conclusive...\n1     2026-04-30 ab_test    ship       30.00    0.0454   Variant performs 30.00%...\n```\n\n**Compare output example:**\n```\nEXPERIMENT COMPARISON\n==================================================\nExperiments compared: 3\n\nSummary by decision:\n  SHIP: 2 experiment(s)\n  ESCALATE: 1 experiment(s)\n\nSummary by mode:\n  ab_test: 2 experiment(s)\n  did: 1 experiment(s)\n\nLift summary: max=30.00%, min=6.25%, avg=17.64% (3 experiments)\n\nAttention needed: 2 experiments recommend ship. Review if they test the same metric.\n```\n\n**Persistence:**\n- SQLite DB stored at `~/.agent-causal/history.db`\n- All experiment modes supported: `ab_test`, `did`, `planning`\n- Full raw JSON preserved for audit reconstruction\n- Filter by mode, limit results, name experiments for later reference\n\n## Decision Reference\n\n| Decision | Meaning | When |\n|----------|---------|------|\n| `ship` | Deploy variant | p < 0.05 AND positive lift |\n| `keep_running` | Continue experiment | p < 0.3, trending positive |\n| `reject` | Do not deploy | p < 0.05 AND negative lift |\n| `escalate` | Needs human review | Not conclusive or critical warnings |\n| `targeted_rollout` | Ship to specific segment only | Strong signal in one segment, aggregate inconclusive |\n| `full_rollout` | Ship to all users | All segments positive |\n| `abandon_segment` | Do not ship to specific segment | Strong negative in one segment despite aggregate ship |\n| `confirm_rejection` | Confirm abandonment | All segments negative |\n\n## Python API\n\n```python\nimport sys\nsys.path.insert(0, '~/clawd/agent-causal-decision-tool')\n\nfrom src.ab_test import calculate_ab\n\nresult = calculate_ab({\n    \"control_conversions\": 100,\n    \"control_total\": 5000,\n    \"variant_conversions\": 130,\n    \"variant_total\": 5000\n})\n\nif result.recommendation.decision == \"ship\":\n    # Deploy variant\n    pass\n```\n\n## Warnings & Limitations\n\n- **LOW_TRAFFIC**: Sample size below 1000 per group\n- **SMALL_EFFECT**: Lift < 1%, may not be practically significant\n- **INCONCLUSIVE**: Result not statistically significant or strong enough to act on\n- **NOT_SIGNIFICANT**: Far from significant; consider stopping the experiment\n- **BORDERLINE_P_VALUE**: p-value between 0.05 and 0.10 — weak evidence, not conclusive\n- **CORRECTION_CONSERVATIVE**: Multiple comparison correction applied; may increase false negatives\n- **SEQUENTIAL_EARLY_STOP**: Experiment stopped early via sequential testing; interpret with caution\n- **SEQUENTIAL_CONDITIONS_NOT_MET**: Early stopping conditions not met; normal decision applied\n- **MAX_RUNTIME_EXCEEDED**: Hard runtime cap exceeded without strong result; escalating\n- **ZERO_BASELINE**: Pre-period values cannot be zero for reliable DiD\n- **PARALLEL_TRENDS_VIOLATED**: Control and treated groups show very different pre-to-post ratios (critical)\n- **PARALLEL_TRENDS_WEAK**: Ratios diverge somewhat; monitor closely\n- **BOTH_GROUPS_GREW**: Both groups grew; cannot separate treatment effect from time trend\n- **AGGREGATE_DATA**: Analysis on aggregated data; use individual-level data for robust inference\n- **AGGREGATE_DATA_DID**: DiD result with high caution; not equivalent to a randomized experiment\n- **SINGLE_PRE_PERIOD**: Only one pre-period observation; parallel trends cannot be assessed\n- **SMALL_SAMPLE**: Sample size small; estimates unreliable\n- **IMBALANCED_GROUPS**: Treatment/control group sizes very different; may bias DiD estimate\n- **LARGE_EFFECT_SMALL_SAMPLE**: Large effect estimate from small sample; prioritize replication\n- **PARALLEL_TRENDS_NO_DATA**: No pre-period count provided; parallel trends cannot be assessed\n- **BOOTSTRAP_CI_UNRELIABLE**: Bootstrap CI not computed — count < 100 or zero baseline\n- **BOOTSTRAP_CI_WIDE**: Bootstrap CI range > 2×|DiD estimate|; point estimate uncertain\n- **DID_CI_CROSSES_ZERO**: Bootstrap CI crosses zero; effect direction uncertain\n- **SLOW_EXPERIMENT**: Estimated duration > 30 days; seasonal effects may confound results\n- **INFEASIBLE_EXPERIMENT**: Duration too long; consider DiD instead\n- **SMALL_MDE**: MDE very small; may require impossibly large sample\n- **BASELINE_VERY_LOW**: Baseline rate < 0.5%; estimations may be unreliable\n- **BASELINE_NEAR_ZERO**: Baseline rate < 0.1%; do not run experiment without careful review\n- **PRIOR_DOMINATES**: Very low total traffic; Jeffreys prior dominates posterior; interpret with caution\n- **CREDIBLE_INTERVAL_WIDE**: Bayesian credible interval is very wide; estimate uncertain\n\n## Schema Contract\n\nThe tool exposes a versioned schema contract for agent consumption:\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli schema\n```\n\nThis prints `schema.json` — a wrapper containing `schema_version`, `schema_coverage` (`ab`, `did`, `plan`, `bayes`), `schema_coverage_pending` (`cohort`), `severity_contract`, and `definitions` (JSON Schema from Pydantic models).\n\nAll output models include `schema_version` field injected from package metadata — never hardcoded.\n\n## Location\n\n- **GitHub:** https://github.com/ZhuMorris/agent-causal-decision-tool\n- **Local:** `~/clawd/agent-causal-decision-tool/`\n\n## Dependencies\n\n- Python 3.9+\n- click >= 8.1.0\n- scipy >= 1.11.0\n- numpy >= 1.26.0\n- pydantic >= 2.10.0\n\n## External Connectors\n\nFetch experiment data directly from external sources. The `connect` action normalizes external data into the internal experiment schema before running a decision.\n\n### PostHog\n\n```bash\n# Health check (validates credentials, no data fetched)\nPYTHONPATH=. python3 -m src.cli connect posthog --dry-run\n\n# Fetch experiment and print normalized data\nPYTHONPATH=. python3 -m src.cli connect posthog --experiment-id <id>\n\n# Fetch and run through decision workflow automatically\nPYTHONPATH=. python3 -m src.cli connect posthog --experiment-id <id> --decide\n\n# JSON-RPC call\n{\"jsonrpc\":\"2.0\",\"method\":\"connect\",\"params\":{\"source\":\"posthog\",\"experiment_id\":\"<id>\"},\"id\":\"1\"}\n```\n\n**Environment / config:**\n- `POSTHOG_API_KEY` + `POSTHOG_PROJECT_ID` env vars, OR\n- `~/.posthogrc` with `api_key`, `project_id`, `instance_url` fields\n\n**Connector result schema:**\n```json\n{\n  \"data\": { \"control_conversions\": 120, \"control_total\": 5000, \"variant_conversions\": 145, \"variant_total\": 5000 },\n  \"source_metadata\": { \"connector\": \"posthog\", \"experiment_id\": \"...\", \"fetch_timestamp\": \"...\" },\n  \"warnings\": []\n}\n```\n\n**Errors:**\n- `INSUFFICIENT_DATA` — experiment found but missing required fields\n- `ConnectorAuthError` — invalid/missing API key\n- `ConnectorNotFoundError` — experiment not found\n\nFile v0.10.0:_meta.json\n\n{\n  \"ownerId\": \"kn73d2hycfhnph00er7p4gsn8h85r7cy\",\n  \"slug\": \"agent-causal\",\n  \"version\": \"0.10.0\",\n  \"publishedAt\": 1778147091468\n}\n\nArchive v0.9.9: 2 files, 10225 bytes\n\nFiles: _meta.json (131b), SKILL.md (27469b)\n\nFile v0.9.9:SKILL.md\n\n---\nname: agent-causal\ndescription: \"Agent Causal Decision Tool helps you and your AI agents answer one question from experiment data: should we ship this change, keep running the test, or roll it back? Returns structured JSON decisions, key statistics, and audit trails from A/B tests (frequentist + Bayesian), DiD, cohort/segment analysis, and sequential early stopping.\"\nmetadata:\n  openclaw:\n    category: data-science\n    version: \"0.9.9\"\n    license: Apache-2.0\n    tools: [exec]\n    requires:\n      bins: [python3, git, pip]\n      python_packages: [click, scipy, numpy, pydantic]\n    source: https://github.com/ZhuMorris/agent-causal-decision-tool\n---\n\n# Agent Causal Decision Tool\n\nA causal decision and audit tool for AI agents. Evaluate product changes using A/B testing, Difference-in-Differences, and sequential early stopping.\n\n**Source:** https://github.com/ZhuMorris/agent-causal-decision-tool\n\n## What is this?\n\nAgent Causal Decision Tool helps you and your AI agents answer one question from experiment data: \"should we ship this change, keep running the test, or roll it back?\" It takes in simple A/B or rollout summaries and returns a structured JSON decision, key statistics, and an audit record you can store or review later.\n\nRather than being a full experimentation platform, it is a **decision engine**. You bring the data (from your logs, BI tool, or CSV); it handles the stats, decision logic, and audit trail.\n\n### Why it exists\n\nIn many teams, experiment decisions happen in ad hoc spreadsheets or dashboards. People glance at lift, argue about whether the sample size is enough, and sometimes ship features based on noisy or biased results. Agents make this worse if they are wired to react to any small uplift they see.\n\nThis tool wraps a few standard methods into one consistent, agent‑friendly interface:\n\n- **Easy-mode dispatcher (`decide`)** — no need to know which statistical method to use. Paste your numbers and it auto-selects A/B, Bayesian, DiD, or planning from your input fields.\n- **Frequentist A/B testing** for classic \"control vs variant\" questions.\n- **Bayesian A/B testing** when you want answers like \"there is a 93% chance B is better than A\" instead of only p‑values.\n- **Difference‑in‑differences (DiD)** for quasi‑experiments like staged rollouts or region‑based launches where you cannot randomize perfectly.\n- **Cohort / segment breakdown** when an aggregate result is inconclusive — you can slice by user segment to find hidden signals, with Benjamini-Hochberg correction for 4+ segments.\n- **Planning and power checks** so you can see if a test is realistic before you start it.\n- **Decision audit** so humans can see what the agent did, why it did it, and how strong the evidence really was.\n- **External connectors** — pull experiment data directly from PostHog, normalize it, and run a decision in one step. No manual export needed.\n\nThe goal is not to replace your analytics stack, but to give agents a small, reliable decision block they can call inside workflows.\n\n### When to use it\n\nUse this tool whenever you or your agents have experiment or rollout results and need a decision you can defend:\n\n- You ran an A/B test and want to know whether to ship, keep running, or reject the variant.\n- You're not sure which method to use — let `decide` auto-detect from your numbers.\n- You ran an A/B test and it was inconclusive — you want to know if a specific user segment is driving (or diluting) the effect.\n- You rolled out a feature to one region or cohort first and want a DiD estimate of impact compared to a similar control group.\n- You prefer a Bayesian summary (\"95% chance B is better; expected lift 3–5%\") to drive thresholds in automated workflows.\n- You need an audit trail with experiment period, traffic size, assumptions, thresholds, and warnings so product, data, or risk teams can review agent decisions later.\n- You want to plan an experiment (sample size, minimum detectable effect, expected duration) or compare current results to previous experiments to see which wins are robust.\n- Your experiment data lives in PostHog — you want to fetch, normalize, and decide without any manual CSV export.\n\n---\n\n## Security Model\n\nThis skill runs as a local CLI tool only — no code is fetched from remote at runtime.\n\n**Setup (one-time, before first use):**\n```bash\n# Clone once to local disk — no runtime network access needed\ngit clone https://github.com/ZhuMorris/agent-causal-decision-tool.git ~/clawd/agent-causal-decision-tool\npip install ~/clawd/agent-causal-decision-tool -q\n```\n\nAfter installation, the `agent-causal` command is available locally. The skill itself only reads your experiment data and runs local statistical calculations — it does not fetch code, pull external dependencies at runtime, or make outbound network requests during analysis.\n\n**Tools used:** `exec` (for running the `agent-causal` CLI commands you specify). No subprocess spawning with unsanitized input.\n\n---\n\n## Setup\n\nBefore using this skill, install the tool (one-time):\n\n```bash\ngit clone https://github.com/ZhuMorris/agent-causal-decision-tool.git ~/clawd/agent-causal-decision-tool\npip install ~/clawd/agent-causal-decision-tool -q\n```\n\nAfter this, `agent-causal` is available as a local command. No further network access is required.\n\n---\n\n## Agent-Native Actions (JSON-RPC 2.0)\n\nFor AI agent integrations, Agent Causal exposes a JSON-RPC 2.0 API over both stdio and HTTP.\n\n### Stdio mode (for OpenClaw, Codex, Claude Code)\n\n```bash\npython -m src.api stdio\n```\n\n### HTTP mode\n\n```bash\npython -m src.api http --port 8000\n```\n\n### Actions\n\n| Action | Description |\n|--------|-------------|\n| `decide` | **Easy-mode dispatcher** — auto-selects A/B, Bayesian, DiD, or planning from your input fields |\n| `decide_ab` | Frequentist A/B test (`mode: frequentist`) or Bayesian (`mode: bayesian`) |\n| `decide_rollout` | DiD for staged rollouts / quasi-experiments |\n| `plan_test` | Experiment planning (sample size, MDE, feasibility) |\n| `audit_result` | Full audit of a stored result by ID |\n| `save_result` | Persist a decision result to SQLite history |\n| `get_result` | Retrieve a stored result by ID |\n| `compare_results` | Compare multiple stored experiments |\n| `connect` | Fetch experiment data from an external connector (e.g. PostHog) |\n\n### Request format\n\n```json\n{\n  \"jsonrpc\": \"2.0\",\n  \"method\": \"decide_ab\",\n  \"params\": {\n    \"mode\": \"frequentist\",\n    \"input\": {\n      \"control_conversions\": 100,\n      \"control_total\": 5000,\n      \"variant_conversions\": 130,\n      \"variant_total\": 5000\n    }\n  },\n  \"id\": 1\n}\n```\n\n### Unified response schema\n\n```json\n{\n  \"decision\": \"ship|keep_running|reject|escalate\",\n  \"recommended_next_action\": \"Deploy variant — statistical significance achieved with positive lift.\",\n  \"selected_method\": \"ab_test|bayesian_ab|did|planning\",\n  \"selection_reason\": \"Why this method was chosen\",\n  \"confidence\": \"high|medium|low\",\n  \"effect_summary\": \"Estimated lift: +30.00% (positive)\",\n  \"warnings\": [{\"code\": \"LOW_TRAFFIC\", \"message\": \"...\", \"severity\": \"info\"}],\n  \"limitations\": [\"No multiple testing correction applied\"],\n  \"audit_summary\": \"ab_test: Decision\",\n  \"source_metadata\": {\"connector\": \"langsmith\", \"dataset_id\": \"ds-001\"},\n  \"internal_result\": { ... }\n}\n```\n\n### Error format\n\n```json\n{\n  \"code\": \"VALIDATION_ERROR\",\n  \"message\": \"Invalid A/B test inputs\",\n  \"data\": {\n    \"details\": [{\"field\": \"control_total\", \"issue\": \"must be >= 1\"}],\n    \"request_id\": null\n  }\n}\n```\n\n---\n\n## Commands\n\n### Easy-mode Dispatcher (`decide`)\n\n**Don't know which method you need?** `decide` auto-detects from your input fields — no need to pick the right command:\n\n```bash\n# A/B test (auto-detected)\nPYTHONPATH=. python3 -m src.cli decide --control 100/5000 --variant 130/5000\nPYTHONPATH=. python3 -m src.cli decide --control 100/5000 --variant 130/5000 --format text\n\n# Bayesian A/B (--bayesian flag)\nPYTHONPATH=. python3 -m src.cli decide --control 100/5000 --variant 130/5000 --bayesian\n\n# DiD / staged rollout (auto-detected from pre/post treated fields)\nPYTHONPATH=. python3 -m src.cli decide --pre-control 1000 --post-control 1200 --pre-treated 200 --post-treated 280\n\n# Experiment planning (auto-detected from --baseline + --mde)\nPYTHONPATH=. python3 -m src.cli decide --baseline 0.05 --mde 10 --traffic 10000\n```\n\n\n**Auto-detection matrix:**\n| You provide... | It runs... |\n|---|---|\n| `--control` + `--variant` | Frequentist A/B |\n| `--control` + `--variant` + `--bayesian` | Bayesian A/B |\n| `--pre-control` + `--post-control` + `--pre-treated` + `--post-treated` | DiD (Difference-in-Differences) |\n| `--baseline` + `--mde` | Experiment planning |\n\n**JSON-RPC:** `{\"jsonrpc\":\"2.0\",\"method\":\"decide\",\"params\":{...fields...},\"id\":\"1\"}` — same auto-detection over stdio or HTTP.\n\n---\n\n### Experiment Planning (ab_plan)\n\nEstimate required sample size, duration, and feasibility before running an experiment:\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli plan --baseline 0.02 --mde 5 --traffic 5000\n```\n\n**Parameters:**\n- `--baseline` (required): Baseline conversion rate (e.g., `0.02` for 2%)\n- `--mde` (required): Minimum detectable effect as % lift (e.g., `5` for 5% lift)\n- `--traffic` (required): Daily traffic per arm\n- `--confidence` (default `0.95`): Confidence level\n- `--power` (default `0.8`): Statistical power\n- `--allocation`: `equal` (default) or `custom`\n- `--allocation-ratio`: Custom ratio when allocation=custom (e.g., `0.3/0.7`)\n- `--format`: `json` (default) or `text`\n\n**Planning output:**\n```json\n{\n  \"mode\": \"planning\",\n  \"recommendation\": {\n    \"decision\": \"feasible|slow|not_recommended\",\n    \"confidence\": \"high|medium|low\",\n    \"summary\": \"...\"\n  },\n  \"planning\": {\n    \"required_sample_per_arm\": 182934,\n    \"total_required\": 365868,\n    \"estimated_days\": 36.6,\n    \"feasibility\": \"slow\",\n    \"allocation_used\": {\"control\": 0.5, \"variant\": 0.5}\n  },\n  \"warnings\": [...]\n}\n```\n\n**Feasibility thresholds:**\n- `feasible`: ≤14 days\n- `slow`: 15–60 days\n- `not_recommended`: >60 days\n\n### A/B Test Analysis (Frequentist)\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli ab --control 100/5000 --variant 130/5000\n```\n\n**Parameters:**\n- `--control`: Control group conversions/total (e.g., `100/5000`)\n- `--variant`: Variant group conversions/total (e.g., `130/5000`)\n- `--name`: Variant name (optional, default: `variant_1`)\n- `--format`: Output format `json` (default) or `text`\n\n**Sequential / Early Stopping (optional):**\n- `--sequential/--no-sequential`: Enable sequential early stopping evaluation\n- `--experiment-start`, `--experiment-end`: ISO 8601 timestamps for runtime calculation\n- `--min-runtime-days` (default 7): Minimum days before early stop is considered\n- `--min-sample-per-arm` (default 2000): Minimum sample per arm before early stop\n- `--early-stop-p` (default 0.01): p-value threshold for early stop\n- `--max-runtime-days`: Hard cap; escalates if exceeded without strong result\n\n**Trigger logic:** Both min-runtime AND min-sample-per-arm must be met, AND p-value below `--early-stop-p`. Max runtime exceeded always escalates.\n```bash\nPYTHONPATH=. python3 -m src.cli ab --control 100/5000 --variant 130/5000\n```\n\n**Example Output:**\n```json\n{\n  \"schema_version\": \"0.8.0\",\n  \"mode\": \"ab_test\",\n  \"recommendation\": {\n    \"decision\": \"ship\",\n    \"confidence\": \"medium\",\n    \"summary\": \"Variant performs 30.00% better (p=0.0454). Ship it.\",\n    \"primary_metricLift\": 30.0,\n    \"p_value\": 0.045361\n  },\n  \"statistics\": {\n    \"control_rate\": 0.02,\n    \"variant_rate\": 0.026,\n    \"relative_lift_pct\": 30.0,\n    \"z_score\": 2.0013,\n    \"p_value\": 0.045361,\n    \"lift_ci_95\": [0.000124, 0.011876],\n    \"relative_lift_ci_95\": [0.619, 59.381]\n  },\n  \"traffic_stats\": {\n    \"control_size\": 5000,\n    \"variant_size\": 5000,\n    \"total_size\": 10000\n  },\n  \"warnings\": [],\n  \"next_steps\": [\"Deploy variant\", \"Monitor over time for regression\"],\n  \"audit\": {\n    \"decision_path\": [\n      {\"step\": \"Input validation\", \"passed\": true},\n      {\"step\": \"Traffic check\", \"passed\": true},\n      {\"step\": \"Conversion rate calculation\", \"passed\": true},\n      {\"step\": \"Statistical significance test\", \"passed\": true},\n      {\"step\": \"Effect size check\", \"passed\": true},\n      {\"step\": \"Decision\", \"passed\": true}\n    ]\n  }\n}\n```\n\n### Bayesian A/B Test\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli bayes --control 100/5000 --variant 130/5000\n```\n\n**Uses Beta-Binomial conjugate model with Jeffreys prior.**\n- Prior: Beta(0.5, 0.5) — uninformative\n- Posterior: Beta(α + successes, β + failures)\n- Decision via Monte Carlo simulation (20k samples)\n- Thresholds: P(variant wins) ≥ 0.95 → ship, ≤ 0.05 → reject\n\n**Parameters:**\n- `--control`, `--variant`: Conversions/total (same as `ab`)\n- `--name`: Variant name\n- `--format`: `json` (default) or `text`\n- `--samples`: Monte Carlo samples (default: 20000)\n\n**Example output:**\n```json\n{\n  \"schema_version\": \"0.8.0\",\n  \"timestamp\": \"2026-05-06T13:00:00.000Z\",\n  \"mode\": \"bayesian_ab\",\n  \"recommendation\": {\n    \"decision\": \"ship\",\n    \"confidence\": \"medium\",\n    \"summary\": \"Variant wins with P(better)=0.976. Median lift=30.10%. Ship.\",\n    \"primary_metricLift\": 30.10,\n    \"p_value\": 0.9758\n  },\n  \"statistics\": {\n    \"control_rate_observed\": 0.0200,\n    \"variant_rate_observed\": 0.0260,\n    \"relative_lift_pct\": 30.00,\n    \"posterior_control\": {\"alpha\": 100.5, \"beta\": 4900.5, \"mean\": 0.0201},\n    \"posterior_variant\": {\"alpha\": 130.5, \"beta\": 4870.5, \"mean\": 0.0261},\n    \"p_variant_wins\": 0.9758,\n    \"p_control_wins\": 0.0239,\n    \"p_tie\": 0.0003,\n    \"lift_median_pct\": 30.10,\n    \"lift_95ci_pct\": [0.20, 69.15],\n    \"expected_lift_hdi_95\": [0.0004, 0.0014],\n    \"relative_lift_hdi_95\": [2.00, 70.00],\n    \"monte_carlo_samples\": 20000,\n    \"prior_used\": {\"alpha\": 0.5, \"beta\": 0.5, \"type\": \"Jeffreys\"}\n  },\n  \"traffic_stats\": {\n    \"control_size\": 5000,\n    \"variant_size\": 5000,\n    \"total_size\": 10000\n  },\n  \"warnings\": [],\n  \"next_steps\": [\"Deploy variant\", \"Monitor for regression\"],\n  \"audit\": {\n    \"experiment_type\": \"bayesian_ab\",\n    \"thresholds_applied\": {\"ship\": 0.95, \"reject\": 0.05},\n    \"assumptions\": [\"Independent observations between groups\", \"No selection bias in group assignment\", \"Jeffrey's prior is appropriate for conversion rates\"],\n    \"limitations\": [\"Monte Carlo simulation has finite sampling error\", \"No multiple testing correction applied\"]\n  },\n  \"inputs\": {\n    \"control_conversions\": 100,\n    \"control_total\": 5000,\n    \"variant_conversions\": 130,\n    \"variant_total\": 5000\n  }\n}\n```\n\nAccess fields via `.` attribute (e.g. `result.recommendation.decision`) or `.model_dump()` for dict. Serialize with `.model_dump_json()`.\n\n**When to use Bayesian vs Frequentist:**\n\n### DiD Analysis\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli did --pre-control 1000 --post-control 1100 --pre-treated 900 --post-treated 1150\n```\n\n**Parameters:**\n- `--pre-control`: Control group metric before treatment\n- `--post-control`: Control group metric after treatment\n- `--pre-treated`: Treated group metric before treatment\n- `--post-treated`: Treated group metric after treatment\n- `--n-bootstrap` (default 2000, range 500–10000): Number of bootstrap resamples for DiD CI\n\n### Cohort / Segment Breakdown\n\nWhen an aggregate A/B or DiD result is inconclusive, break down results by user segment to find hidden signals:\n\n```bash\ncd ~/clawd/agent-causal-decision-tool\nPYTHONPATH=. python3 -m src.cli cohort-breakdown --file segments.json\n```\n\n**Input format (JSON):**\n```json\n{\n  \"experiment_id\": \"checkout-v3\",\n  \"metric\": \"conversion_rate\",\n  \"prior_result_id\": \"dec_20260501_001\",\n  \"prior_decision\": \"wait\",\n  \"segments\": [\n    {\n      \"segment_name\": \"new_users\",\n      \"segment_definition_note\": \"Users registered within last 30 days\",\n      \"control_conversions\": 21,\n      \"control_total\": 1000,\n      \"variant_conversions\": 67,\n      \"variant_total\": 1000\n    },\n    {\n      \"segment_name\": \"returning_users\",\n      \"segment_definition_note\": \"Users registered more than 30 days ago\",\n      \"control_conversions\": 220,\n      \"control_total\": 4000,\n      \"variant_conversions\": 228,\n      \"variant_total\": 4000\n    }\n  ]\n}\n```\n\n**Input format (CSV alternative):**\n```\nsegment_name,segment_definition_note,arm,conversions,total\nnew_users,Users registered within last 30 days,control,21,1000\nnew_users,Users registered within last 30 days,variant,67,1000\nreturning_users,Users registered more than 30 days ago,control,220,4000\nreturning_users,Users registered more than 30 days ago,variant,228,4000\n```\n\n**Parameters:**\n- `--file`: Path to JSON or CSV segment file\n- `--json`: JSON input string (alternative to `--file`)\n- `--format`: Output format `json` (default) or `text`\n- `--save`: Save result to experiment history\n\n**Multiple comparison correction:**\n- 4+ segments: Benjamini-Hochberg FDR correction applied automatically\n- 5+ segments: Also offers Bonferroni as alternative via `--method bonferroni`\n\n**Example output:**\n```json\n{\n  \"method\": \"experiment_cohort_breakdown\",\n  \"cohort_decision_override\": true,\n  \"cohort_override_reason\": \"Strong positive signal in 'new_users' (lift=219.0%, adj-p=0.0000) contradicts aggregate decision 'wait'\",\n  \"interaction_flag\": false,\n  \"segments\": [\n    {\n      \"segment_name\": \"new_users\",\n      \"control_rate\": 0.021,\n      \"variant_rate\": 0.067,\n      \"relative_lift_pct\": 219.05,\n      \"p_value_raw\": 0.0000,\n      \"p_value_adjusted\": 0.0000,\n      \"decision\": \"strongly_positive\",\n      \"priority_rank\": 1\n    }\n  ],\n  \"priority_ranking\": [\n    {\"rank\": 1, \"segment\": \"new_users\", \"rationale\": \"Strong positive effect (lift=219.1%, adj-p=0.0000). Highest priority.\"}\n  ],\n  \"summary\": \"new_users drives the effect. 1 segment(s) positive.\",\n  \"recommended_next_action\": \"targeted_rollout\"\n}\n```\n\n**When to use cohort breakdown:**\n- Aggregate A/B result is `keep_running` or `escalate` — segment analysis may reveal a hidden signal\n- One segment is strongly positive while another is strongly negative (interaction flag)\n- You want to ship only to specific segments rather than all users\n\n**Key features:**\n- Per-segment two-proportion z-test with 95% confidence\n- Benjamini-Hochberg FDR correction for 4+ segments (controls false-discovery rate)\n- Priority ranking by absolute lift magnitude\n- Cohort decision override: fires when a segment contradicts the aggregate decision\n- Interaction flag: triggered when segments show opposing strongly-significant directions\n\n### Decision Audit\n\nReconstruct and explain a previous decision:\n\n```bash\n# Save result to file\nPYTHONPATH=. python3 -m src.cli ab --control 100/5000 --variant 130/5000 > /tmp/result.json\n\n# Audit it (human-readable)\nPYTHONPATH=. python3 -m src.cli audit /tmp/result.json --format text\n\n# Audit with experiment maturity assessment\nPYTHONPATH=. python3 -m src.cli audit /tmp/result.json --maturity\n```\n\n**Maturity assessment** (with `--maturity` flag):\n- Scores experiments 0–100 across 8 checks\n- Labels: `mature` (≥90), `adequate` (≥70), `immature` (≥50), `inadequate` (<50)\n- Checks: decision path completeness, critical warnings, limitations documented, traffic sufficiency, confidence level, step documentation\n\n**Example audit output:**\n```\n-- DECISION PATH --\n1. Input validation [✓]\n   control_total: 5000, variant_total: 5000\n2. Traffic check [✓]\n   control_size: 5000, min_required: 1000\n3. Conversion rate calculation [✓]\n   control_rate: 0.02, variant_rate: 0.026\n4. Statistical significance test [✓]\n   p_value: 0.045361, alpha: 0.05\n5. Effect size check [✓]\n   lift_pct: 30.0, threshold: 1\n6. Decision [✓]\n   decision: ship, confidence: medium\n\n-- FINAL DECISION --\n  Decision: SHIP\n```\n\n### Experiment History & Persistence\n\nAll commands support `--save` to persist results to local SQLite history:\n\n```bash\n# Run and save in one step\nPYTHONPATH=. python3 -m src.cli ab --control 100/5000 --variant 130/5000 --save\nPYTHONPATH=. python3 -m src.cli did --pre-control 1000 --post-control 1100 --pre-treated 900 --post-treated 1150 --save\nPYTHONPATH=. python3 -m src.cli plan --baseline 0.02 --mde 5 --traffic 5000 --save\n```\n\n**History commands:**\n\n```bash\n# List recent experiments\nPYTHONPATH=. python3 -m src.cli history\nPYTHONPATH=. python3 -m src.cli history --mode ab_test --limit 10\n\n# Compare multiple experiments by ID\nPYTHONPATH=. python3 -m src.cli compare 1 2 3\n\n# Save a prior JSON result file to history\nPYTHONPATH=. python3 -m src.cli save /tmp/result.json --name \"checkout-v3-test\"\n```\n\n**History output example:**\n```\nID    Date       Mode       Decision   Lift     P-value  Summary\n--------------------------------------------------------------------\n3     2026-04-30 did        ship       16.67    -        Treatment effect is 150.00...\n2     2026-04-30 ab_test    escalate   6.25     0.6947   Results not conclusive...\n1     2026-04-30 ab_test    ship       30.00    0.0454   Variant performs 30.00%...\n```\n\n**Compare output example:**\n```\nEXPERIMENT COMPARISON\n==================================================\nExperiments compared: 3\n\nSummary by decision:\n  SHIP: 2 experiment(s)\n  ESCALATE: 1 experiment(s)\n\nSummary by mode:\n  ab_test: 2 experiment(s)\n  did: 1 experiment(s)\n\nLift summary: max=30.00%, min=6.25%, avg=17.64% (3 experiments)\n\nAttention needed: 2 experiments recommend ship. Review if they test the same metric.\n```\n\n**Persistence:**\n- SQLite DB stored at `~/.agent-causal/history.db`\n- All experiment modes supported: `ab_test`, `did`, `planning`\n- Full raw JSON preserved for audit reconstruction\n- Filter by mode, limit results, name experiments for later reference\n\n## Decision Reference\n\n| Decision | Meaning | When |\n|----------|---------|------|\n| `ship` | Deploy variant | p < 0.05 AND positive lift |\n| `keep_running` | Continue experiment | p < 0.3, trending positive |\n| `reject` | Do not deploy | p < 0.05 AND negative lift |\n| `escalate` | Needs human review | Not conclusive or critical warnings |\n| `targeted_rollout` | Ship to specific segment only | Strong signal in one segment, aggregate inconclusive |\n| `full_rollout` | Ship to all users | All segments positive |\n| `abandon_segment` | Do not ship to specific segment | Strong negative in one segment despite aggregate ship |\n| `confirm_rejection` | Confirm abandonment | All segments negative |\n\n## Python API\n\n```python\nimport sys\nsys.path.insert(0, '~/clawd/agent-causal-decision-tool')\n\nfrom src.ab_test import calculate_ab\n\nresult = calculate_ab({\n    \"control_conversions\": 100,\n    \"control_total\": 5000,\n    \"variant_conversions\": 130,\n    \"variant_total\": 5000\n})\n\nif result.recommendation.decision == \"ship\":\n    # Deploy variant\n    pass\n```\n\n## Warnings & Limitations\n\n- **LOW_TRAFFIC**: Sample size below 1000 per group\n- **SMALL_EFFECT**: Lift < 1%, may not be practically significant\n- **INCONCLUSIVE**: Result not statistically significant or strong enough to act on\n- **NOT_SIGNIFICANT**: Far from significant; consider stopping the experiment\n- **BORDERLINE_P_VALUE**: p-value between 0.05 and 0.10 — weak evidence, not conclusive\n- **CORRECTION_CONSERVATIVE**: Multiple comparison correction applied; may increase false negatives\n- **SEQUENTIAL_EARLY_STOP**: Experiment stopped early via sequential testing; interpret with caution\n- **SEQUENTIAL_CONDITIONS_NOT_MET**: Early stopping conditions not met; normal decision applied\n- **MAX_RUNTIME_EXCEEDED**: Hard runtime cap exceeded without strong result; escalating\n- **ZERO_BASELINE**: Pre-period values cannot be zero for reliable DiD\n- **PARALLEL_TRENDS_VIOLATED**: Control and treated groups show very different pre-to-post ratios (critical)\n- **PARALLEL_TRENDS_WEAK**: Ratios diverge somewhat; monitor closely\n- **BOTH_GROUPS_GREW**: Both groups grew; cannot separate treatment effect from time trend\n- **AGGREGATE_DATA**: Analysis on aggregated data; use individual-level data for robust inference\n- **AGGREGATE_DATA_DID**: DiD result with high caution; not equivalent to a randomized experiment\n- **SINGLE_PRE_PERIOD**: Only one pre-period observation; parallel trends cannot be assessed\n- **SMALL_SAMPLE**: Sample size small; estimates unreliable\n- **IMBALANCED_GROUPS**: Treatment/control group sizes very different; may bias DiD estimate\n- **LARGE_EFFECT_SMALL_SAMPLE**: Large effect estimate from small sample; prioritize replication\n- **PARALLEL_TRENDS_NO_DATA**: No pre-period count provided; parallel trends cannot be assessed\n- **BOOTSTRAP_CI_UNRELIABLE**: Bootstrap CI not computed — count < 100 or zero baseline\n- **BOOTSTRAP_CI_WIDE**: Bootstrap CI range > 2×|DiD estimate|; point estimate uncertain\n- **DID_CI_CROSSES_ZERO**: Bootstrap CI crosses zero; effect direction uncertain\n- **SLOW_EXPERIMENT**: Estimated duration > 30 days; seasonal effects may confound results\n- **INFEASIBLE_EXPERIMENT**: Duration too long; consider DiD instead\n- **SMALL_MDE**: MDE very small; may require impossibly large sample\n- **BASELINE_VERY_LOW**: Baseline rate < 0.5%; estimations may be unreliable\n- **BASELINE_NEAR_ZERO**: Baseline rate < 0.1%; do not run experiment without careful review\n- **PRIOR_DOMINATES**: Very low total traffic; Jeffreys prior dominates posterior; interpret with caution\n- **CREDIBLE_INTERVAL_WIDE**: Bayesian credible interval is very wide; estimate uncertain\n\n## Schema Contract\n\nThe tool exposes a versi\n\nArchive v0.9.8: 2 files, 10224 bytes\n\nFiles: _meta.json (131b), SKILL.md (27469b)\n\nArchive v0.9.7: 2 files, 10224 bytes\n\nFiles: _meta.json (131b), SKILL.md (27469b)\n\nArchive v0.9.6: 2 files, 10224 bytes\n\nFiles: _meta.json (131b), SKILL.md (27469b)\n\nArchive v0.9.5: 2 files, 10224 bytes\n\nFiles: _meta.json (131b), SKILL.md (27469b)\n\nArchive v0.9.4: 2 files, 10224 bytes\n\nFiles: _meta.json (131b), SKILL.md (27469b)","readmeExcerpt":"Skill: Agent Causal Owner: zhumorris Summary: Agent Causal Decision Tool helps you and your AI agents answer one question from experiment data: should we ship this change, keep running the test, or roll... Tags: latest:0.10.3 Version history: v0.10.3 | 2026-05-07T11:46:56.976Z | user Skill version 0.10.3 aligned with v0.10.2 git tag — SKILL.md tarball URL now uses v0.10.2 (matching the published skill version). Also ","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"curl -sL https://github.com/ZhuMorris/agent-causal-decision-tool/archive/refs/tags/v0.10.2.tar.gz -o agent-causal.tar.gz"},{"language":"bash","snippet":"# Download the release tarball — no git clone needed\ncurl -sL https://github.com/ZhuMorris/agent-causal-decision-tool/archive/refs/tags/v0.10.2.tar.gz -o agent-causal.tar.gz\ntar -xzf agent-causal.tar.gz\npip install agent-causal-decision-tool-0.10.2/ -q"},{"language":"bash","snippet":"git clone https://github.com/ZhuMorris/agent-causal-decision-tool.git ~/clawd/agent-causal-decision-tool\npip install ~/clawd/agent-causal-decision-tool -q"},{"language":"bash","snippet":"python -m src.api stdio"},{"language":"bash","snippet":"python -m src.api http --port 8000"},{"language":"json","snippet":"{\n  \"jsonrpc\": \"2.0\",\n  \"method\": \"decide_ab\",\n  \"params\": {\n    \"mode\": \"frequentist\",\n    \"input\": {\n      \"control_conversions\": 100,\n      \"control_total\": 5000,\n      \"variant_conversions\": 130,\n      \"variant_total\": 5000\n    }\n  },\n  \"id\": 1\n}"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: agent-causal\ndescription: \"Agent Causal Decision Tool helps you and your AI agents answer one question from experiment data: should we ship this change, keep running the test, or roll it back? Returns structured JSON decisions, key statistics, and audit trails from A/B tests (frequentist + Bayesian), DiD, cohort/segment analysis, and sequential early stopping.\"\nmetadata:\n  openclaw:\n    category: data-science\n    version: \"0.10.3\"\n    license: Apache-2.0\n    tools: [exec]\n    requires:\n      bins: [python3, git, pip]\n      python_packages: [click, scipy, numpy, pydantic]\n    source: https://github.com/ZhuMorris/agent-causal-decision-tool\n---\n\n# Agent Causal Decision Tool\n\nA causal decision and audit tool for AI agents. Evaluate product changes using A/B testing, Difference-in-Differences, and sequential early stopping.\n\n**Source:** https://github.com/ZhuMorris/agent-causal-decision-tool\n\n## What is this?\n\nAgent Causal Decision Tool helps you and your AI agents answer one question from experiment data: \"should we ship this change, keep running the test, or roll it back?\" It takes in simple A/B or rollout summaries and returns a structured JSON decision, key statistics, and an audit record you can store or review later.\n\nRather than being a full experimentation platform, it is a **decision engine**. You bring the data (from your logs, BI tool, or CSV); it handles the stats, decision logic, and audit trail.\n\n### Why it exists\n\nIn many teams, experiment decisions happen in ad hoc spreadsheets or dashboards. People glance at lift, argue about whether the sample size is enough, and sometimes ship features based on noisy or biased results. Agents make this worse if they are wired to react to any small uplift they see.\n\nThis tool wraps a few standard methods into one consistent, agent‑friendly interface:\n\n- **Easy-mode dispatcher (`decide`)** — no need to know which statistical method to use. Paste your numbers and it auto-selects A/B, Bayesian, DiD, or planning from your input fields.\n- **Frequentist A/B testing** for classic \"control vs variant\" questions.\n- **Bayesian A/B testing** when you want answers like \"there is a 93% chance B is better than A\" instead of only p‑values.\n- **Difference‑in‑differences (DiD)** for quasi‑experiments like staged rollouts or region‑based launches where you cannot randomize perfectly.\n- **Cohort / segment breakdown** when an aggregate result is inconclusive — you can slice by user segment to find hidden signals, with Benjamini-Hochberg correction for 4+ segments.\n- **Planning and power checks** so you can see if a test is realistic before you start it.\n- **Decision audit** so humans can see what the agent did, why it did it, and how strong the evidence really was.\n- **External connectors** — pull experiment data directly from PostHog, normalize it, and run a decision in one step. No manual export needed.\n\nThe goal is not to replace your analytics stack, but to give agents a small, reliable decision block they"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn73d2hycfhnph00er7p4gsn8h85r7cy\",\n  \"slug\": \"agent-causal\",\n  \"version\": \"0.10.3\",\n  \"publishedAt\": 1778154416976\n}"},{"path":"skill-card.md","content":"## Description:\n\nAgent Causal Decision Tool returns structured JSON decisions, key statistics, and audit trails that help agents decide whether to ship, continue, reject, or escalate experiment changes across A/B tests, Bayesian A/B tests, Difference-in-Differences, cohort analysis, planning, and sequential early stopping.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[zhumorris](https://clawhub.ai/user/zhumorris)\n\n### License/Terms of Use:\n\nApache-2.0\n\n## Use Case:\n\nDevelopers, product analysts, and AI agents use this skill to evaluate experiment or rollout summaries and produce defensible ship, continue, reject, targeted rollout, or review decisions with statistics and audit records.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Installation can fetch remote code, and the security evidence flags the mutable git clone install path as a review concern.\n\nMitigation: Prefer a pinned, checksummed release or isolated environment and avoid the mutable git clone install path.\n\nRisk: HTTP JSON-RPC mode can expose a network-facing service if bound or routed beyond local development use.\n\nMitigation: Keep HTTP mode on localhost unless independent authentication and transport controls are added.\n\nRisk: The PostHog connector makes outbound HTTPS requests and uses API credentials when explicitly invoked.\n\nMitigation: Use read-only PostHog tokens with minimal scopes and keep credentials in environment variables or local config rather than prompts or logs.\n\nRisk: Experiment recommendations can be misleading when inputs are underpowered, aggregate-only, or violate method assumptions.\n\nMitigation: Review emitted warnings, confidence, limitations, and audit records before acting on ship, reject, or targeted rollout recommendations.\n\n## Reference(s):\n\n- [Agent Causal source repository](https://github.com/ZhuMorris/agent-causal-decision-tool)\n- [Agent Causal ClawHub skill page](https://clawhub.ai/zhumorris/skills/agent-causal)\n\n## Skill Output:\n\n**Output Type(s):** [JSON, Text, Shell commands, Configuration, Guidance]\n\n**Output Format:** [Structured JSON decision objects, text summaries, and Markdown guidance with command examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May persist local SQLite audit and experiment history records when save or history commands are used.]\n\n## Skill Version(s):\n\n0.10.3 (source: artifact/_meta.json, SKILL.md metadata.openclaw.version, evidence.release.version)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Agent Causal Decision Tool helps you and your AI agents answer one question from experiment data: should we ship this change, keep running the test, or roll... Skill: Agent Causal Owner: zhumorris Summary: Agent Causal Decision Tool helps you and your AI agents answer one question from experiment data: should we ship this change, keep running the test, or roll... Tags: latest:0.10.3 Version history: v0.10.3 | 2026-05-07T11:46:56.976Z | user Skill version 0.10.3 aligned with v0.10.2 git tag — SKILL.md tarball URL now uses v0.10.2 (matching the published skill version). Also","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1412,"uniquenessScore":53,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T00:10:40.806Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T00:10:40.806Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T03:55:52.203Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}