{"id":"9ce2612a-d731-40f5-802e-e7c0aa12280a","entityType":"agent","slug":"clawhub-deciqai-falsifiability","name":"Falsifiability","canonicalUrl":"https://www.xpersona.co/agent/clawhub-deciqai-falsifiability","canonicalPath":"/agent/clawhub-deciqai-falsifiability","generatedAt":"2026-10-11T14:15:08.416Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-11T11:19:57.408Z","emptyReason":null},"description":"Activate when: user says 'what would prove this wrong', 'how do we know if our strategy is working', 'what would change your mind', 'this claim feels unfalsi... Skill: Falsifiability Owner: deciqai Summary: Activate when: user says 'what would prove this wrong', 'how do we know if our strategy is working', 'what would change your mind', 'this claim feels unfalsi... Tags: latest:1.0.5 Version history: v1.0.5 | 2026-07-16T17:59:34.650Z | user Description tail link + agents machine-readable metadata line (deciqai.com/s/falsifiability.json) v1.0.4 | 2026-07-10T10:25:40.288Z | us","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.1K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s17a4mqcnk515kvaca5ze55d0x88pfpx:falsifiability","sourceUrl":"https://clawhub.ai/deciqai/falsifiability","homepage":"https://clawhub.ai/deciqai/skills/falsifiability","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/deciqai/falsifiability","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/deciqai/skills/falsifiability","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":61,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Activate when: user says 'what would prove this wrong', 'how do we know if our strategy is working', 'what would change your mind', 'this claim feels unfalsi..."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T11:19:57.408Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T11:19:57.408Z","emptyReason":null},"stars":null,"forks":null,"downloads":1078,"packageName":null,"latestVersion":"1.0.5","tractionLabel":"1.1K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T11:19:57.395Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T11:19:57.408Z","lastCrawledAt":"2026-10-11T11:19:57.395Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T11:19:57.395Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.5","createdAt":"2026-07-16T17:59:34.650Z","changelog":"Description tail link + agents machine-readable metadata line (deciqai.com/s/falsifiability.json)","fileCount":6,"zipByteSize":12653},{"version":"1.0.4","createdAt":"2026-07-10T10:25:40.288Z","changelog":"Add 2024-2026 AI-era worked example + updated sources","fileCount":6,"zipByteSize":12610},{"version":"1.0.3","createdAt":"2026-07-08T11:02:16.593Z","changelog":"Footer now uses /c/<slug> short link (fixes UTM truncation when SKILL.md is read in a terminal)","fileCount":5,"zipByteSize":8426},{"version":"1.0.2","createdAt":"2026-07-08T00:47:39.419Z","changelog":"Refreshed content + GitHub star link in footer","fileCount":5,"zipByteSize":8651},{"version":"1.0.1","createdAt":"2026-07-07T20:33:07.815Z","changelog":"Add catalog categories and topics","fileCount":5,"zipByteSize":8515},{"version":"1.0.0","createdAt":"2026-06-28T07:20:16.280Z","changelog":"Initial publish","fileCount":5,"zipByteSize":8690}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17a4mqcnk515kvaca5ze55d0x88pfpx:falsifiability","setupComplexity":"low","setupSteps":["Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-deciqai-falsifiability/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-deciqai-falsifiability/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-deciqai-falsifiability/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-deciqai-falsifiability/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-deciqai-falsifiability/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-deciqai-falsifiability/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T14:15:08.411Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-deciqai-falsifiability/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-deciqai-falsifiability/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-deciqai-falsifiability/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-deciqai-falsifiability/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-11T11:19:57.408Z","emptyReason":null},"readme":"Skill: Falsifiability\n\nOwner: deciqai\n\nSummary: Activate when: user says 'what would prove this wrong', 'how do we know if our strategy is working', 'what would change your mind', 'this claim feels unfalsi...\n\nTags: latest:1.0.5\n\nVersion history:\n\nv1.0.5 | 2026-07-16T17:59:34.650Z | user\n\nDescription tail link + agents machine-readable metadata line (deciqai.com/s/falsifiability.json)\n\nv1.0.4 | 2026-07-10T10:25:40.288Z | user\n\nAdd 2024-2026 AI-era worked example + updated sources\n\nv1.0.3 | 2026-07-08T11:02:16.593Z | user\n\nFooter now uses /c/<slug> short link (fixes UTM truncation when SKILL.md is read in a terminal)\n\nv1.0.2 | 2026-07-08T00:47:39.419Z | user\n\nRefreshed content + GitHub star link in footer\n\nv1.0.1 | 2026-07-07T20:33:07.815Z | user\n\nAdd catalog categories and topics\n\nv1.0.0 | 2026-06-28T07:20:16.280Z | user\n\nInitial publish\n\nArchive index:\n\nArchive v1.0.5: 6 files, 12653 bytes\n\nFiles: examples/ai-claims-agi-is-near-vs-testable-predictions-2024-2026.md (6719b), examples/popper-1934-eddington-1919-eclipse-modern-applications.md (5999b), references/sources.md (1903b), skill-card.md (2231b), SKILL.md (7283b), _meta.json (133b)\n\nFile v1.0.5:SKILL.md\n\n---\nname: falsifiability\ndescription: \"Activate when: user says 'what would prove this wrong', 'how do we know if our strategy is working', 'what would change your mind', 'this claim feels unfalsifiable', or is designing a hypothesis/experiment/investment thesis that needs to be made testable.\n  Do NOT activate when: the claim is genuinely non-empirical (ethical, aesthetic, philosophical); or the cost of running the test exceeds the value of the knowledge. More: deciqai.com/c/falsifiability\"\n---\n\n# Falsifiability\n\n## Overview\n\nA meaningful empirical claim must specify what observations would refute it. Claims that resist all possible refutation are not science — they are unfalsifiable belief. Formalized by Karl Popper (1934): science progresses not by accumulating confirmations but by surviving rigorous attempts at falsification. More-specific claims are more falsifiable; ad-hoc modifications that explain away failures destroy a claim's scientific status.\n\nComposes with `confirmation-bias` (falsifiability is the structural counter), `abductive-reasoning` (generates hypotheses; this skill tests them), `bayesian-reasoning`, `critical-thinking`.\n\n## When to Use\n\n- Designing OKRs, KPIs, or strategic goals; writing or evaluating investment theses\n- Designing experiments (A/B tests, product hypotheses, market entry)\n- Evaluating consultant/advisor recommendations or diagnosing vague leadership claims\n- Someone says \"what would change your mind,\" \"how would you know you're wrong,\" \"Popper\"\n- Stress-testing an AI/AGI hype claim (\"AGI is near,\" \"the model truly understands,\" \"our AI adoption is working\") — demand what evidence would disprove the capability or safety claim\n\n**Not when:** genuinely non-empirical (philosophical, ethical, aesthetic); test cost exceeds its value.\n\n## Coaching Novices (Adaptive Front Door)\n\n- **Engine mode:** user has a specific claim → run The Process directly.\n- **Coach mode:** user is new → guide step by step.\n\nIn Coach mode, respond one step at a time. Each [WAIT] is a hard stop — output only that step's question, then stop.\n\n1. One-line: before acting on a claim, ask what observation would refute it — if \"nothing would,\" it's belief, not knowledge.\n2. Check fit: if the claim is genuinely metaphysical (ethical, aesthetic), falsifiability doesn't apply.\n3. Elicit the claim and its current evidential basis.\n> **[WAIT — do not advance until user responds]**\n4. Ask: what specific observation would falsify this? When and how would you observe it?\n> **[WAIT — do not advance until user responds]**\n5. Close: falsifiability conditions specified + monitoring plan + commitment to act on disconfirming evidence.\n> **[WAIT — do not advance until user responds]**\n\n## The Process\n\n**Step 1 — State the claim:** claim / who asserts it / decision dependent on it / current evidential basis.\n\n**Step 2 — Test whether empirical:** claim about how the world works (empirical) or values/aesthetics (non-empirical)? If non-empirical, stop here.\n\n**Step 3 — Specify falsification conditions:** complete \"This claim would be falsified if I observed: ___\" — specific, observable, time-bounded. If you cannot complete it, the claim is not falsifiable as stated.\n\n**Step 4 — Plan to observe:** when is the observation possible / how measured / who tracks it / threshold for \"falsified.\"\n\n**Step 5 — Pre-commit to action:** if the falsifying observation occurs, what will you do? Is any theory modification itself falsifiable?\n\n**Step 6 — Iterate:** confirmed → keep monitoring; falsified → revise/abandon; unfalsifiable → recognize as belief.\n\n## Output Template\n\n```markdown\n# Falsifiability Analysis: <claim>\nClaim: | Asserted by: | Decision at stake: | Current basis:\nEmpirical: Y/N | Observation type:\nFalsified if: | Threshold: | Observable when/how: | Owner:\nIf falsified, action: | Ad-hoc preservation risk:\n```\n\n*→ Method in Action: [Popper 1934 + Eddington 1919 Eclipse + Modern Applications](examples/popper-1934-eddington-1919-eclipse-modern-applications.md)*\n*→ 2026 lens: [Separating falsifiable from unfalsifiable AI claims (2024–2026)](examples/ai-claims-agi-is-near-vs-testable-predictions-2024-2026.md)*\n\n## Pack: Vague → Falsifiable\n\n| Vague (unfalsifiable) | Falsifiable form |\n|---|---|\n| \"We have PMF\" | \"≥40% of users would be 'very disappointed' without the product\" |\n| \"Our outbound is working\" | \"5% of cold emails convert to qualified opps within 30 days\" |\n| \"Our culture is strong\" | \"Employee NPS ≥40 in next quarterly survey\" |\n| \"This stock is undervalued\" | \"Stock <12 P/E within 12 months; otherwise thesis is wrong\" |\n| \"Users want feature X\" | \"≥30% complete the new flow within 14 days of launch\" |\n\n## Applying It Well\n\n- Specify falsifiability conditions *before* the observation window opens — post-result conditions are rationalization.\n- More specific = more falsifiable. Quantify wherever possible.\n- When a prediction fails, ask: is the modification of your theory itself falsifiable?\n- Apply to others: \"What would change your mind?\" is the highest-leverage critical-thinking question.\n\n*→ Primary sources: [references/sources.md](references/sources.md)*\n\n## Common Rationalizations\n\n**[D] = designed upfront | [O] = observed in real use. [O] entries are more valuable.**\n\n| Fake move | Reality |\n|---|---|\n| [D] \"I just have a strong gut feeling\" | Gut feelings are not falsifiable. Without a test, you can't tell right from wrong. |\n| [D] \"We need more data to decide\" | If you can't specify what data would change your mind, you're not doing data-driven analysis. |\n| [D] \"The strategy just needs more time\" | Specify the timeframe and metrics in advance. |\n| [D] \"It's because of external factors\" | If external factors can always be invoked, the original claim was unfalsifiable. |\n| [D] \"It's too early to evaluate\" | If you can't specify when evaluation is appropriate, the claim is unfalsifiable. |\n| [D] \"We're in a special situation\" | This defense converts a falsifiable claim into an unfalsifiable one. Resist. |\n| *→ Add [O] entries here after each real use — paste the actual failure pattern* | *What went wrong and why* |\n\n## Red Flags\n\n- Claim asserted without specifying what would refute it\n- \"The strategy is working\" without metrics or thresholds\n- Predictions modified ad-hoc to accommodate failures\n- \"External factors\" invoked to explain disconfirming evidence\n- Person cannot specify what would change their mind\n\n## Verification\n\n- [ ] Claim specifically stated; empirical vs. non-empirical determined\n- [ ] Specific, observable, time-bounded falsification conditions defined\n- [ ] Monitoring mechanism in place with a named owner\n- [ ] Advance commitment to act on falsification made\n- [ ] Risk of ad-hoc preservation recognized\n\n---\n\n*Part of **deciqAI Knowledge Skills** — 227 open-source thinking skills that make rigor executable for AI agents. The same skills power every deciqAI agent, which runs them autonomously to operate your company. **See it run → https://www.deciqai.com/c/falsifiability** · ⭐ Star the repo → https://github.com/deciqAI/knowledge-skills · Contributions welcome.*\n\n*Agents: latest version & machine-readable metadata → https://www.deciqai.com/s/falsifiability.json*\n\nFile v1.0.5:_meta.json\n\n{\n  \"ownerId\": \"kn754b8sk22s8c6gjxt02bftbn88q7ye\",\n  \"slug\": \"falsifiability\",\n  \"version\": \"1.0.5\",\n  \"publishedAt\": 1784224774650\n}\n\nFile v1.0.5:references/sources.md\n\n# Sources — falsifiability\n\n> *Primary sources for the [falsifiability](../SKILL.md) skill.*\n\n## Sources\n\n- Popper, K. R. (1934). *Logik der Forschung.* Vienna: Springer. English: *The Logic of Scientific Discovery* (1959). London: Hutchinson. ISBN 978-0415278447.\n- Popper, K. R. (1963). *Conjectures and Refutations: The Growth of Scientific Knowledge.* London: Routledge. ISBN 978-0415285940.\n- Eddington, A. S. (1919). \"Joint Eclipse Meeting of the Royal Society and the Royal Astronomical Society.\" *Nature*, 104, 354. The relativity confirmation.\n- Kuhn, T. S. (1962). *The Structure of Scientific Revolutions.* University of Chicago Press. ISBN 978-0226458120. The paradigm-shift complement.\n- Lakatos, I. (1970). \"Falsification and the Methodology of Scientific Research Programmes.\" in Lakatos & Musgrave (eds.), *Criticism and the Growth of Knowledge*. Cambridge University Press.\n- Ries, E. (2011). *The Lean Startup.* Crown Business. ISBN 978-0307887894. The startup application.\n- Tetlock, P. E. (2015). *Superforecasting.* Crown. ISBN 978-0804136693. Calibrated falsifiability in prediction.\n- Mayo, D. G. (1996). *Error and the Growth of Experimental Knowledge.* University of Chicago Press. The error-statistical framework.\n- Chollet, F. (2019). \"On the Measure of Intelligence.\" arXiv:1911.01547. Operationalizes \"intelligence\" into the falsifiable ARC / ARC-AGI benchmark — a template for turning \"AGI is near\" into a testable claim (2024–2026 AI application).\n- On benchmark contamination and reasoning robustness (2023–2025): peer-reviewed and arXiv work documenting train/test data contamination in LLM evaluations and accuracy drops when surface features of reasoning problems are perturbed — why a high AI benchmark score does not, by itself, confirm a capability claim. (Consult current surveys; specific reported scores are vendor figures, not independently audited.)\n\nFile v1.0.5:examples/ai-claims-agi-is-near-vs-testable-predictions-2024-2026.md\n\n# Method in Action: Separating Falsifiable from Unfalsifiable AI Claims (2024–2026)\n\n> *Example for the [falsifiability](../SKILL.md) skill.*\n\nBetween 2024 and 2026, public discourse around large AI models filled with two very different kinds of statement. Some were slogans — \"AGI is near,\" \"the model truly understands,\" \"scaling will just keep working\" — and some were concrete, dated predictions with numbers attached. Popper's criterion sorts them cleanly: a claim is empirical knowledge only if you can say in advance what observation would prove it wrong. This walkthrough runs the anchor claims through the falsifiability skill's own six-step Process.\n\n## Step 1 — State the claim\n\nTake three representative claims from the period:\n\n- **Claim A (slogan):** \"AGI is near.\"\n- **Claim B (mentalistic):** \"The model *truly understands* what it's saying.\"\n- **Claim C (testable):** \"By the end of 2025, a frontier model will exceed a specified accuracy threshold on the competition-mathematics benchmark AIME,\" or similar dated, metric-bound predictions of the kind researchers publish.\n\n*Who asserts them:* executives, commentators, and researchers, respectively. *Decision at stake:* whether an enterprise should bet a roadmap (or an investor a position) on imminent general capability. *Current evidential basis:* rapid, genuine benchmark gains from roughly 2023 onward, plus extrapolation.\n\n## Step 2 — Test whether empirical\n\n- **Claim A (\"AGI is near\")** is empirical *only if* \"AGI\" and \"near\" are defined. As typically used, neither is: \"AGI\" has no agreed operational definition, and \"near\" has no date. Without those, it is not yet a testable claim — it is a mood.\n- **Claim B (\"truly understands\")** invokes an inner mental state. As stated it is closer to metaphysics than to empirical science: no external observation is specified that would distinguish \"truly understands\" from \"produces the same outputs without understanding.\" Step 2 says: if it stays non-empirical, stop and label it belief.\n- **Claim C** is empirical: it is a statement about a measurable score on a fixed benchmark by a fixed date.\n\n## Step 3 — Specify falsification conditions\n\nForce each claim to complete: *\"This would be falsified if I observed: ___.\"*\n\n- **Claim A** can be *rescued into* an empirical claim by pinning it down — e.g. \"A single model will pass [a specified operationalization, such as the ARC-AGI abstraction-and-reasoning benchmark at human-level, or a stated economically-valuable-task bar] before 31 December 2026.\" Now it can fail. Note the discipline: the version that can be proven wrong is the only version worth arguing about.\n- **Claim B** resists completion. \"Understanding\" that predicts no observable difference from \"not understanding\" has no falsification condition. The productive move is to *replace* it with a behavioral proxy that does — e.g. \"the model will maintain accuracy when the same problem is presented with surface features (names, numbers, framing) changed,\" which is testable and, as documented in 2024–2025 work on reasoning robustness, sometimes fails.\n- **Claim C** already has its condition: *falsified if the reported score is below the stated threshold on the stated date.*\n\n## Step 4 — Plan to observe\n\n- *When observable:* Claim A at its stated deadline; Claim C at the benchmark's next reported evaluation; the Claim-B proxy on demand via a perturbed test set.\n- *How measured / who tracks:* published benchmark results and independent replications; assign a named owner to record the number when it lands, not to reinterpret the slogan afterward.\n- *Threshold for \"falsified\":* the pre-stated number and date. A crucial guardrail specific to AI benchmarks — **contamination**: if the evaluation items may have appeared in training data, a high score does not confirm the capability claim, so the observation plan must include held-out or post-cutoff items.\n\n## Step 5 — Pre-commit to action\n\n- If the operationalized **Claim A** deadline passes unmet: downgrade \"imminent AGI\" from a planning premise to a possibility, and stop staking irreversible roadmap bets on it.\n- For **Claim B**: if perturbation collapses performance, treat the system as pattern-competent but not robustly understanding, and design guardrails accordingly.\n- **Ad-hoc preservation risk (the core failure mode here):** the stock defenses — \"it's *almost* AGI,\" \"the benchmark was flawed,\" \"true AGI is just around the corner,\" \"you're moving the goalposts\" — are exactly the moves Popper flagged. Each rescue that explains away a missed prediction without a *new* falsifiable commitment converts the claim back into unfalsifiable belief.\n\n## Step 6 — Iterate\n\n- **Confirmed** (Claim C threshold met on clean data): keep monitoring; a passed test is not a general guarantee, only a survived attempt.\n- **Falsified** (deadline missed, proxy collapses): revise or abandon the stronger claim rather than defending it.\n- **Unfalsifiable** (Claim B left as \"it just understands\"): recognize it as belief and route the actual decision through the behavioral proxies instead.\n\n## The operational lesson\n\nThe right question to put to any 2024–2026 AI capability or safety claim is the skill's highest-leverage one: **\"What specific evidence, observable by when, would prove this wrong?\"** Claims that can answer — \"score below X on held-out set Y by date Z\" — are doing empirical work. Claims that cannot — \"it's near,\" \"it truly understands,\" \"trust the trajectory\" — are marketing or metaphysics wearing the costume of a prediction. The same test applies to safety claims: \"the model is safe\" is unfalsifiable until it is rewritten as \"the model will not produce [specified harmful output] under [specified red-team protocol],\" which can, and sometimes does, fail.\n\n*Sources: Karl Popper, *The Logic of Scientific Discovery* (1959; orig. *Logik der Forschung*, 1934) and *Conjectures and Refutations* (1963) for the falsifiability criterion. On operationalized AGI-style benchmarks: François Chollet, \"On the Measure of Intelligence\" (arXiv:1911.01547, 2019) and the ARC-AGI benchmark. On competition-math evaluation of frontier models: the AIME/MATH benchmark literature and vendors' publicly reported 2024–2025 results (treat specific figures as reported, not independently audited). On benchmark contamination and on reasoning robustness under input perturbation: peer-reviewed and arXiv work published 2023–2025 (e.g., studies of train/test contamination and of accuracy drops when numeric/surface features of problems are altered). Exact scores and dates vary by model and report; the reasoning here does not depend on any single unverified figure.*\n\nFile v1.0.5:examples/popper-1934-eddington-1919-eclipse-modern-applications.md\n\n# Method in Action: Popper 1934 + Eddington 1919 Eclipse + Modern Applications\n\n> *Example for the [falsifiability](../SKILL.md) skill.*\n\n**Karl Popper** (1902-1994) was an Austrian-British philosopher whose 1934 *Logik der Forschung* (translated as *The Logic of Scientific Discovery* in 1959) reshaped 20th-century philosophy of science. Popper had observed the rise of psychoanalysis (Freud, Adler) and Marxism in Vienna during the 1920s, and was struck by how their adherents claimed they \"explained\" everything — every observed event could be interpreted as confirming the theory. Popper's diagnosis: any theory that explains everything explains nothing. Such theories make no risky predictions; they cannot fail; therefore they are not empirical knowledge.\n\nPopper contrasted this with Einstein's general relativity (1915). Einstein had made a specific, quantitative prediction: light passing through the Sun's gravitational field during an eclipse would be deflected by 1.75 arcseconds, twice the Newtonian prediction. The 1919 total solar eclipse provided the empirical test. **Arthur Eddington's measurements during the eclipse** (in Príncipe and Brazil) confirmed the relativistic prediction within experimental error:\n\n> Eddington, A. S. (1919). \"Joint Eclipse Meeting of the Royal Society and the Royal Astronomical Society.\" *Nature*, 104, 354.\n\nPopper's emphasis: the test was real because relativity could have failed. If the measured deflection had been close to the Newtonian value or no deflection, relativity would have been falsified. The test mattered because failure was possible. The theory's confirmation was meaningful only because falsification was possible. This is the operational criterion of science.\n\nPopper's 1963 *Conjectures and Refutations* extended the framework to many domains:\n\n> \"The history of science, like the history of all human ideas, is a history of irresponsible dreams, of obstinacy, and of error. But science is one of the very few human activities — perhaps the only one — in which errors are systematically criticized and fairly often, in time, corrected. This is why we can say that, in science, we often learn from our mistakes, and why we can speak clearly and sensibly about making progress there.\"\n>\n> — Popper (1963), p. 216.\n\nThe framework has been applied extensively:\n\n**Lean Startup methodology.** Eric Ries's 2011 *The Lean Startup* is implicitly Popperian. Each iteration generates falsifiable hypotheses (\"if we A/B test this checkout flow change, conversion will improve by ≥5%\"), tests them empirically, and either confirms or rejects them based on data. The build-measure-learn cycle is operationalized falsifiability applied to product development.\n\n**Hypothesis-driven product development.** Modern product organizations (Amazon, Google, Microsoft) make explicit predictions about what user behavior change a feature will produce, then measure the actual change. Features that don't deliver the predicted impact are rolled back. The discipline is Popperian falsifiability at scale.\n\n**Investment thesis discipline.** Sophisticated investors specify what observations would falsify their investment thesis. If a thesis is \"Company X will reach $1B ARR by 2027,\" then quarterly milestones are set as falsifiability triggers. If Q3 ARR is below threshold X, the thesis is at risk; if it's below threshold Y, the thesis is falsified and the position is reconsidered.\n\n**Strategic planning.** Sophisticated strategy processes specify what conditions would change the strategy. \"Our differentiation rests on X capability; if competitors X' achieves parity in this capability, we need to revisit.\" The conditions are falsifiability triggers for the strategic thesis.\n\n**Scientific peer review and reproducibility.** The \"reproducibility crisis\" in social sciences (2010s onward) revealed that many published findings could not be replicated. Modern reforms (pre-registration of hypotheses, public datasets, replication studies) are explicit Popperian: making research findings falsifiable in practice, not just in principle.\n\n**Defense against unfalsifiable claims:** Several domains demonstrate the operational cost of unfalsifiable claims:\n\n- **Cult and conspiracy theory dynamics.** True believers find ways to \"explain\" any disconfirming evidence (the lack of UFO landing means the aliens are testing us; the lack of cult prediction fulfillment means we weren't ready). The structural feature is unfalsifiability — no observation could change their mind.\n- **Some macroeconomic claims.** \"Austerity will eventually produce growth\" or \"more stimulus will eventually produce inflation\" — without timeframes and thresholds, these claims are unfalsifiable. Any short-term failure can be attributed to \"not yet.\"\n- **Vague leadership pronouncements.** \"Our culture is strong\" or \"we have great execution\" without metrics or thresholds are unfalsifiable. They sound informative but contain no testable content.\n\nThree operational lessons:\n\n**First, the falsifiability question is the highest-leverage critical-thinking tool.** \"What would change your mind?\" — applied to any claim, your own or another's — produces dramatically better thinking than the alternative of accumulating confirmations.\n\n**Second, ad-hoc preservation is the failure mode.** When a prediction fails, the temptation is to modify the theory to accommodate the failure (\"the strategy worked, but X happened to prevent it\"). Each ad-hoc modification reduces the scientific status of the claim. Sometimes modifications are justified; usually they are face-saving rationalization that compounds error.\n\n**Third, specifying falsifiability conditions in advance is the discipline.** Conditions that you specify after the result are not falsifiability; they are post-hoc rationalization. The Popperian discipline is to commit to falsification triggers before the relevant observations are made — when you might genuinely be wrong, not after the fact.\n\nFile v1.0.5:skill-card.md\n\n## Description:\n\nHelps agents turn empirical claims, strategies, hypotheses, investment theses, and AI capability claims into testable statements with explicit falsification conditions.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[deciqai](https://clawhub.ai/user/deciqai)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nEmployees, external users, and developers use this skill to stress-test claims before making decisions. It guides agents to state the claim, decide whether it is empirical, define observable falsification thresholds, assign monitoring, and pre-commit to action if the claim fails.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Business, investment, and AI examples could be mistaken for financial or operational advice.\n\nMitigation: Treat the examples as prompts for analysis and have qualified owners review consequential decisions independently.\n\nRisk: The falsifiability workflow can be over-applied to ethical, aesthetic, philosophical, or uneconomical questions.\n\nMitigation: Use the skill's fit check first and stop when a claim is non-empirical or the cost of testing exceeds the value of the knowledge.\n\n## Reference(s):\n\n- [Sources - falsifiability](references/sources.md)\n- [Popper 1934 + Eddington 1919 Eclipse + Modern Applications](examples/popper-1934-eddington-1919-eclipse-modern-applications.md)\n- [Separating falsifiable from unfalsifiable AI claims (2024-2026)](examples/ai-claims-agi-is-near-vs-testable-predictions-2024-2026.md)\n- [ClawHub skill page](https://clawhub.ai/deciqai/skills/falsifiability)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, guidance]\n\n**Output Format:** [Markdown analysis template with concise guidance]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May ask one question at a time in coach mode before producing the final analysis.]\n\n## Skill Version(s):\n\n1.0.5 (source: server release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v1.0.4: 6 files, 12610 bytes\n\nFiles: examples/ai-claims-agi-is-near-vs-testable-predictions-2024-2026.md (6719b), examples/popper-1934-eddington-1919-eclipse-modern-applications.md (5999b), references/sources.md (1903b), skill-card.md (2467b), SKILL.md (7144b), _meta.json (133b)\n\nFile v1.0.4:SKILL.md\n\n---\nname: falsifiability\ndescription: \"Activate when: user says 'what would prove this wrong', 'how do we know if our strategy is working', 'what would change your mind', 'this claim feels unfalsifiable', or is designing a hypothesis/experiment/investment thesis that needs to be made testable.\n  Do NOT activate when: the claim is genuinely non-empirical (ethical, aesthetic, philosophical); or the cost of running the test exceeds the value of the knowledge.\"\n---\n\n# Falsifiability\n\n## Overview\n\nA meaningful empirical claim must specify what observations would refute it. Claims that resist all possible refutation are not science — they are unfalsifiable belief. Formalized by Karl Popper (1934): science progresses not by accumulating confirmations but by surviving rigorous attempts at falsification. More-specific claims are more falsifiable; ad-hoc modifications that explain away failures destroy a claim's scientific status.\n\nComposes with `confirmation-bias` (falsifiability is the structural counter), `abductive-reasoning` (generates hypotheses; this skill tests them), `bayesian-reasoning`, `critical-thinking`.\n\n## When to Use\n\n- Designing OKRs, KPIs, or strategic goals; writing or evaluating investment theses\n- Designing experiments (A/B tests, product hypotheses, market entry)\n- Evaluating consultant/advisor recommendations or diagnosing vague leadership claims\n- Someone says \"what would change your mind,\" \"how would you know you're wrong,\" \"Popper\"\n- Stress-testing an AI/AGI hype claim (\"AGI is near,\" \"the model truly understands,\" \"our AI adoption is working\") — demand what evidence would disprove the capability or safety claim\n\n**Not when:** genuinely non-empirical (philosophical, ethical, aesthetic); test cost exceeds its value.\n\n## Coaching Novices (Adaptive Front Door)\n\n- **Engine mode:** user has a specific claim → run The Process directly.\n- **Coach mode:** user is new → guide step by step.\n\nIn Coach mode, respond one step at a time. Each [WAIT] is a hard stop — output only that step's question, then stop.\n\n1. One-line: before acting on a claim, ask what observation would refute it — if \"nothing would,\" it's belief, not knowledge.\n2. Check fit: if the claim is genuinely metaphysical (ethical, aesthetic), falsifiability doesn't apply.\n3. Elicit the claim and its current evidential basis.\n> **[WAIT — do not advance until user responds]**\n4. Ask: what specific observation would falsify this? When and how would you observe it?\n> **[WAIT — do not advance until user responds]**\n5. Close: falsifiability conditions specified + monitoring plan + commitment to act on disconfirming evidence.\n> **[WAIT — do not advance until user responds]**\n\n## The Process\n\n**Step 1 — State the claim:** claim / who asserts it / decision dependent on it / current evidential basis.\n\n**Step 2 — Test whether empirical:** claim about how the world works (empirical) or values/aesthetics (non-empirical)? If non-empirical, stop here.\n\n**Step 3 — Specify falsification conditions:** complete \"This claim would be falsified if I observed: ___\" — specific, observable, time-bounded. If you cannot complete it, the claim is not falsifiable as stated.\n\n**Step 4 — Plan to observe:** when is the observation possible / how measured / who tracks it / threshold for \"falsified.\"\n\n**Step 5 — Pre-commit to action:** if the falsifying observation occurs, what will you do? Is any theory modification itself falsifiable?\n\n**Step 6 — Iterate:** confirmed → keep monitoring; falsified → revise/abandon; unfalsifiable → recognize as belief.\n\n## Output Template\n\n```markdown\n# Falsifiability Analysis: <claim>\nClaim: | Asserted by: | Decision at stake: | Current basis:\nEmpirical: Y/N | Observation type:\nFalsified if: | Threshold: | Observable when/how: | Owner:\nIf falsified, action: | Ad-hoc preservation risk:\n```\n\n*→ Method in Action: [Popper 1934 + Eddington 1919 Eclipse + Modern Applications](examples/popper-1934-eddington-1919-eclipse-modern-applications.md)*\n*→ 2026 lens: [Separating falsifiable from unfalsifiable AI claims (2024–2026)](examples/ai-claims-agi-is-near-vs-testable-predictions-2024-2026.md)*\n\n## Pack: Vague → Falsifiable\n\n| Vague (unfalsifiable) | Falsifiable form |\n|---|---|\n| \"We have PMF\" | \"≥40% of users would be 'very disappointed' without the product\" |\n| \"Our outbound is working\" | \"5% of cold emails convert to qualified opps within 30 days\" |\n| \"Our culture is strong\" | \"Employee NPS ≥40 in next quarterly survey\" |\n| \"This stock is undervalued\" | \"Stock <12 P/E within 12 months; otherwise thesis is wrong\" |\n| \"Users want feature X\" | \"≥30% complete the new flow within 14 days of launch\" |\n\n## Applying It Well\n\n- Specify falsifiability conditions *before* the observation window opens — post-result conditions are rationalization.\n- More specific = more falsifiable. Quantify wherever possible.\n- When a prediction fails, ask: is the modification of your theory itself falsifiable?\n- Apply to others: \"What would change your mind?\" is the highest-leverage critical-thinking question.\n\n*→ Primary sources: [references/sources.md](references/sources.md)*\n\n## Common Rationalizations\n\n**[D] = designed upfront | [O] = observed in real use. [O] entries are more valuable.**\n\n| Fake move | Reality |\n|---|---|\n| [D] \"I just have a strong gut feeling\" | Gut feelings are not falsifiable. Without a test, you can't tell right from wrong. |\n| [D] \"We need more data to decide\" | If you can't specify what data would change your mind, you're not doing data-driven analysis. |\n| [D] \"The strategy just needs more time\" | Specify the timeframe and metrics in advance. |\n| [D] \"It's because of external factors\" | If external factors can always be invoked, the original claim was unfalsifiable. |\n| [D] \"It's too early to evaluate\" | If you can't specify when evaluation is appropriate, the claim is unfalsifiable. |\n| [D] \"We're in a special situation\" | This defense converts a falsifiable claim into an unfalsifiable one. Resist. |\n| *→ Add [O] entries here after each real use — paste the actual failure pattern* | *What went wrong and why* |\n\n## Red Flags\n\n- Claim asserted without specifying what would refute it\n- \"The strategy is working\" without metrics or thresholds\n- Predictions modified ad-hoc to accommodate failures\n- \"External factors\" invoked to explain disconfirming evidence\n- Person cannot specify what would change their mind\n\n## Verification\n\n- [ ] Claim specifically stated; empirical vs. non-empirical determined\n- [ ] Specific, observable, time-bounded falsification conditions defined\n- [ ] Monitoring mechanism in place with a named owner\n- [ ] Advance commitment to act on falsification made\n- [ ] Risk of ad-hoc preservation recognized\n\n---\n\n*Part of **deciqAI Knowledge Skills** — 223 open-source thinking skills that make rigor executable for AI agents. The same skills power every deciqAI agent, which runs them autonomously to operate your company. **See it run → https://www.deciqai.com/c/falsifiability** · ⭐ Star the repo → https://github.com/deciqAI/knowledge-skills · Contributions welcome.*\n\nFile v1.0.4:_meta.json\n\n{\n  \"ownerId\": \"kn754b8sk22s8c6gjxt02bftbn88q7ye\",\n  \"slug\": \"falsifiability\",\n  \"version\": \"1.0.4\",\n  \"publishedAt\": 1783679140288\n}\n\nFile v1.0.4:references/sources.md\n\n# Sources — falsifiability\n\n> *Primary sources for the [falsifiability](../SKILL.md) skill.*\n\n## Sources\n\n- Popper, K. R. (1934). *Logik der Forschung.* Vienna: Springer. English: *The Logic of Scientific Discovery* (1959). London: Hutchinson. ISBN 978-0415278447.\n- Popper, K. R. (1963). *Conjectures and Refutations: The Growth of Scientific Knowledge.* London: Routledge. ISBN 978-0415285940.\n- Eddington, A. S. (1919). \"Joint Eclipse Meeting of the Royal Society and the Royal Astronomical Society.\" *Nature*, 104, 354. The relativity confirmation.\n- Kuhn, T. S. (1962). *The Structure of Scientific Revolutions.* University of Chicago Press. ISBN 978-0226458120. The paradigm-shift complement.\n- Lakatos, I. (1970). \"Falsification and the Methodology of Scientific Research Programmes.\" in Lakatos & Musgrave (eds.), *Criticism and the Growth of Knowledge*. Cambridge University Press.\n- Ries, E. (2011). *The Lean Startup.* Crown Business. ISBN 978-0307887894. The startup application.\n- Tetlock, P. E. (2015). *Superforecasting.* Crown. ISBN 978-0804136693. Calibrated falsifiability in prediction.\n- Mayo, D. G. (1996). *Error and the Growth of Experimental Knowledge.* University of Chicago Press. The error-statistical framework.\n- Chollet, F. (2019). \"On the Measure of Intelligence.\" arXiv:1911.01547. Operationalizes \"intelligence\" into the falsifiable ARC / ARC-AGI benchmark — a template for turning \"AGI is near\" into a testable claim (2024–2026 AI application).\n- On benchmark contamination and reasoning robustness (2023–2025): peer-reviewed and arXiv work documenting train/test data contamination in LLM evaluations and accuracy drops when surface features of reasoning problems are perturbed — why a high AI benchmark score does not, by itself, confirm a capability claim. (Consult current surveys; specific reported scores are vendor figures, not independently audited.)\n\nFile v1.0.4:examples/ai-claims-agi-is-near-vs-testable-predictions-2024-2026.md\n\n# Method in Action: Separating Falsifiable from Unfalsifiable AI Claims (2024–2026)\n\n> *Example for the [falsifiability](../SKILL.md) skill.*\n\nBetween 2024 and 2026, public discourse around large AI models filled with two very different kinds of statement. Some were slogans — \"AGI is near,\" \"the model truly understands,\" \"scaling will just keep working\" — and some were concrete, dated predictions with numbers attached. Popper's criterion sorts them cleanly: a claim is empirical knowledge only if you can say in advance what observation would prove it wrong. This walkthrough runs the anchor claims through the falsifiability skill's own six-step Process.\n\n## Step 1 — State the claim\n\nTake three representative claims from the period:\n\n- **Claim A (slogan):** \"AGI is near.\"\n- **Claim B (mentalistic):** \"The model *truly understands* what it's saying.\"\n- **Claim C (testable):** \"By the end of 2025, a frontier model will exceed a specified accuracy threshold on the competition-mathematics benchmark AIME,\" or similar dated, metric-bound predictions of the kind researchers publish.\n\n*Who asserts them:* executives, commentators, and researchers, respectively. *Decision at stake:* whether an enterprise should bet a roadmap (or an investor a position) on imminent general capability. *Current evidential basis:* rapid, genuine benchmark gains from roughly 2023 onward, plus extrapolation.\n\n## Step 2 — Test whether empirical\n\n- **Claim A (\"AGI is near\")** is empirical *only if* \"AGI\" and \"near\" are defined. As typically used, neither is: \"AGI\" has no agreed operational definition, and \"near\" has no date. Without those, it is not yet a testable claim — it is a mood.\n- **Claim B (\"truly understands\")** invokes an inner mental state. As stated it is closer to metaphysics than to empirical science: no external observation is specified that would distinguish \"truly understands\" from \"produces the same outputs without understanding.\" Step 2 says: if it stays non-empirical, stop and label it belief.\n- **Claim C** is empirical: it is a statement about a measurable score on a fixed benchmark by a fixed date.\n\n## Step 3 — Specify falsification conditions\n\nForce each claim to complete: *\"This would be falsified if I observed: ___.\"*\n\n- **Claim A** can be *rescued into* an empirical claim by pinning it down — e.g. \"A single model will pass [a specified operationalization, such as the ARC-AGI abstraction-and-reasoning benchmark at human-level, or a stated economically-valuable-task bar] before 31 December 2026.\" Now it can fail. Note the discipline: the version that can be proven wrong is the only version worth arguing about.\n- **Claim B** resists completion. \"Understanding\" that predicts no observable difference from \"not understanding\" has no falsification condition. The productive move is to *replace* it with a behavioral proxy that does — e.g. \"the model will maintain accuracy when the same problem is presented with surface features (names, numbers, framing) changed,\" which is testable and, as documented in 2024–2025 work on reasoning robustness, sometimes fails.\n- **Claim C** already has its condition: *falsified if the reported score is below the stated threshold on the stated date.*\n\n## Step 4 — Plan to observe\n\n- *When observable:* Claim A at its stated deadline; Claim C at the benchmark's next reported evaluation; the Claim-B proxy on demand via a perturbed test set.\n- *How measured / who tracks:* published benchmark results and independent replications; assign a named owner to record the number when it lands, not to reinterpret the slogan afterward.\n- *Threshold for \"falsified\":* the pre-stated number and date. A crucial guardrail specific to AI benchmarks — **contamination**: if the evaluation items may have appeared in training data, a high score does not confirm the capability claim, so the observation plan must include held-out or post-cutoff items.\n\n## Step 5 — Pre-commit to action\n\n- If the operationalized **Claim A** deadline passes unmet: downgrade \"imminent AGI\" from a planning premise to a possibility, and stop staking irreversible roadmap bets on it.\n- For **Claim B**: if perturbation collapses performance, treat the system as pattern-competent but not robustly understanding, and design guardrails accordingly.\n- **Ad-hoc preservation risk (the core failure mode here):** the stock defenses — \"it's *almost* AGI,\" \"the benchmark was flawed,\" \"true AGI is just around the corner,\" \"you're moving the goalposts\" — are exactly the moves Popper flagged. Each rescue that explains away a missed prediction without a *new* falsifiable commitment converts the claim back into unfalsifiable belief.\n\n## Step 6 — Iterate\n\n- **Confirmed** (Claim C threshold met on clean data): keep monitoring; a passed test is not a general guarantee, only a survived attempt.\n- **Falsified** (deadline missed, proxy collapses): revise or abandon the stronger claim rather than defending it.\n- **Unfalsifiable** (Claim B left as \"it just understands\"): recognize it as belief and route the actual decision through the behavioral proxies instead.\n\n## The operational lesson\n\nThe right question to put to any 2024–2026 AI capability or safety claim is the skill's highest-leverage one: **\"What specific evidence, observable by when, would prove this wrong?\"** Claims that can answer — \"score below X on held-out set Y by date Z\" — are doing empirical work. Claims that cannot — \"it's near,\" \"it truly understands,\" \"trust the trajectory\" — are marketing or metaphysics wearing the costume of a prediction. The same test applies to safety claims: \"the model is safe\" is unfalsifiable until it is rewritten as \"the model will not produce [specified harmful output] under [specified red-team protocol],\" which can, and sometimes does, fail.\n\n*Sources: Karl Popper, *The Logic of Scientific Discovery* (1959; orig. *Logik der Forschung*, 1934) and *Conjectures and Refutations* (1963) for the falsifiability criterion. On operationalized AGI-style benchmarks: François Chollet, \"On the Measure of Intelligence\" (arXiv:1911.01547, 2019) and the ARC-AGI benchmark. On competition-math evaluation of frontier models: the AIME/MATH benchmark literature and vendors' publicly reported 2024–2025 results (treat specific figures as reported, not independently audited). On benchmark contamination and on reasoning robustness under input perturbation: peer-reviewed and arXiv work published 2023–2025 (e.g., studies of train/test contamination and of accuracy drops when numeric/surface features of problems are altered). Exact scores and dates vary by model and report; the reasoning here does not depend on any single unverified figure.*\n\nFile v1.0.4:examples/popper-1934-eddington-1919-eclipse-modern-applications.md\n\n# Method in Action: Popper 1934 + Eddington 1919 Eclipse + Modern Applications\n\n> *Example for the [falsifiability](../SKILL.md) skill.*\n\n**Karl Popper** (1902-1994) was an Austrian-British philosopher whose 1934 *Logik der Forschung* (translated as *The Logic of Scientific Discovery* in 1959) reshaped 20th-century philosophy of science. Popper had observed the rise of psychoanalysis (Freud, Adler) and Marxism in Vienna during the 1920s, and was struck by how their adherents claimed they \"explained\" everything — every observed event could be interpreted as confirming the theory. Popper's diagnosis: any theory that explains everything explains nothing. Such theories make no risky predictions; they cannot fail; therefore they are not empirical knowledge.\n\nPopper contrasted this with Einstein's general relativity (1915). Einstein had made a specific, quantitative prediction: light passing through the Sun's gravitational field during an eclipse would be deflected by 1.75 arcseconds, twice the Newtonian prediction. The 1919 total solar eclipse provided the empirical test. **Arthur Eddington's measurements during the eclipse** (in Príncipe and Brazil) confirmed the relativistic prediction within experimental error:\n\n> Eddington, A. S. (1919). \"Joint Eclipse Meeting of the Royal Society and the Royal Astronomical Society.\" *Nature*, 104, 354.\n\nPopper's emphasis: the test was real because relativity could have failed. If the measured deflection had been close to the Newtonian value or no deflection, relativity would have been falsified. The test mattered because failure was possible. The theory's confirmation was meaningful only because falsification was possible. This is the operational criterion of science.\n\nPopper's 1963 *Conjectures and Refutations* extended the framework to many domains:\n\n> \"The history of science, like the history of all human ideas, is a history of irresponsible dreams, of obstinacy, and of error. But science is one of the very few human activities — perhaps the only one — in which errors are systematically criticized and fairly often, in time, corrected. This is why we can say that, in science, we often learn from our mistakes, and why we can speak clearly and sensibly about making progress there.\"\n>\n> — Popper (1963), p. 216.\n\nThe framework has been applied extensively:\n\n**Lean Startup methodology.** Eric Ries's 2011 *The Lean Startup* is implicitly Popperian. Each iteration generates falsifiable hypotheses (\"if we A/B test this checkout flow change, conversion will improve by ≥5%\"), tests them empirically, and either confirms or rejects them based on data. The build-measure-learn cycle is operationalized falsifiability applied to product development.\n\n**Hypothesis-driven product development.** Modern product organizations (Amazon, Google, Microsoft) make explicit predictions about what user behavior change a feature will produce, then measure the actual change. Features that don't deliver the predicted impact are rolled back. The discipline is Popperian falsifiability at scale.\n\n**Investment thesis discipline.** Sophisticated investors specify what observations would falsify their investment thesis. If a thesis is \"Company X will reach $1B ARR by 2027,\" then quarterly milestones are set as falsifiability triggers. If Q3 ARR is below threshold X, the thesis is at risk; if it's below threshold Y, the thesis is falsified and the position is reconsidered.\n\n**Strategic planning.** Sophisticated strategy processes specify what conditions would change the strategy. \"Our differentiation rests on X capability; if competitors X' achieves parity in this capability, we need to revisit.\" The conditions are falsifiability triggers for the strategic thesis.\n\n**Scientific peer review and reproducibility.** The \"reproducibility crisis\" in social sciences (2010s onward) revealed that many published findings could not be replicated. Modern reforms (pre-registration of hypotheses, public datasets, replication studies) are explicit Popperian: making research findings falsifiable in practice, not just in principle.\n\n**Defense against unfalsifiable claims:** Several domains demonstrate the operational cost of unfalsifiable claims:\n\n- **Cult and conspiracy theory dynamics.** True believers find ways to \"explain\" any disconfirming evidence (the lack of UFO landing means the aliens are testing us; the lack of cult prediction fulfillment means we weren't ready). The structural feature is unfalsifiability — no observation could change their mind.\n- **Some macroeconomic claims.** \"Austerity will eventually produce growth\" or \"more stimulus will eventually produce inflation\" — without timeframes and thresholds, these claims are unfalsifiable. Any short-term failure can be attributed to \"not yet.\"\n- **Vague leadership pronouncements.** \"Our culture is strong\" or \"we have great execution\" without metrics or thresholds are unfalsifiable. They sound informative but contain no testable content.\n\nThree operational lessons:\n\n**First, the falsifiability question is the highest-leverage critical-thinking tool.** \"What would change your mind?\" — applied to any claim, your own or another's — produces dramatically better thinking than the alternative of accumulating confirmations.\n\n**Second, ad-hoc preservation is the failure mode.** When a prediction fails, the temptation is to modify the theory to accommodate the failure (\"the strategy worked, but X happened to prevent it\"). Each ad-hoc modification reduces the scientific status of the claim. Sometimes modifications are justified; usually they are face-saving rationalization that compounds error.\n\n**Third, specifying falsifiability conditions in advance is the discipline.** Conditions that you specify after the result are not falsifiability; they are post-hoc rationalization. The Popperian discipline is to commit to falsification triggers before the relevant observations are made — when you might genuinely be wrong, not after the fact.\n\nFile v1.0.4:skill-card.md\n\n## Description: <br>\nHelps agents turn empirical claims, strategies, experiments, and investment theses into testable falsification conditions and monitoring plans. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[deciqai](https://clawhub.ai/user/deciqai) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nEmployees, external users, and developers use this skill to make empirical claims testable before acting on them. It supports strategy reviews, product experiments, investment theses, and AI capability claims by defining falsification conditions, observation plans, and actions if the claim fails. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Reasoning guidance may influence business, investment, product, or capability decisions if users treat the framework as a substitute for review. <br>\nMitigation: Review the framework and resulting falsification criteria before relying on them for high-stakes decisions. <br>\nRisk: User-specific examples may contain sensitive strategy, investment, or product information if persisted back into the skill. <br>\nMitigation: Avoid saving sensitive examples unless those notes are intentionally approved for persistence. <br>\n\n\n## Reference(s): <br>\n- [ClawHub release page](https://clawhub.ai/deciqai/skills/falsifiability) <br>\n- [Sources - falsifiability](references/sources.md) <br>\n- [Method in Action: Popper 1934 + Eddington 1919 Eclipse + Modern Applications](examples/popper-1934-eddington-1919-eclipse-modern-applications.md) <br>\n- [Separating falsifiable from unfalsifiable AI claims (2024-2026)](examples/ai-claims-agi-is-near-vs-testable-predictions-2024-2026.md) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, guidance] <br>\n**Output Format:** [Markdown analysis template and coaching prompts] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Produces a structured falsifiability analysis with claim, evidence basis, falsification threshold, observation plan, owner, and action if falsified.] <br>\n\n## Skill Version(s): <br>\n1.0.4 (source: server release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.0.3: 5 files, 8426 bytes\n\nFiles: examples/popper-1934-eddington-1919-eclipse-modern-applications.md (5999b), references/sources.md (1242b), skill-card.md (1720b), SKILL.md (6804b), _meta.json (133b)\n\nFile v1.0.3:SKILL.md\n\n---\nname: falsifiability\ndescription: \"Activate when: user says 'what would prove this wrong', 'how do we know if our strategy is working', 'what would change your mind', 'this claim feels unfalsifiable', or is designing a hypothesis/experiment/investment thesis that needs to be made testable.\n  Do NOT activate when: the claim is genuinely non-empirical (ethical, aesthetic, philosophical); or the cost of running the test exceeds the value of the knowledge.\"\n---\n\n# Falsifiability\n\n## Overview\n\nA meaningful empirical claim must specify what observations would refute it. Claims that resist all possible refutation are not science — they are unfalsifiable belief. Formalized by Karl Popper (1934): science progresses not by accumulating confirmations but by surviving rigorous attempts at falsification. More-specific claims are more falsifiable; ad-hoc modifications that explain away failures destroy a claim's scientific status.\n\nComposes with `confirmation-bias` (falsifiability is the structural counter), `abductive-reasoning` (generates hypotheses; this skill tests them), `bayesian-reasoning`, `critical-thinking`.\n\n## When to Use\n\n- Designing OKRs, KPIs, or strategic goals; writing or evaluating investment theses\n- Designing experiments (A/B tests, product hypotheses, market entry)\n- Evaluating consultant/advisor recommendations or diagnosing vague leadership claims\n- Someone says \"what would change your mind,\" \"how would you know you're wrong,\" \"Popper\"\n\n**Not when:** genuinely non-empirical (philosophical, ethical, aesthetic); test cost exceeds its value.\n\n## Coaching Novices (Adaptive Front Door)\n\n- **Engine mode:** user has a specific claim → run The Process directly.\n- **Coach mode:** user is new → guide step by step.\n\nIn Coach mode, respond one step at a time. Each [WAIT] is a hard stop — output only that step's question, then stop.\n\n1. One-line: before acting on a claim, ask what observation would refute it — if \"nothing would,\" it's belief, not knowledge.\n2. Check fit: if the claim is genuinely metaphysical (ethical, aesthetic), falsifiability doesn't apply.\n3. Elicit the claim and its current evidential basis.\n> **[WAIT — do not advance until user responds]**\n4. Ask: what specific observation would falsify this? When and how would you observe it?\n> **[WAIT — do not advance until user responds]**\n5. Close: falsifiability conditions specified + monitoring plan + commitment to act on disconfirming evidence.\n> **[WAIT — do not advance until user responds]**\n\n## The Process\n\n**Step 1 — State the claim:** claim / who asserts it / decision dependent on it / current evidential basis.\n\n**Step 2 — Test whether empirical:** claim about how the world works (empirical) or values/aesthetics (non-empirical)? If non-empirical, stop here.\n\n**Step 3 — Specify falsification conditions:** complete \"This claim would be falsified if I observed: ___\" — specific, observable, time-bounded. If you cannot complete it, the claim is not falsifiable as stated.\n\n**Step 4 — Plan to observe:** when is the observation possible / how measured / who tracks it / threshold for \"falsified.\"\n\n**Step 5 — Pre-commit to action:** if the falsifying observation occurs, what will you do? Is any theory modification itself falsifiable?\n\n**Step 6 — Iterate:** confirmed → keep monitoring; falsified → revise/abandon; unfalsifiable → recognize as belief.\n\n## Output Template\n\n```markdown\n# Falsifiability Analysis: <claim>\nClaim: | Asserted by: | Decision at stake: | Current basis:\nEmpirical: Y/N | Observation type:\nFalsified if: | Threshold: | Observable when/how: | Owner:\nIf falsified, action: | Ad-hoc preservation risk:\n```\n\n*→ Method in Action: [Popper 1934 + Eddington 1919 Eclipse + Modern Applications](examples/popper-1934-eddington-1919-eclipse-modern-applications.md)*\n\n## Pack: Vague → Falsifiable\n\n| Vague (unfalsifiable) | Falsifiable form |\n|---|---|\n| \"We have PMF\" | \"≥40% of users would be 'very disappointed' without the product\" |\n| \"Our outbound is working\" | \"5% of cold emails convert to qualified opps within 30 days\" |\n| \"Our culture is strong\" | \"Employee NPS ≥40 in next quarterly survey\" |\n| \"This stock is undervalued\" | \"Stock <12 P/E within 12 months; otherwise thesis is wrong\" |\n| \"Users want feature X\" | \"≥30% complete the new flow within 14 days of launch\" |\n\n## Applying It Well\n\n- Specify falsifiability conditions *before* the observation window opens — post-result conditions are rationalization.\n- More specific = more falsifiable. Quantify wherever possible.\n- When a prediction fails, ask: is the modification of your theory itself falsifiable?\n- Apply to others: \"What would change your mind?\" is the highest-leverage critical-thinking question.\n\n*→ Primary sources: [references/sources.md](references/sources.md)*\n\n## Common Rationalizations\n\n**[D] = designed upfront | [O] = observed in real use. [O] entries are more valuable.**\n\n| Fake move | Reality |\n|---|---|\n| [D] \"I just have a strong gut feeling\" | Gut feelings are not falsifiable. Without a test, you can't tell right from wrong. |\n| [D] \"We need more data to decide\" | If you can't specify what data would change your mind, you're not doing data-driven analysis. |\n| [D] \"The strategy just needs more time\" | Specify the timeframe and metrics in advance. |\n| [D] \"It's because of external factors\" | If external factors can always be invoked, the original claim was unfalsifiable. |\n| [D] \"It's too early to evaluate\" | If you can't specify when evaluation is appropriate, the claim is unfalsifiable. |\n| [D] \"We're in a special situation\" | This defense converts a falsifiable claim into an unfalsifiable one. Resist. |\n| *→ Add [O] entries here after each real use — paste the actual failure pattern* | *What went wrong and why* |\n\n## Red Flags\n\n- Claim asserted without specifying what would refute it\n- \"The strategy is working\" without metrics or thresholds\n- Predictions modified ad-hoc to accommodate failures\n- \"External factors\" invoked to explain disconfirming evidence\n- Person cannot specify what would change their mind\n\n## Verification\n\n- [ ] Claim specifically stated; empirical vs. non-empirical determined\n- [ ] Specific, observable, time-bounded falsification conditions defined\n- [ ] Monitoring mechanism in place with a named owner\n- [ ] Advance commitment to act on falsification made\n- [ ] Risk of ad-hoc preservation recognized\n\n---\n\n*Part of **deciqAI Knowledge Skills** — 164 open-source thinking skills that make rigor executable for AI agents. The same skills power every deciqAI agent, which runs them autonomously to operate your company. **See it run → https://www.deciqai.com/c/falsifiability** · ⭐ Star the repo → https://github.com/deciqAI/knowledge-skills · Contributions welcome.*\n\nFile v1.0.3:_meta.json\n\n{\n  \"ownerId\": \"kn754b8sk22s8c6gjxt02bftbn88q7ye\",\n  \"slug\": \"falsifiability\",\n  \"version\": \"1.0.3\",\n  \"publishedAt\": 1783508536593\n}\n\nFile v1.0.3:references/sources.md\n\n# Sources — falsifiability\n\n> *Primary sources for the [falsifiability](../SKILL.md) skill.*\n\n## Sources\n\n- Popper, K. R. (1934). *Logik der Forschung.* Vienna: Springer. English: *The Logic of Scientific Discovery* (1959). London: Hutchinson. ISBN 978-0415278447.\n- Popper, K. R. (1963). *Conjectures and Refutations: The Growth of Scientific Knowledge.* London: Routledge. ISBN 978-0415285940.\n- Eddington, A. S. (1919). \"Joint Eclipse Meeting of the Royal Society and the Royal Astronomical Society.\" *Nature*, 104, 354. The relativity confirmation.\n- Kuhn, T. S. (1962). *The Structure of Scientific Revolutions.* University of Chicago Press. ISBN 978-0226458120. The paradigm-shift complement.\n- Lakatos, I. (1970). \"Falsification and the Methodology of Scientific Research Programmes.\" in Lakatos & Musgrave (eds.), *Criticism and the Growth of Knowledge*. Cambridge University Press.\n- Ries, E. (2011). *The Lean Startup.* Crown Business. ISBN 978-0307887894. The startup application.\n- Tetlock, P. E. (2015). *Superforecasting.* Crown. ISBN 978-0804136693. Calibrated falsifiability in prediction.\n- Mayo, D. G. (1996). *Error and the Growth of Experimental Knowledge.* University of Chicago Press. The error-statistical framework.\n\nFile v1.0.3:examples/popper-1934-eddington-1919-eclipse-modern-applications.md\n\n# Method in Action: Popper 1934 + Eddington 1919 Eclipse + Modern Applications\n\n> *Example for the [falsifiability](../SKILL.md) skill.*\n\n**Karl Popper** (1902-1994) was an Austrian-British philosopher whose 1934 *Logik der Forschung* (translated as *The Logic of Scientific Discovery* in 1959) reshaped 20th-century philosophy of science. Popper had observed the rise of psychoanalysis (Freud, Adler) and Marxism in Vienna during the 1920s, and was struck by how their adherents claimed they \"explained\" everything — every observed event could be interpreted as confirming the theory. Popper's diagnosis: any theory that explains everything explains nothing. Such theories make no risky predictions; they cannot fail; therefore they are not empirical knowledge.\n\nPopper contrasted this with Einstein's general relativity (1915). Einstein had made a specific, quantitative prediction: light passing through the Sun's gravitational field during an eclipse would be deflected by 1.75 arcseconds, twice the Newtonian prediction. The 1919 total solar eclipse provided the empirical test. **Arthur Eddington's measurements during the eclipse** (in Príncipe and Brazil) confirmed the relativistic prediction within experimental error:\n\n> Eddington, A. S. (1919). \"Joint Eclipse Meeting of the Royal Society and the Royal Astronomical Society.\" *Nature*, 104, 354.\n\nPopper's emphasis: the test was real because relativity could have failed. If the measured deflection had been close to the Newtonian value or no deflection, relativity would have been falsified. The test mattered because failure was possible. The theory's confirmation was meaningful only because falsification was possible. This is the operational criterion of science.\n\nPopper's 1963 *Conjectures and Refutations* extended the framework to many domains:\n\n> \"The history of science, like the history of all human ideas, is a history of irresponsible dreams, of obstinacy, and of error. But science is one of the very few human activities — perhaps the only one — in which errors are systematically criticized and fairly often, in time, corrected. This is why we can say that, in science, we often learn from our mistakes, and why we can speak clearly and sensibly about making progress there.\"\n>\n> — Popper (1963), p. 216.\n\nThe framework has been applied extensively:\n\n**Lean Startup methodology.** Eric Ries's 2011 *The Lean Startup* is implicitly Popperian. Each iteration generates falsifiable hypotheses (\"if we A/B test this checkout flow change, conversion will improve by ≥5%\"), tests them empirically, and either confirms or rejects them based on data. The build-measure-learn cycle is operationalized falsifiability applied to product development.\n\n**Hypothesis-driven product development.** Modern product organizations (Amazon, Google, Microsoft) make explicit predictions about what user behavior change a feature will produce, then measure the actual change. Features that don't deliver the predicted impact are rolled back. The discipline is Popperian falsifiability at scale.\n\n**Investment thesis discipline.** Sophisticated investors specify what observations would falsify their investment thesis. If a thesis is \"Company X will reach $1B ARR by 2027,\" then quarterly milestones are set as falsifiability triggers. If Q3 ARR is below threshold X, the thesis is at risk; if it's below threshold Y, the thesis is falsified and the position is reconsidered.\n\n**Strategic planning.** Sophisticated strategy processes specify what conditions would change the strategy. \"Our differentiation rests on X capability; if competitors X' achieves parity in this capability, we need to revisit.\" The conditions are falsifiability triggers for the strategic thesis.\n\n**Scientific peer review and reproducibility.** The \"reproducibility crisis\" in social sciences (2010s onward) revealed that many published findings could not be replicated. Modern reforms (pre-registration of hypotheses, public datasets, replication studies) are explicit Popperian: making research findings falsifiable in practice, not just in principle.\n\n**Defense against unfalsifiable claims:** Several domains demonstrate the operational cost of unfalsifiable claims:\n\n- **Cult and conspiracy theory dynamics.** True believers find ways to \"explain\" any disconfirming evidence (the lack of UFO landing means the aliens are testing us; the lack of cult prediction fulfillment means we weren't ready). The structural feature is unfalsifiability — no observation could change their mind.\n- **Some macroeconomic claims.** \"Austerity will eventually produce growth\" or \"more stimulus will eventually produce inflation\" — without timeframes and thresholds, these claims are unfalsifiable. Any short-term failure can be attributed to \"not yet.\"\n- **Vague leadership pronouncements.** \"Our culture is strong\" or \"we have great execution\" without metrics or thresholds are unfalsifiable. They sound informative but contain no testable content.\n\nThree operational lessons:\n\n**First, the falsifiability question is the highest-leverage critical-thinking tool.** \"What would change your mind?\" — applied to any claim, your own or another's — produces dramatically better thinking than the alternative of accumulating confirmations.\n\n**Second, ad-hoc preservation is the failure mode.** When a prediction fails, the temptation is to modify the theory to accommodate the failure (\"the strategy worked, but X happened to prevent it\"). Each ad-hoc modification reduces the scientific status of the claim. Sometimes modifications are justified; usually they are face-saving rationalization that compounds error.\n\n**Third, specifying falsifiability conditions in advance is the discipline.** Conditions that you specify after the result are not falsifiability; they are post-hoc rationalization. The Popperian discipline is to commit to falsification triggers before the relevant observations are made — when you might genuinely be wrong, not after the fact.\n\nFile v1.0.3:skill-card.md\n\n## Description: <br>\nHelps agents turn empirical claims, strategies, experiments, and theses into specific observations that could prove them wrong. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[deciqai](https://clawhub.ai/user/deciqai) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nEmployees, external users, and developers use this skill to evaluate whether empirical claims are testable, define falsification conditions, and plan what to measure before acting on a hypothesis. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Review before execution as proposals could introduce incorrect or misleading guidance into skills. <br>\nMitigation: Review and scan skill before deployment. <br>\n\n## Reference(s): <br>\n- [Sources - falsifiability](references/sources.md) <br>\n- [Method in Action: Popper 1934 + Eddington 1919 Eclipse + Modern Applications](examples/popper-1934-eddington-1919-eclipse-modern-applications.md) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, guidance] <br>\n**Output Format:** [Markdown] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Produces reasoning and structured analysis only; the release security summary reports no hidden execution, sensitive access, or persistence.] <br>\n\n## Skill Version(s): <br>\n1.0.3 (source: server release metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.0.2: 5 files, 8651 bytes\n\nFiles: examples/popper-1934-eddington-1919-eclipse-modern-applications.md (5999b), references/sources.md (1242b), skill-card.md (2148b), SKILL.md (6908b), _meta.json (133b)\n\nFile v1.0.2:SKILL.md\n\n---\nname: falsifiability\ndescription: \"Activate when: user says 'what would prove this wrong', 'how do we know if our strategy is working', 'what would change your mind', 'this claim feels unfalsifiable', or is designing a hypothesis/experiment/investment thesis that needs to be made testable.\n  Do NOT activate when: the claim is genuinely non-empirical (ethical, aesthetic, philosophical); or the cost of running the test exceeds the value of the knowledge.\"\n---\n\n# Falsifiability\n\n## Overview\n\nA meaningful empirical claim must specify what observations would refute it. Claims that resist all possible refutation are not science — they are unfalsifiable belief. Formalized by Karl Popper (1934): science progresses not by accumulating confirmations but by surviving rigorous attempts at falsification. More-specific claims are more falsifiable; ad-hoc modifications that explain away failures destroy a claim's scientific status.\n\nComposes with `confirmation-bias` (falsifiability is the structural counter), `abductive-reasoning` (generates hypotheses; this skill tests them), `bayesian-reasoning`, `critical-thinking`.\n\n## When to Use\n\n- Designing OKRs, KPIs, or strategic goals; writing or evaluating investment theses\n- Designing experiments (A/B tests, product hypotheses, market entry)\n- Evaluating consultant/advisor recommendations or diagnosing vague leadership claims\n- Someone says \"what would change your mind,\" \"how would you know you're wrong,\" \"Popper\"\n\n**Not when:** genuinely non-empirical (philosophical, ethical, aesthetic); test cost exceeds its value.\n\n## Coaching Novices (Adaptive Front Door)\n\n- **Engine mode:** user has a specific claim → run The Process directly.\n- **Coach mode:** user is new → guide step by step.\n\nIn Coach mode, respond one step at a time. Each [WAIT] is a hard stop — output only that step's question, then stop.\n\n1. One-line: before acting on a claim, ask what observation would refute it — if \"nothing would,\" it's belief, not knowledge.\n2. Check fit: if the claim is genuinely metaphysical (ethical, aesthetic), falsifiability doesn't apply.\n3. Elicit the claim and its current evidential basis.\n> **[WAIT — do not advance until user responds]**\n4. Ask: what specific observation would falsify this? When and how would you observe it?\n> **[WAIT — do not advance until user responds]**\n5. Close: falsifiability conditions specified + monitoring plan + commitment to act on disconfirming evidence.\n> **[WAIT — do not advance until user responds]**\n\n## The Process\n\n**Step 1 — State the claim:** claim / who asserts it / decision dependent on it / current evidential basis.\n\n**Step 2 — Test whether empirical:** claim about how the world works (empirical) or values/aesthetics (non-empirical)? If non-empirical, stop here.\n\n**Step 3 — Specify falsification conditions:** complete \"This claim would be falsified if I observed: ___\" — specific, observable, time-bounded. If you cannot complete it, the claim is not falsifiable as stated.\n\n**Step 4 — Plan to observe:** when is the observation possible / how measured / who tracks it / threshold for \"falsified.\"\n\n**Step 5 — Pre-commit to action:** if the falsifying observation occurs, what will you do? Is any theory modification itself falsifiable?\n\n**Step 6 — Iterate:** confirmed → keep monitoring; falsified → revise/abandon; unfalsifiable → recognize as belief.\n\n## Output Template\n\n```markdown\n# Falsifiability Analysis: <claim>\nClaim: | Asserted by: | Decision at stake: | Current basis:\nEmpirical: Y/N | Observation type:\nFalsified if: | Threshold: | Observable when/how: | Owner:\nIf falsified, action: | Ad-hoc preservation risk:\n```\n\n*→ Method in Action: [Popper 1934 + Eddington 1919 Eclipse + Modern Applications](examples/popper-1934-eddington-1919-eclipse-modern-applications.md)*\n\n## Pack: Vague → Falsifiable\n\n| Vague (unfalsifiable) | Falsifiable form |\n|---|---|\n| \"We have PMF\" | \"≥40% of users would be 'very disappointed' without the product\" |\n| \"Our outbound is working\" | \"5% of cold emails convert to qualified opps within 30 days\" |\n| \"Our culture is strong\" | \"Employee NPS ≥40 in next quarterly survey\" |\n| \"This stock is undervalued\" | \"Stock <12 P/E within 12 months; otherwise thesis is wrong\" |\n| \"Users want feature X\" | \"≥30% complete the new flow within 14 days of launch\" |\n\n## Applying It Well\n\n- Specify falsifiability conditions *before* the observation window opens — post-result conditions are rationalization.\n- More specific = more falsifiable. Quantify wherever possible.\n- When a prediction fails, ask: is the modification of your theory itself falsifiable?\n- Apply to others: \"What would change your mind?\" is the highest-leverage critical-thinking question.\n\n*→ Primary sources: [references/sources.md](references/sources.md)*\n\n## Common Rationalizations\n\n**[D] = designed upfront | [O] = observed in real use. [O] entries are more valuable.**\n\n| Fake move | Reality |\n|---|---|\n| [D] \"I just have a strong gut feeling\" | Gut feelings are not falsifiable. Without a test, you can't tell right from wrong. |\n| [D] \"We need more data to decide\" | If you can't specify what data would change your mind, you're not doing data-driven analysis. |\n| [D] \"The strategy just needs more time\" | Specify the timeframe and metrics in advance. |\n| [D] \"It's because of external factors\" | If external factors can always be invoked, the original claim was unfalsifiable. |\n| [D] \"It's too early to evaluate\" | If you can't specify when evaluation is appropriate, the claim is unfalsifiable. |\n| [D] \"We're in a special situation\" | This defense converts a falsifiable claim into an unfalsifiable one. Resist. |\n| *→ Add [O] entries here after each real use — paste the actual failure pattern* | *What went wrong and why* |\n\n## Red Flags\n\n- Claim asserted without specifying what would refute it\n- \"The strategy is working\" without metrics or thresholds\n- Predictions modified ad-hoc to accommodate failures\n- \"External factors\" invoked to explain disconfirming evidence\n- Person cannot specify what would change their mind\n\n## Verification\n\n- [ ] Claim specifically stated; empirical vs. non-empirical determined\n- [ ] Specific, observable, time-bounded falsification conditions defined\n- [ ] Monitoring mechanism in place with a named owner\n- [ ] Advance commitment to act on falsification made\n- [ ] Risk of ad-hoc preservation recognized\n\n---\n\n*Part of **deciqAI Knowledge Skills** — 163 open-source thinking skills that make rigor executable for AI agents. The same skills power every deciqAI agent, which runs them autonomously to operate your company. **See it run → https://www.deciqai.com/skills/falsifiability?utm_source=clawhub&utm_medium=marketplace&utm_campaign=knowledge-skills&utm_content=falsifiability** · ⭐ Star the repo → https://github.com/deciqAI/knowledge-skills · Contributions welcome.*\n\nFile v1.0.2:_meta.json\n\n{\n  \"ownerId\": \"kn754b8sk22s8c6gjxt02bftbn88q7ye\",\n  \"slug\": \"falsifiability\",\n  \"version\": \"1.0.2\",\n  \"publishedAt\": 1783471659419\n}\n\nFile v1.0.2:references/sources.md\n\n# Sources — falsifiability\n\n> *Primary sources for the [falsifiability](../SKILL.md) skill.*\n\n## Sources\n\n- Popper, K. R. (1934). *Logik der Forschung.* Vienna: Springer. English: *The Logic of Scientific Discovery* (1959). London: Hutchinson. ISBN 978-0415278447.\n- Popper, K. R. (1963). *Conjectures and Refutations: The Growth of Scientific Knowledge.* London: Routledge. ISBN 978-0415285940.\n- Eddington, A. S. (1919). \"Joint Eclipse Meeting of the Royal Society and the Royal Astronomical Society.\" *Nature*, 104, 354. The relativity confirmation.\n- Kuhn, T. S. (1962). *The Structure of Scientific Revolutions.* University of Chicago Press. ISBN 978-0226458120. The paradigm-shift complement.\n- Lakatos, I. (1970). \"Falsification and the Methodology of Scientific Research Programmes.\" in Lakatos & Musgrave (eds.), *Criticism and the Growth of Knowledge*. Cambridge University Press.\n- Ries, E. (2011). *The Lean Startup.* Crown Business. ISBN 978-0307887894. The startup application.\n- Tetlock, P. E. (2015). *Superforecasting.* Crown. ISBN 978-0804136693. Calibrated falsifiability in prediction.\n- Mayo, D. G. (1996). *Error and the Growth of Experimental Knowledge.* University of Chicago Press. The error-statistical framework.\n\nFile v1.0.2:examples/popper-1934-eddington-1919-eclipse-modern-applications.md\n\n# Method in Action: Popper 1934 + Eddington 1919 Eclipse + Modern Applications\n\n> *Example for the [falsifiability](../SKILL.md) skill.*\n\n**Karl Popper** (1902-1994) was an Austrian-British philosopher whose 1934 *Logik der Forschung* (translated as *The Logic of Scientific Discovery* in 1959) reshaped 20th-century philosophy of science. Popper had observed the rise of psychoanalysis (Freud, Adler) and Marxism in Vienna during the 1920s, and was struck by how their adherents claimed they \"explained\" everything — every observed event could be interpreted as confirming the theory. Popper's diagnosis: any theory that explains everything explains nothing. Such theories make no risky predictions; they cannot fail; therefore they are not empirical knowledge.\n\nPopper contrasted this with Einstein's general relativity (1915). Einstein had made a specific, quantitative prediction: light passing through the Sun's gravitational field during an eclipse would be deflected by 1.75 arcseconds, twice the Newtonian prediction. The 1919 total solar eclipse provided the empirical test. **Arthur Eddington's measurements during the eclipse** (in Príncipe and Brazil) confirmed the relativistic prediction within experimental error:\n\n> Eddington, A. S. (1919). \"Joint Eclipse Meeting of the Royal Society and the Royal Astronomical Society.\" *Nature*, 104, 354.\n\nPopper's emphasis: the test was real because relativity could have failed. If the measured deflection had been close to the Newtonian value or no deflection, relativity would have been falsified. The test mattered because failure was possible. The theory's confirmation was meaningful only because falsification was possible. This is the operational criterion of science.\n\nPopper's 1963 *Conjectures and Refutations* extended the framework to many domains:\n\n> \"The history of science, like the history of all human ideas, is a history of irresponsible dreams, of obstinacy, and of error. But science is one of the very few human activities — perhaps the only one — in which errors are systematically criticized and fairly often, in time, corrected. This is why we can say that, in science, we often learn from our mistakes, and why we can speak clearly and sensibly about making progress there.\"\n>\n> — Popper (1963), p. 216.\n\nThe framework has been applied extensively:\n\n**Lean Startup methodology.** Eric Ries's 2011 *The Lean Startup* is implicitly Popperian. Each iteration generates falsifiable hypotheses (\"if we A/B test this checkout flow change, conversion will improve by ≥5%\"), tests them empirically, and either confirms or rejects them based on data. The build-measure-learn cycle is operationalized falsifiability applied to product development.\n\n**Hypothesis-driven product development.** Modern product organizations (Amazon, Google, Microsoft) make explicit predictions about what user behavior change a feature will produce, then measure the actual change. Features that don't deliver the predicted impact are rolled back. The discipline is Popperian falsifiability at scale.\n\n**Investment thesis discipline.** Sophisticated investors specify what observations would falsify their investment thesis. If a thesis is \"Company X will reach $1B ARR by 2027,\" then quarterly milestones are set as falsifiability triggers. If Q3 ARR is below threshold X, the thesis is at risk; if it's below threshold Y, the thesis is falsified and the position is reconsidered.\n\n**Strategic planning.** Sophisticated strategy processes specify what conditions would change the strategy. \"Our differentiation rests on X capability; if competitors X' achieves parity in this capability, we need to revisit.\" The conditions are falsifiability triggers for the strategic thesis.\n\n**Scientific peer review and reproducibility.** The \"reproducibility crisis\" in social sciences (2010s onward) revealed that many published findings could not be replicated. Modern reforms (pre-registration of hypotheses, public datasets, replication studies) are explicit Popperian: making research findings falsifiable in practice, not just in principle.\n\n**Defense against unfalsifiable claims:** Several domains demonstrate the operational cost of unfalsifiable claims:\n\n- **Cult and conspiracy theory dynamics.** True believers find ways to \"explain\" any disconfirming evidence (the lack of UFO landing means the aliens are testing us; the lack of cult prediction fulfillment means we weren't ready). The structural feature is unfalsifiability — no observation could change their mind.\n- **Some macroeconomic claims.** \"Austerity will eventually produce growth\" or \"more stimulus will eventually produce inflation\" — without timeframes and thresholds, these claims are unfalsifiable. Any short-term failure can be attributed to \"not yet.\"\n- **Vague leadership pronouncements.** \"Our culture is strong\" or \"we have great execution\" without metrics or thresholds are unfalsifiable. They sound informative but contain no testable content.\n\nThree operational lessons:\n\n**First, the falsifiability question is the highest-leverage critical-thinking tool.** \"What would change your mind?\" — applied to any claim, your own or another's — produces dramatically better thinking than the alternative of accumulating confirmations.\n\n**Second, ad-hoc preservation is the failure mode.** When a prediction fails, the temptation is to modify the theory to accommodate the failure (\"the strategy worked, but X happened to prevent it\"). Each ad-hoc modification reduces the scientific status of the claim. Sometimes modifications are justified; usually they are face-saving rationalization that compounds error.\n\n**Third, specifying falsifiability conditions in advance is the discipline.** Conditions that you specify after the result are not falsifiability; they are post-hoc rationalization. The Popperian discipline is to commit to falsification triggers before the relevant observations are made — when you might genuinely be wrong, not after the fact.\n\nFile v1.0.2:skill-card.md\n\n## Description: <br>\nFalsifiability helps an agent turn empirical claims, hypotheses, strategies, experiments, and investment theses into specific observations that could refute them. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[deciqai](https://clawhub.ai/user/deciqai) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nEmployees, developers, and analysts use this skill to test whether a claim is empirical, define observable falsification conditions, plan measurement, and pre-commit to action if the claim is refuted. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill can be misapplied to ethical, aesthetic, philosophical, or otherwise non-empirical claims. <br>\nMitigation: Use the documented fit check and stop when falsifiability does not apply. <br>\nRisk: A falsifiability analysis can still produce misleading thresholds or actions if the claim, measurement window, or owner is poorly specified. <br>\nMitigation: Review the claim, observation method, threshold, owner, and pre-committed action before using the output for decisions. <br>\n\n\n## Reference(s): <br>\n- [Primary Sources](references/sources.md) <br>\n- [Method in Action: Popper 1934 + Eddington 1919 Eclipse + Modern Applications](examples/popper-1934-eddington-1919-eclipse-modern-applications.md) <br>\n- [ClawHub Skill Page](https://clawhub.ai/deciqai/skills/falsifiability) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, guidance] <br>\n**Output Format:** [Markdown analysis template or step-by-step coaching prompts] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [May stop at WAIT checkpoints in coach mode so the agent asks one question at a time.] <br>\n\n## Skill Version(s): <br>\n1.0.2 (source: server release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.0.1: 5 files, 8515 bytes\n\nFiles: examples/popper-1934-eddington-1919-eclipse-modern-applications.md (5999b), references/sources.md (1242b), skill-card.md (2083b), SKILL.md (6787b), _meta.json (133b)\n\nFile v1.0.1:SKILL.md\n\n---\nname: falsifiability\ndescription: \"Activate when: user says 'what would prove this wrong', 'how do we know if our strategy is working', 'what would change your mind', 'this claim feels unfalsifiable', or is designing a hypothesis/experiment/investment thesis that needs to be made testable.\n  Do NOT activate when: the claim is genuinely non-empirical (ethical, aesthetic, philosophical); or the cost of running the test exceeds the value of the knowledge.\"\n---\n\n# Falsifiability\n\n## Overview\n\nA meaningful empirical claim must specify what observations would refute it. Claims that resist all possible refutation are not science — they are unfalsifiable belief. Formalized by Karl Popper (1934): science progresses not by accumulating confirmations but by surviving rigorous attempts at falsification. More-specific claims are more falsifiable; ad-hoc modifications that explain away failures destroy a claim's scientific status.\n\nComposes with [`confirmation-bias`](../confirmation-bias/SKILL.md) (falsifiability is the structural counter), [`abductive-reasoning`](../abductive-reasoning/SKILL.md) (generates hypotheses; this skill tests them), [`bayesian-reasoning`](../bayesian-reasoning/SKILL.md), [`critical-thinking`](../critical-thinking/SKILL.md).\n\n## When to Use\n\n- Designing OKRs, KPIs, or strategic goals; writing or evaluating investment theses\n- Designing experiments (A/B tests, product hypotheses, market entry)\n- Evaluating consultant/advisor recommendations or diagnosing vague leadership claims\n- Someone says \"what would change your mind,\" \"how would you know you're wrong,\" \"Popper\"\n\n**Not when:** genuinely non-empirical (philosophical, ethical, aesthetic); test cost exceeds its value.\n\n## Coaching Novices (Adaptive Front Door)\n\n- **Engine mode:** user has a specific claim → run The Process directly.\n- **Coach mode:** user is new → guide step by step.\n\nIn Coach mode, respond one step at a time. Each [WAIT] is a hard stop — output only that step's question, then stop.\n\n1. One-line: before acting on a claim, ask what observation would refute it — if \"nothing would,\" it's belief, not knowledge.\n2. Check fit: if the claim is genuinely metaphysical (ethical, aesthetic), falsifiability doesn't apply.\n3. Elicit the claim and its current evidential basis.\n> **[WAIT — do not advance until user responds]**\n4. Ask: what specific observation would falsify this? When and how would you observe it?\n> **[WAIT — do not advance until user responds]**\n5. Close: falsifiability conditions specified + monitoring plan + commitment to act on disconfirming evidence.\n> **[WAIT — do not advance until user responds]**\n\n## The Process\n\n**Step 1 — State the claim:** claim / who asserts it / decision dependent on it / current evidential basis.\n\n**Step 2 — Test whether empirical:** claim about how the world works (empirical) or values/aesthetics (non-empirical)? If non-empirical, stop here.\n\n**Step 3 — Specify falsification conditions:** complete \"This claim would be falsified if I observed: ___\" — specific, observable, time-bounded. If you cannot complete it, the claim is not falsifiable as stated.\n\n**Step 4 — Plan to observe:** when is the observation possible / how measured / who tracks it / threshold for \"falsified.\"\n\n**Step 5 — Pre-commit to action:** if the falsifying observation occurs, what will you do? Is any theory modification itself falsifiable?\n\n**Step 6 — Iterate:** confirmed → keep monitoring; falsified → revise/abandon; unfalsifiable → recognize as belief.\n\n## Output Template\n\n```markdown\n# Falsifiability Analysis: <claim>\nClaim: | Asserted by: | Decision at stake: | Current basis:\nEmpirical: Y/N | Observation type:\nFalsified if: | Threshold: | Observable when/how: | Owner:\nIf falsified, action: | Ad-hoc preservation risk:\n```\n\n*→ Method in Action: [Popper 1934 + Eddington 1919 Eclipse + Modern Applications](examples/popper-1934-eddington-1919-eclipse-modern-applications.md)*\n\n## Pack: Vague → Falsifiable\n\n| Vague (unfalsifiable) | Falsifiable form |\n|---|---|\n| \"We have PMF\" | \"≥40% of users would be 'very disappointed' without the product\" |\n| \"Our outbound is working\" | \"5% of cold emails convert to qualified opps within 30 days\" |\n| \"Our culture is strong\" | \"Employee NPS ≥40 in next quarterly survey\" |\n| \"This stock is undervalued\" | \"Stock <12 P/E within 12 months; otherwise thesis is wrong\" |\n| \"Users want feature X\" | \"≥30% complete the new flow within 14 days of launch\" |\n\n## Applying It Well\n\n- Specify falsifiability conditions *before* the observation window opens — post-result conditions are rationalization.\n- More specific = more falsifiable. Quantify wherever possible.\n- When a prediction fails, ask: is the modification of your theory itself falsifiable?\n- Apply to others: \"What would change your mind?\" is the highest-leverage critical-thinking question.\n\n*→ Primary sources: [references/sources.md](references/sources.md)*\n\n## Common Rationalizations\n\n**[D] = designed upfront | [O] = observed in real use. [O] entries are more valuable.**\n\n| Fake move | Reality |\n|---|---|\n| [D] \"I just have a strong gut feeling\" | Gut feelings are not falsifiable. Without a test, you can't tell right from wrong. |\n| [D] \"We need more data to decide\" | If you can't specify what data would change your mind, you're not doing data-driven analysis. |\n| [D] \"The strategy just needs more time\" | Specify the timeframe and metrics in advance. |\n| [D] \"It's because of external factors\" | If external factors can always be invoked, the original claim was unfalsifiable. |\n| [D] \"It's too early to evaluate\" | If you can't specify when evaluation is appropriate, the claim is unfalsifiable. |\n| [D] \"We're in a special situation\" | This defense converts a falsifiable claim into an unfalsifiable one. Resist. |\n| *→ Add [O] entries here after each real use — paste the actual failure pattern* | *What went wrong and why* |\n\n## Red Flags\n\n- Claim asserted without specifying what would refute it\n- \"The strategy is working\" without metrics or thresholds\n- Predictions modified ad-hoc to accommodate failures\n- \"External factors\" invoked to explain disconfirming evidence\n- Person cannot specify what would change their mind\n\n## Verification\n\n- [ ] Claim specifically stated; empirical vs. non-empirical determined\n- [ ] Specific, observable, time-bounded falsification conditions defined\n- [ ] Monitoring mechanism in place with a named owner\n- [ ] Advance commitment to act on falsification made\n- [ ] Risk of ad-hoc preservation recognized\n\n---\n\n*Part of **deciqAI Knowledge Skills** — open-source thinking skills that make rigor executable for AI agents. Built by deciqAI · https://deciqai.com · Contributions welcome — see the template at the repo root.*\n\nFile v1.0.1:_meta.json\n\n{\n  \"ownerId\": \"kn754b8sk22s8c6gjxt02bftbn88q7ye\",\n  \"slug\": \"falsifiability\",\n  \"version\": \"1.0.1\",\n  \"publishedAt\": 1783456387815\n}\n\nFile v1.0.1:references/sources.md\n\n# Sources — falsifiability\n\n> *Primary sources for the [falsifiability](../SKILL.md) skill.*\n\n## Sources\n\n- Popper, K. R. (1934). *Logik der Forschung.* Vienna: Springer. English: *The Logic of Scientific Discovery* (1959). London: Hutchinson. ISBN 978-0415278447.\n- Popper, K. R. (1963). *Conjectures and Refutations: The Growth of Scientific Knowledge.* London: Routledge. ISBN 978-0415285940.\n- Eddington, A. S. (1919). \"Joint Eclipse Meeting of the Royal Society and the Royal Astronomical Society.\" *Nature*, 104, 354. The relativity confirmation.\n- Kuhn, T. S. (1962). *The Structure of Scientific Revolutions.* University of Chicago Press. ISBN 978-0226458120. The paradigm-shift complement.\n- Lakatos, I. (1970). \"Falsification and the Methodology of Scientific Research Programmes.\" in Lakatos & Musgrave (eds.), *Criticism and the Growth of Knowledge*. Cambridge University Press.\n- Ries, E. (2011). *The Lean Startup.* Crown Business. ISBN 978-0307887894. The startup application.\n- Tetlock, P. E. (2015). *Superforecasting.* Crown. ISBN 978-0804136693. Calibrated falsifiability in prediction.\n- Mayo, D. G. (1996). *Error and the Growth of Experimental Knowledge.* University of Chicago Press. The error-statistical framework.\n\nFile v1.0.1:examples/popper-1934-eddington-1919-eclipse-modern-applications.md\n\n# Method in Action: Popper 1934 + Eddington 1919 Eclipse + Modern Applications\n\n> *Example for the [falsifiability](../SKILL.md) skill.*\n\n**Karl Popper** (1902-1994) was an Austrian-British philosopher whose 1934 *Logik der Forschung* (translated as *The Logic of Scientific Discovery* in 1959) reshaped 20th-century philosophy of science. Popper had observed the rise of psychoanalysis (Freud, Adler) and Marxism in Vienna during the 1920s, and was struck by how their adherents claimed they \"explained\" everything — every observed event could be interpreted as confirming the theory. Popper's diagnosis: any theory that explains everything explains nothing. Such theories make no risky predictions; they cannot fail; therefore they are not empirical knowledge.\n\nPopper contrasted this with Einstein's general relativity (1915). Einstein had made a specific, quantitative prediction: light passing through the Sun's gravitational field during an eclipse would be deflected by 1.75 arcseconds, twice the Newtonian prediction. The 1919 total solar eclipse provided the empirical test. **Arthur Eddington's measurements during the eclipse** (in Príncipe and Brazil) confirmed the relativistic prediction within experimental error:\n\n> Eddington, A. S. (1919). \"Joint Eclipse Meeting of the Royal Society and the Royal Astronomical Society.\" *Nature*, 104, 354.\n\nPopper's emphasis: the test was real because relativity could have failed. If the measured deflection had been close to the Newtonian value or no deflection, relativity would have been falsified. The test mattered because failure was possible. The theory's confirmation was meaningful only because falsification was possible. This is the operational criterion of science.\n\nPopper's 1963 *Conjectures and Refutations* extended the framework to many domains:\n\n> \"The history of science, like the history of all human ideas, is a history of irresponsible dreams, of obstinacy, and of error. But science is one of the very few human activities — perhaps the only one — in which errors are systematically criticized and fairly often, in time, corrected. This is why we can say that, in science, we often learn from our mistakes, and why we can speak clearly and sensibly about making progress there.\"\n>\n> — Popper (1963), p. 216.\n\nThe framework has been applied extensively:\n\n**Lean Startup methodology.** Eric Ries's 2011 *The Lean Startup* is implicitly Popperian. Each iteration generates falsifiable hypotheses (\"if we A/B test this checkout flow change, conversion will improve by ≥5%\"), tests them empirically, and either confirms or rejects them based on data. The build-measure-learn cycle is operationalized falsifiability applied to product development.\n\n**Hypothesis-driven product development.** Modern product organizations (Amazon, Google, Microsoft) make explicit predictions about what user behavior change a feature will produce, then measure the actual change. Features that don't deliver the predicted impact are rolled back. The discipline is Popperian falsifiability at scale.\n\n**Investment thesis discipline.** Sophisticated investors specify what observations would falsify their investment thesis. If a thesis is \"Company X will reach $1B ARR by 2027,\" then quarterly milestones are set as falsifiability triggers. If Q3 ARR is below threshold X, the thesis is at risk; if it's below threshold Y, the thesis is falsified and the position is reconsidered.\n\n**Strategic planning.** Sophisticated strategy processes specify what conditions would change the strategy. \"Our differentiation rests on X capability; if competitors X' achieves parity in this capability, we need to revisit.\" The conditions are falsifiability triggers for the strategic thesis.\n\n**Scientific peer review and reproducibility.** The \"reproducibility crisis\" in social sciences (2010s onward) revealed that many published findings could not be replicated. Modern reforms (pre-registration of hypotheses, public datasets, replication studies) are explicit Popperian: making research findings falsifiable in practice, not just in principle.\n\n**Defense against unfalsifiable claims:** Several domains demonstrate the operational cost of unfalsifiable claims:\n\n- **Cult and conspiracy theory dynamics.** True believers find ways to \"explain\" any disconfirming evidence (the lack of UFO landing means the aliens are testing us; the lack of cult prediction fulfillment means we weren't ready). The structural feature is unfalsifiability — no observation could change their mind.\n- **Some macroeconomic claims.** \"Austerity will eventually produce growth\" or \"more stimulus will eventually produce inflation\" — without timeframes and thresholds, these claims are unfalsifiable. Any short-term failure can be attributed to \"not yet.\"\n- **Vague leadership pronouncements.** \"Our culture is strong\" or \"we have great execution\" without metrics or thresholds are unfalsifiable. They sound informative but contain no testable content.\n\nThree operational lessons:\n\n**First, the falsifiability question is the highest-leverage critical-thinking tool.** \"What would change your mind?\" — applied to any claim, your own or another's — produces dramatically better thinking than the alternative of accumulating confirmations.\n\n**Second, ad-hoc preservation is the failure mode.** When a prediction fails, the temptation is to modify the theory to accommodate the failure (\"the strategy worked, but X happened to prevent it\"). Each ad-hoc modification reduces the scientific status of the claim. Sometimes modifications are justified; usually they are face-saving rationalization that compounds error.\n\n**Third, specifying falsifiability conditions in advance is the discipline.** Conditions that you specify after the result are not falsifiability; they are post-hoc rationalization. The Popperian discipline is to commit to falsification triggers before the relevant observations are made — when you might genuinely be wrong, not after the fact.\n\nFile v1.0.1:skill-card.md\n\n## Description: <br>\nHelps agents turn empirical claims, strategies, experiments, and investment theses into testable falsification conditions and monitoring plans. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[deciqai](https://clawhub.ai/user/deciqai) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers, analysts, product teams, strategy teams, and decision makers use this skill to test whether an empirical claim can be refuted and to define observable thresholds before acting on it. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill may produce misleading analysis if a vague or non-empirical claim is forced into a testable frame. <br>\nMitigation: Confirm that the claim is empirical before applying the process, and stop when the claim is ethical, aesthetic, philosophical, or too costly to test. <br>\nRisk: Users may treat generated falsification thresholds as final decisions rather than reviewable reasoning aids. <br>\nMitigation: Review the claim, threshold, observation window, and proposed action before relying on the analysis. <br>\n\n\n## Reference(s): <br>\n- [Sources - falsifiability](references/sources.md) <br>\n- [Method in Action: Popper 1934 + Eddington 1919 Eclipse + Modern Applications](examples/popper-1934-eddington-1919-eclipse-modern-applications.md) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, guidance] <br>\n**Output Format:** [Markdown analysis template with concise questions and recommendations] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [May pause for user input in coach mode before continuing the analysis.] <br>\n\n## Skill Version(s): <br>\n1.0.1 (source: server release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.0.0: 5 files, 8690 bytes\n\nFiles: examples/popper-1934-eddington-1919-eclipse-modern-applications.md (5999b), references/sources.md (1242b), skill-card.md (2494b), SKILL.md (6787b), _meta.json (133b)\n\nFile v1.0.0:SKILL.md\n\n---\nname: falsifiability\ndescription: \"Activate when: user says 'what would prove this wrong', 'how do we know if our strategy is working', 'what would change your mind', 'this claim feels unfalsifiable', or is designing a hypothesis/experiment/investment thesis that needs to be made testable.\n  Do NOT activate when: the claim is genuinely non-empirical (ethical, aesthetic, philosophical); or the cost of running the test exceeds the value of the knowledge.\"\n---\n\n# Falsifiability\n\n## Overview\n\nA meaningful empirical claim must specify what observations would refute it. Claims that resist all possible refutation are not science — they are unfalsifiable belief. Formalized by Karl Popper (1934): science progresses not by accumulating confirmations but by surviving rigorous attempts at falsification. More-specific claims are more falsifiable; ad-hoc modifications that explain away failures destroy a claim's scientific status.\n\nComposes with [`confirmation-bias`](../confirmation-bias/SKILL.md) (falsifiability is the structural counter), [`abductive-reasoning`](../abductive-reasoning/SKILL.md) (generates hypotheses; this skill tests them), [`bayesian-reasoning`](../bayesian-reasoning/SKILL.md), [`critical-thinking`](../critical-thinking/SKILL.md).\n\n## When to Use\n\n- Designing OKRs, KPIs, or strategic goals; writing or evaluating investment theses\n- Designing experiments (A/B tests, product hypotheses, market entry)\n- Evaluating consultant/advisor recommendations or diagnosing vague leadership claims\n- Someone says \"what would change your mind,\" \"how would you know you're wrong,\" \"Popper\"\n\n**Not when:** genuinely non-empirical (philosophical, ethical, aesthetic); test cost exceeds its value.\n\n## Coaching Novices (Adaptive Front Door)\n\n- **Engine mode:** user has a specific claim → run The Process directly.\n- **Coach mode:** user is new → guide step by step.\n\nIn Coach mode, respond one step at a time. Each [WAIT] is a hard stop — output only that step's question, then stop.\n\n1. One-line: before acting on a claim, ask what observation would refute it — if \"nothing would,\" it's belief, not knowledge.\n2. Check fit: if the claim is genuinely metaphysical (ethical, aesthetic), falsifiability doesn't apply.\n3. Elicit the claim and its current evidential basis.\n> **[WAIT — do not advance until user responds]**\n4. Ask: what specific observation would falsify this? When and how would you observe it?\n> **[WAIT — do not advance until user responds]**\n5. Close: falsifiability conditions specified + monitoring plan + commitment to act on disconfirming evidence.\n> **[WAIT — do not advance until user responds]**\n\n## The Process\n\n**Step 1 — State the claim:** claim / who asserts it / decision dependent on it / current evidential basis.\n\n**Step 2 — Test whether empirical:** claim about how the world works (empirical) or values/aesthetics (non-empirical)? If non-empirical, stop here.\n\n**Step 3 — Specify falsification conditions:** complete \"This claim would be falsified if I observed: ___\" — specific, observable, time-bounded. If you cannot complete it, the claim is not falsifiable as stated.\n\n**Step 4 — Plan to observe:** when is the observation possible / how measured / who tracks it / threshold for \"falsified.\"\n\n**Step 5 — Pre-commit to action:** if the falsifying observation occurs, what will you do? Is any theory modification itself falsifiable?\n\n**Step 6 — Iterate:** confirmed → keep monitoring; falsified → revise/abandon; unfalsifiable → recognize as belief.\n\n## Output Template\n\n```markdown\n# Falsifiability Analysis: <claim>\nClaim: | Asserted by: | Decision at stake: | Current basis:\nEmpirical: Y/N | Observation type:\nFalsified if: | Threshold: | Observable when/how: | Owner:\nIf falsified, action: | Ad-hoc preservation risk:\n```\n\n*→ Method in Action: [Popper 1934 + Eddington 1919 Eclipse + Modern Applications](examples/popper-1934-eddington-1919-eclipse-modern-applications.md)*\n\n## Pack: Vague → Falsifiable\n\n| Vague (unfalsifiable) | Falsifiable form |\n|---|---|\n| \"We have PMF\" | \"≥40% of users would be 'very disappointed' without the product\" |\n| \"Our outbound is working\" | \"5% of cold emails convert to qualified opps within 30 days\" |\n| \"Our culture is strong\" | \"Employee NPS ≥40 in next quarterly survey\" |\n| \"This stock is undervalued\" | \"Stock <12 P/E within 12 months; otherwise thesis is wrong\" |\n| \"Users want feature X\" | \"≥30% complete the new flow within 14 days of launch\" |\n\n## Applying It Well\n\n- Specify falsifiability conditions *before* the observation window opens — post-result conditions are rationalization.\n- More specific = more falsifiable. Quantify wherever possible.\n- When a prediction fails, ask: is the modification of your theory itself falsifiable?\n- Apply to others: \"What would change your mind?\" is the highest-leverage critical-thinking question.\n\n*→ Primary sources: [references/sources.md](references/sources.md)*\n\n## Common Rationalizations\n\n**[D] = designed upfront | [O] = observed in real use. [O] entries are more valuable.**\n\n| Fake move | Reality |\n|---|---|\n| [D] \"I just have a strong gut feeling\" | Gut feelings are not falsifiable. Without a test, you can't tell right from wrong. |\n| [D] \"We need more data to decide\" | If you can't specify what data would change your mind, you're not doing data-driven analysis. |\n| [D] \"The strategy just needs more time\" | Specify the timeframe and metrics in advance. |\n| [D] \"It's because of external factors\" | If external factors can always be invoked, the original claim was unfalsifiable. |\n| [D] \"It's too early to evaluate\" | If you can't specify when evaluation is appropriate, the claim is unfalsifiable. |\n| [D] \"We're in a special situation\" | This defense converts a falsifiable claim into an unfalsifiable one. Resist. |\n| *→ Add [O] entries here after each real use — paste the actual failure pattern* | *What went wrong and why* |\n\n## Red Flags\n\n- Claim asserted without specifying what would refute it\n- \"The strategy is working\" without metrics or thresholds\n- Predictions modified ad-hoc to accommodate failures\n- \"External factors\" invoked to explain disconfirming evidence\n- Person cannot specify what would change their mind\n\n## Verification\n\n- [ ] Claim specifically stated; empirical vs. non-empirical determined\n- [ ] Specific, observable, time-bounded falsification conditions defined\n- [ ] Monitoring mechanism in place with a named owner\n- [ ] Advance commitment to act on falsification made\n- [ ] Risk of ad-hoc preservation recognized\n\n---\n\n*Part of **deciqAI Knowledge Skills** — open-source thinking skills that make rigor executable for AI agents. Built by deciqAI · https://deciqai.com · Contributions welcome — see the template at the repo root.*\n\nFile v1.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn754b8sk22s8c6gjxt02bftbn88q7ye\",\n  \"slug\": \"falsifiability\",\n  \"version\": \"1.0.0\",\n  \"publishedAt\": 1782631216280\n}\n\nFile v1.0.0:references/sources.md\n\n# Sources — falsifiability\n\n> *Primary sources for the [falsifiability](../SKILL.md) skill.*\n\n## Sources\n\n- Popper, K. R. (1934). *Logik der Forschung.* Vienna: Springer. English: *The Logic of Scientific Discovery* (1959). London: Hutchinson. ISBN 978-0415278447.\n- Popper, K. R. (1963). *Conjectures and Refutations: The Growth of Scientific Knowledge.* London: Routledge. ISBN 978-0415285940.\n- Eddington, A. S. (1919). \"Joint Eclipse Meeting of the Royal Society and the Royal Astronomical Society.\" *Nature*, 104, 354. The relativity confirmation.\n- Kuhn, T. S. (1962). *The Structure of Scientific Revolutions.* University of Chicago Press. ISBN 978-0226458120. The paradigm-shift complement.\n- Lakatos, I. (1970). \"Falsification and the Methodology of Scientific Research Programmes.\" in Lakatos & Musgrave (eds.), *Criticism and the Growth of Knowledge*. Cambridge University Press.\n- Ries, E. (2011). *The Lean Startup.* Crown Business. ISBN 978-0307887894. The startup application.\n- Tetlock, P. E. (2015). *Superforecasting.* Crown. ISBN 978-0804136693. Calibrated falsifiability in prediction.\n- Mayo, D. G. (1996). *Error and the Growth of Experimental Knowledge.* University of Chicago Press. The error-statistical framework.\n\nFile v1.0.0:examples/popper-1934-eddington-1919-eclipse-modern-applications.md\n\n# Method in Action: Popper 1934 + Eddington 1919 Eclipse + Modern Applications\n\n> *Example for the [falsifiability](../SKILL.md) skill.*\n\n**Karl Popper** (1902-1994) was an Austrian-British philosopher whose 1934 *Logik der Forschung* (translated as *The Logic of Scientific Discovery* in 1959) reshaped 20th-century philosophy of science. Popper had observed the rise of psychoanalysis (Freud, Adler) and Marxism in Vienna during the 1920s, and was struck by how their adherents claimed they \"explained\" everything — every observed event could be interpreted as confirming the theory. Popper's diagnosis: any theory that explains everything explains nothing. Such theories make no risky predictions; they cannot fail; therefore they are not empirical knowledge.\n\nPopper contrasted this with Einstein's general relativity (1915). Einstein had made a specific, quantitative prediction: light passing through the Sun's gravitational field during an eclipse would be deflected by 1.75 arcseconds, twice the Newtonian prediction. The 1919 total solar eclipse provided the empirical test. **Arthur Eddington's measurements during the eclipse** (in Príncipe and Brazil) confirmed the relativistic prediction within experimental error:\n\n> Eddington, A. S. (1919). \"Joint Eclipse Meeting of the Royal Society and the Royal Astronomical Society.\" *Nature*, 104, 354.\n\nPopper's emphasis: the test was real because relativity could have failed. If the measured deflection had been close to the Newtonian value or no deflection, relativity would have been falsified. The test mattered because failure was possible. The theory's confirmation was meaningful only because falsification was possible. This is the operational criterion of science.\n\nPopper's 1963 *Conjectures and Refutations* extended the framework to many domains:\n\n> \"The history of science, like the history of all human ideas, is a history of irresponsible dreams, of obstinacy, and of error. But science is one of the very few human activities — perhaps the only one — in which errors are systematically criticized and fairly often, in time, corrected. This is why we can say that, in science, we often learn from our mistakes, and why we can speak clearly and sensibly about making progress there.\"\n>\n> — Popper (1963), p. 216.\n\nThe framework has been applied extensively:\n\n**Lean Startup methodology.** Eric Ries's 2011 *The Lean Startup* is implicitly Popperian. Each iteration generates falsifiable hypotheses (\"if we A/B test this checkout flow change, conversion will improve by ≥5%\"), tests them empirically, and either confirms or rejects them based on data. The build-measure-learn cycle is operationalized falsifiability applied to product development.\n\n**Hypothesis-driven product development.** Modern product organizations (Amazon, Google, Microsoft) make explicit predictions about what user behavior change a feature will produce, then measure the actual change. Features that don't deliver the predicted impact are rolled back. The discipline is Popperian falsifiability at scale.\n\n**Investment thesis discipline.** Sophisticated investors specify what observations would falsify their investment thesis. If a thesis is \"Company X will reach $1B ARR by 2027,\" then quarterly milestones are set as falsifiability triggers. If Q3 ARR is below threshold X, the thesis is at risk; if it's below threshold Y, the thesis is falsified and the position is reconsidered.\n\n**Strategic planning.** Sophisticated strategy processes specify what conditions would change the strategy. \"Our differentiation rests on X capability; if competitors X' achieves parity in this capability, we need to revisit.\" The conditions are falsifiability triggers for the strategic thesis.\n\n**Scientific peer review and reproducibility.** The \"reproducibility crisis\" in social sciences (2010s onward) revealed that many published findings could not be replicated. Modern reforms (pre-registration of hypotheses, public datasets, replication studies) are explicit Popperian: making research findings falsifiable in practice, not just in principle.\n\n**Defense against unfalsifiable claims:** Several domains demonstrate the operational cost of unfalsifiable claims:\n\n- **Cult and conspiracy theory dynamics.** True believers find ways to \"explain\" any disconfirming evidence (the lack of UFO landing means the aliens are testing us; the lack of cult prediction fulfillment means we weren't ready). The structural feature is unfalsifiability — no observation could change their mind.\n- **Some macroeconomic claims.** \"Austerity will eventually produce growth\" or \"more stimulus will eventually produce inflation\" — without timeframes and thresholds, these claims are unfalsifiable. Any short-term failure can be attributed to \"not yet.\"\n- **Vague leadership pronouncements.** \"Our culture is strong\" or \"we have great execution\" without metrics or thresholds are unfalsifiable. They sound informative but contain no testable content.\n\nThree operational lessons:\n\n**First, the falsifiability question is the highest-leverage critical-thinking tool.** \"What would change your mind?\" — applied to any claim, your own or another's — produces dramatically better thinking than the alternative of accumulating confirmations.\n\n**Second, ad-hoc preservation is the failure mode.** When a prediction fails, the temptation is to modify the theory to accommodate the failure (\"the strategy worked, but X happened to prevent it\"). Each ad-hoc modification reduces the scientific status of the claim. Sometimes modifications are justified; usually they are face-saving rationalization that compounds error.\n\n**Third, specifying falsifiability conditions in advance is the discipline.** Conditions that you specify after the result are not falsifiability; they are post-hoc rationalization. The Popperian discipline is to commit to falsification triggers before the relevant observations are made — when you might genuinely be wrong, not after the fact.\n\nFile v1.0.0:skill-card.md\n\n## Description: <br>\nHelps agents turn empirical claims, strategies, experiments, and investment theses into testable statements with explicit falsification conditions. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[deciqai](https://clawhub.ai/user/deciqai) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nExternal users, developers, and analysts use this skill to evaluate empirical claims, design experiments, define measurable thresholds, and pre-commit to actions when evidence falsifies a claim. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill may over-apply empirical testing to ethical, aesthetic, philosophical, or otherwise non-empirical claims. <br>\nMitigation: First classify whether the claim is empirical and stop or redirect when falsifiability does not apply. <br>\nRisk: The skill may encourage extra questioning or test design when the cost of running a test exceeds the value of the knowledge. <br>\nMitigation: Check test value before proceeding and avoid recommending measurement work that is not worth the decision impact. <br>\nRisk: Generated thresholds or action plans could become misleading if accepted without domain review. <br>\nMitigation: Treat outputs as structured guidance and have the user or domain owner confirm metrics, time windows, and actions before relying on them. <br>\n\n\n## Reference(s): <br>\n- [Sources - falsifiability](artifact/references/sources.md) <br>\n- [Method in Action: Popper 1934 + Eddington 1919 Eclipse + Modern Applications](artifact/examples/popper-1934-eddington-1919-eclipse-modern-applications.md) <br>\n- [ClawHub Skill Page](https://clawhub.ai/deciqai/skills/falsifiability) <br>\n- [Publisher Website](https://deciqai.com) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, guidance] <br>\n**Output Format:** [Markdown] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Structured falsifiability analysis with claim, evidence basis, falsification threshold, observation plan, owner, action, and ad-hoc preservation risk.] <br>\n\n## Skill Version(s): <br>\n1.0.0 (source: release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>","readmeExcerpt":"Skill: Falsifiability Owner: deciqai Summary: Activate when: user says 'what would prove this wrong', 'how do we know if our strategy is working', 'what would change your mind', 'this claim feels unfalsi... Tags: latest:1.0.5 Version history: v1.0.5 | 2026-07-16T17:59:34.650Z | user Description tail link + agents machine-readable metadata line (deciqai.com/s/falsifiability.json) v1.0.4 | 2026-07-10T10:25:40.288Z | us","codeSnippets":[],"executableExamples":[{"language":"markdown","snippet":"# Falsifiability Analysis: <claim>\nClaim: | Asserted by: | Decision at stake: | Current basis:\nEmpirical: Y/N | Observation type:\nFalsified if: | Threshold: | Observable when/how: | Owner:\nIf falsified, action: | Ad-hoc preservation risk:"},{"language":"markdown","snippet":"# Falsifiability Analysis: <claim>\nClaim: | Asserted by: | Decision at stake: | Current basis:\nEmpirical: Y/N | Observation type:\nFalsified if: | Threshold: | Observable when/how: | Owner:\nIf falsified, action: | Ad-hoc preservation risk:"},{"language":"markdown","snippet":"# Falsifiability Analysis: <claim>\nClaim: | Asserted by: | Decision at stake: | Current basis:\nEmpirical: Y/N | Observation type:\nFalsified if: | Threshold: | Observable when/how: | Owner:\nIf falsified, action: | Ad-hoc preservation risk:"},{"language":"markdown","snippet":"# Falsifiability Analysis: <claim>\nClaim: | Asserted by: | Decision at stake: | Current basis:\nEmpirical: Y/N | Observation type:\nFalsified if: | Threshold: | Observable when/how: | Owner:\nIf falsified, action: | Ad-hoc preservation risk:"},{"language":"markdown","snippet":"# Falsifiability Analysis: <claim>\nClaim: | Asserted by: | Decision at stake: | Current basis:\nEmpirical: Y/N | Observation type:\nFalsified if: | Threshold: | Observable when/how: | Owner:\nIf falsified, action: | Ad-hoc preservation risk:"},{"language":"markdown","snippet":"# Falsifiability Analysis: <claim>\nClaim: | Asserted by: | Decision at stake: | Current basis:\nEmpirical: Y/N | Observation type:\nFalsified if: | Threshold: | Observable when/how: | Owner:\nIf falsified, action: | Ad-hoc preservation risk:"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: falsifiability\ndescription: \"Activate when: user says 'what would prove this wrong', 'how do we know if our strategy is working', 'what would change your mind', 'this claim feels unfalsifiable', or is designing a hypothesis/experiment/investment thesis that needs to be made testable.\n  Do NOT activate when: the claim is genuinely non-empirical (ethical, aesthetic, philosophical); or the cost of running the test exceeds the value of the knowledge. More: deciqai.com/c/falsifiability\"\n---\n\n# Falsifiability\n\n## Overview\n\nA meaningful empirical claim must specify what observations would refute it. Claims that resist all possible refutation are not science — they are unfalsifiable belief. Formalized by Karl Popper (1934): science progresses not by accumulating confirmations but by surviving rigorous attempts at falsification. More-specific claims are more falsifiable; ad-hoc modifications that explain away failures destroy a claim's scientific status.\n\nComposes with `confirmation-bias` (falsifiability is the structural counter), `abductive-reasoning` (generates hypotheses; this skill tests them), `bayesian-reasoning`, `critical-thinking`.\n\n## When to Use\n\n- Designing OKRs, KPIs, or strategic goals; writing or evaluating investment theses\n- Designing experiments (A/B tests, product hypotheses, market entry)\n- Evaluating consultant/advisor recommendations or diagnosing vague leadership claims\n- Someone says \"what would change your mind,\" \"how would you know you're wrong,\" \"Popper\"\n- Stress-testing an AI/AGI hype claim (\"AGI is near,\" \"the model truly understands,\" \"our AI adoption is working\") — demand what evidence would disprove the capability or safety claim\n\n**Not when:** genuinely non-empirical (philosophical, ethical, aesthetic); test cost exceeds its value.\n\n## Coaching Novices (Adaptive Front Door)\n\n- **Engine mode:** user has a specific claim → run The Process directly.\n- **Coach mode:** user is new → guide step by step.\n\nIn Coach mode, respond one step at a time. Each [WAIT] is a hard stop — output only that step's question, then stop.\n\n1. One-line: before acting on a claim, ask what observation would refute it — if \"nothing would,\" it's belief, not knowledge.\n2. Check fit: if the claim is genuinely metaphysical (ethical, aesthetic), falsifiability doesn't apply.\n3. Elicit the claim and its current evidential basis.\n> **[WAIT — do not advance until user responds]**\n4. Ask: what specific observation would falsify this? When and how would you observe it?\n> **[WAIT — do not advance until user responds]**\n5. Close: falsifiability conditions specified + monitoring plan + commitment to act on disconfirming evidence.\n> **[WAIT — do not advance until user responds]**\n\n## The Process\n\n**Step 1 — State the claim:** claim / who asserts it / decision dependent on it / current evidential basis.\n\n**Step 2 — Test whether empirical:** claim about how the world works (empirical) or values/aesthetics (non-empirical)? If non-empirical, stop here.\n\n**Ste"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn754b8sk22s8c6gjxt02bftbn88q7ye\",\n  \"slug\": \"falsifiability\",\n  \"version\": \"1.0.5\",\n  \"publishedAt\": 1784224774650\n}"},{"path":"references/sources.md","content":"# Sources — falsifiability\n\n> *Primary sources for the [falsifiability](../SKILL.md) skill.*\n\n## Sources\n\n- Popper, K. R. (1934). *Logik der Forschung.* Vienna: Springer. English: *The Logic of Scientific Discovery* (1959). London: Hutchinson. ISBN 978-0415278447.\n- Popper, K. R. (1963). *Conjectures and Refutations: The Growth of Scientific Knowledge.* London: Routledge. ISBN 978-0415285940.\n- Eddington, A. S. (1919). \"Joint Eclipse Meeting of the Royal Society and the Royal Astronomical Society.\" *Nature*, 104, 354. The relativity confirmation.\n- Kuhn, T. S. (1962). *The Structure of Scientific Revolutions.* University of Chicago Press. ISBN 978-0226458120. The paradigm-shift complement.\n- Lakatos, I. (1970). \"Falsification and the Methodology of Scientific Research Programmes.\" in Lakatos & Musgrave (eds.), *Criticism and the Growth of Knowledge*. Cambridge University Press.\n- Ries, E. (2011). *The Lean Startup.* Crown Business. ISBN 978-0307887894. The startup application.\n- Tetlock, P. E. (2015). *Superforecasting.* Crown. ISBN 978-0804136693. Calibrated falsifiability in prediction.\n- Mayo, D. G. (1996). *Error and the Growth of Experimental Knowledge.* University of Chicago Press. The error-statistical framework.\n- Chollet, F. (2019). \"On the Measure of Intelligence.\" arXiv:1911.01547. Operationalizes \"intelligence\" into the falsifiable ARC / ARC-AGI benchmark — a template for turning \"AGI is near\" into a testable claim (2024–2026 AI application).\n- On benchmark contamination and reasoning robustness (2023–2025): peer-reviewed and arXiv work documenting train/test data contamination in LLM evaluations and accuracy drops when surface features of reasoning problems are perturbed — why a high AI benchmark score does not, by itself, confirm a capability claim. (Consult current surveys; specific reported scores are vendor figures, not independently audited.)"},{"path":"examples/ai-claims-agi-is-near-vs-testable-predictions-2024-2026.md","content":"# Method in Action: Separating Falsifiable from Unfalsifiable AI Claims (2024–2026)\n\n> *Example for the [falsifiability](../SKILL.md) skill.*\n\nBetween 2024 and 2026, public discourse around large AI models filled with two very different kinds of statement. Some were slogans — \"AGI is near,\" \"the model truly understands,\" \"scaling will just keep working\" — and some were concrete, dated predictions with numbers attached. Popper's criterion sorts them cleanly: a claim is empirical knowledge only if you can say in advance what observation would prove it wrong. This walkthrough runs the anchor claims through the falsifiability skill's own six-step Process.\n\n## Step 1 — State the claim\n\nTake three representative claims from the period:\n\n- **Claim A (slogan):** \"AGI is near.\"\n- **Claim B (mentalistic):** \"The model *truly understands* what it's saying.\"\n- **Claim C (testable):** \"By the end of 2025, a frontier model will exceed a specified accuracy threshold on the competition-mathematics benchmark AIME,\" or similar dated, metric-bound predictions of the kind researchers publish.\n\n*Who asserts them:* executives, commentators, and researchers, respectively. *Decision at stake:* whether an enterprise should bet a roadmap (or an investor a position) on imminent general capability. *Current evidential basis:* rapid, genuine benchmark gains from roughly 2023 onward, plus extrapolation.\n\n## Step 2 — Test whether empirical\n\n- **Claim A (\"AGI is near\")** is empirical *only if* \"AGI\" and \"near\" are defined. As typically used, neither is: \"AGI\" has no agreed operational definition, and \"near\" has no date. Without those, it is not yet a testable claim — it is a mood.\n- **Claim B (\"truly understands\")** invokes an inner mental state. As stated it is closer to metaphysics than to empirical science: no external observation is specified that would distinguish \"truly understands\" from \"produces the same outputs without understanding.\" Step 2 says: if it stays non-empirical, stop and label it belief.\n- **Claim C** is empirical: it is a statement about a measurable score on a fixed benchmark by a fixed date.\n\n## Step 3 — Specify falsification conditions\n\nForce each claim to complete: *\"This would be falsified if I observed: ___.\"*\n\n- **Claim A** can be *rescued into* an empirical claim by pinning it down — e.g. \"A single model will pass [a specified operationalization, such as the ARC-AGI abstraction-and-reasoning benchmark at human-level, or a stated economically-valuable-task bar] before 31 December 2026.\" Now it can fail. Note the discipline: the version that can be proven wrong is the only version worth arguing about.\n- **Claim B** resists completion. \"Understanding\" that predicts no observable difference from \"not understanding\" has no falsification condition. The productive move is to *replace* it with a behavioral proxy that does — e.g. \"the model will maintain accuracy when the same problem is presented with surface features (names, numbers, framing) changed,\" wh"},{"path":"examples/popper-1934-eddington-1919-eclipse-modern-applications.md","content":"# Method in Action: Popper 1934 + Eddington 1919 Eclipse + Modern Applications\n\n> *Example for the [falsifiability](../SKILL.md) skill.*\n\n**Karl Popper** (1902-1994) was an Austrian-British philosopher whose 1934 *Logik der Forschung* (translated as *The Logic of Scientific Discovery* in 1959) reshaped 20th-century philosophy of science. Popper had observed the rise of psychoanalysis (Freud, Adler) and Marxism in Vienna during the 1920s, and was struck by how their adherents claimed they \"explained\" everything — every observed event could be interpreted as confirming the theory. Popper's diagnosis: any theory that explains everything explains nothing. Such theories make no risky predictions; they cannot fail; therefore they are not empirical knowledge.\n\nPopper contrasted this with Einstein's general relativity (1915). Einstein had made a specific, quantitative prediction: light passing through the Sun's gravitational field during an eclipse would be deflected by 1.75 arcseconds, twice the Newtonian prediction. The 1919 total solar eclipse provided the empirical test. **Arthur Eddington's measurements during the eclipse** (in Príncipe and Brazil) confirmed the relativistic prediction within experimental error:\n\n> Eddington, A. S. (1919). \"Joint Eclipse Meeting of the Royal Society and the Royal Astronomical Society.\" *Nature*, 104, 354.\n\nPopper's emphasis: the test was real because relativity could have failed. If the measured deflection had been close to the Newtonian value or no deflection, relativity would have been falsified. The test mattered because failure was possible. The theory's confirmation was meaningful only because falsification was possible. This is the operational criterion of science.\n\nPopper's 1963 *Conjectures and Refutations* extended the framework to many domains:\n\n> \"The history of science, like the history of all human ideas, is a history of irresponsible dreams, of obstinacy, and of error. But science is one of the very few human activities — perhaps the only one — in which errors are systematically criticized and fairly often, in time, corrected. This is why we can say that, in science, we often learn from our mistakes, and why we can speak clearly and sensibly about making progress there.\"\n>\n> — Popper (1963), p. 216.\n\nThe framework has been applied extensively:\n\n**Lean Startup methodology.** Eric Ries's 2011 *The Lean Startup* is implicitly Popperian. Each iteration generates falsifiable hypotheses (\"if we A/B test this checkout flow change, conversion will improve by ≥5%\"), tests them empirically, and either confirms or rejects them based on data. The build-measure-learn cycle is operationalized falsifiability applied to product development.\n\n**Hypothesis-driven product development.** Modern product organizations (Amazon, Google, Microsoft) make explicit predictions about what user behavior change a feature will produce, then measure the actual change. Features that don't deliver the predicted impact are rolled back. T"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Activate when: user says 'what would prove this wrong', 'how do we know if our strategy is working', 'what would change your mind', 'this claim feels unfalsi... Skill: Falsifiability Owner: deciqai Summary: Activate when: user says 'what would prove this wrong', 'how do we know if our strategy is working', 'what would change your mind', 'this claim feels unfalsi... Tags: latest:1.0.5 Version history: v1.0.5 | 2026-07-16T17:59:34.650Z | user Description tail link + agents machine-readable metadata line (deciqai.com/s/falsifiability.json) v1.0.4 | 2026-07-10T10:25:40.288Z | us","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1931,"uniquenessScore":50,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T11:19:57.408Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T11:19:57.408Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T14:15:08.416Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}