{"id":"4327a640-37cd-4127-ae47-e587b837b7ef","entityType":"agent","slug":"clawhub-deciqai-goodharts-law","name":"Goodhart's Law","canonicalUrl":"https://www.xpersona.co/agent/clawhub-deciqai-goodharts-law","canonicalPath":"/agent/clawhub-deciqai-goodharts-law","generatedAt":"2026-10-11T03:56:43.688Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-11T00:08:25.475Z","emptyReason":null},"description":"Activate when: our KPI is going up but the real outcome isn't improving; people seem to be gaming the metric; we're about to tie bonuses or promotions to a n... Skill: Goodhart's Law Owner: deciqai Summary: Activate when: our KPI is going up but the real outcome isn't improving; people seem to be gaming the metric; we're about to tie bonuses or promotions to a n... Tags: latest:1.0.5 Version history: v1.0.5 | 2026-07-16T18:01:32.110Z | user Description tail link + agents machine-readable metadata line (deciqai.com/s/goodharts-law.json) v1.0.4 | 2026-07-09T11:17:54.034Z | use","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.2K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s17a4mqcnk515kvaca5ze55d0x88pfpx:goodharts-law","sourceUrl":"https://clawhub.ai/deciqai/goodharts-law","homepage":"https://clawhub.ai/deciqai/skills/goodharts-law","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/deciqai/goodharts-law","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/deciqai/skills/goodharts-law","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":62,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Activate when: our KPI is going up but the real outcome isn't improving; people seem to be gaming the metric; we're about to tie bonuses or promotions to a n..."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T00:08:25.475Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T00:08:25.475Z","emptyReason":null},"stars":null,"forks":null,"downloads":1223,"packageName":null,"latestVersion":"1.0.5","tractionLabel":"1.2K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T00:08:25.406Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T00:08:25.475Z","lastCrawledAt":"2026-10-11T00:08:25.406Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T00:08:25.406Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.5","createdAt":"2026-07-16T18:01:32.110Z","changelog":"Description tail link + agents machine-readable metadata line (deciqai.com/s/goodharts-law.json)","fileCount":6,"zipByteSize":16186},{"version":"1.0.4","createdAt":"2026-07-09T11:17:54.034Z","changelog":"Refresh: 2024-2026 AI-era worked examples added (strategy/leadership + systems/game-theory batch)","fileCount":6,"zipByteSize":16063},{"version":"1.0.3","createdAt":"2026-07-08T11:04:01.897Z","changelog":"Footer now uses /c/<slug> short link (fixes UTM truncation when SKILL.md is read in a terminal)","fileCount":5,"zipByteSize":11189},{"version":"1.0.2","createdAt":"2026-07-08T00:49:19.378Z","changelog":"Refreshed content + GitHub star link in footer","fileCount":5,"zipByteSize":11088},{"version":"1.0.1","createdAt":"2026-07-07T20:33:16.613Z","changelog":"Add catalog categories and topics","fileCount":5,"zipByteSize":11171},{"version":"1.0.0","createdAt":"2026-06-28T11:17:28.608Z","changelog":"Initial publish","fileCount":5,"zipByteSize":11162}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17a4mqcnk515kvaca5ze55d0x88pfpx:goodharts-law","setupComplexity":"low","setupSteps":["Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-deciqai-goodharts-law/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-deciqai-goodharts-law/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-deciqai-goodharts-law/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-deciqai-goodharts-law/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-deciqai-goodharts-law/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-deciqai-goodharts-law/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T03:56:43.684Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-deciqai-goodharts-law/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-deciqai-goodharts-law/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-deciqai-goodharts-law/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-deciqai-goodharts-law/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-11T00:08:25.475Z","emptyReason":null},"readme":"Skill: Goodhart's Law\n\nOwner: deciqai\n\nSummary: Activate when: our KPI is going up but the real outcome isn't improving; people seem to be gaming the metric; we're about to tie bonuses or promotions to a n...\n\nTags: latest:1.0.5\n\nVersion history:\n\nv1.0.5 | 2026-07-16T18:01:32.110Z | user\n\nDescription tail link + agents machine-readable metadata line (deciqai.com/s/goodharts-law.json)\n\nv1.0.4 | 2026-07-09T11:17:54.034Z | user\n\nRefresh: 2024-2026 AI-era worked examples added (strategy/leadership + systems/game-theory batch)\n\nv1.0.3 | 2026-07-08T11:04:01.897Z | user\n\nFooter now uses /c/<slug> short link (fixes UTM truncation when SKILL.md is read in a terminal)\n\nv1.0.2 | 2026-07-08T00:49:19.378Z | user\n\nRefreshed content + GitHub star link in footer\n\nv1.0.1 | 2026-07-07T20:33:16.613Z | user\n\nAdd catalog categories and topics\n\nv1.0.0 | 2026-06-28T11:17:28.608Z | user\n\nInitial publish\n\nArchive index:\n\nArchive v1.0.5: 6 files, 16186 bytes\n\nFiles: examples/ai-benchmark-and-engagement-gaming-2023-2026.md (9141b), examples/goodhart-1975-m3-and-strathern-1997-rae.md (10675b), references/sources.md (2000b), skill-card.md (2572b), SKILL.md (7862b), _meta.json (132b)\n\nFile v1.0.5:SKILL.md\n\n---\nname: goodharts-law\ndescription: \"Activate when: our KPI is going up but the real outcome isn't improving; people seem to be gaming the metric; we're about to tie bonuses or promotions to a number; an algorithm is producing results nobody intended; a test or audit system is being designed.\n  Do NOT activate when: the metric IS the goal with no proxy gap; measurement is purely descriptive with zero stakes attached. More: deciqai.com/c/goodharts-law\"\n---\n\n# Goodhart's Law\n\n## Overview\n\n**Goodhart's Law:** when a metric controls behavior, people optimize the metric rather than the underlying goal. Formulated by economist Charles Goodhart (1975) on UK monetary policy; sharpened by Marilyn Strathern (1997): *\"When a measure becomes a target, it ceases to be a good measure.\"* Four failure mechanisms (Manheim & Garrabrant 2018): **Regressional**, **Extremal**, **Causal**, **Adversarial**. Countermeasure is always multi-metric + audit + rotation.\n\nComposes with `feedback-loops`, `principal-agent`, `okr-goal-setting`, `survivorship-bias`.\n\n## When to Use\n\n- A KPI is being introduced or its weight is increasing in performance evaluation\n- A metric is \"improving\" without corresponding improvement in the underlying goal\n- People are visibly optimizing for a number rather than the work it was meant to track\n- Algorithmic optimization is producing outcomes the designers didn't intend\n- Resource allocation is driven by a single composite score or ranking\n- An AI model, benchmark, or engagement metric is being optimized (or used to justify AI capex / adoption / AI-native competition) and the score is rising faster than real capability or user value\n\n**Not when:** metric and goal are identical; stakes too low for gaming; metric is purely descriptive with no reward/punishment; question is which metric to use, not whether the measurement-reward system is sound.\n\n## Coaching Novices (Adaptive Front Door)\n\n- **Engine mode:** user has a concrete metric or system → run The Process directly.\n- **Coach mode:** user is unfamiliar or has no concrete case → guide step by step.\n\nIn Coach mode, respond one step at a time. Each [WAIT] is a hard stop — output only that step's question, then stop.\n\n1. One-line: before relying on a metric to control behavior, predict how people will game it — choose the system that survives that prediction.\n2. Check fit: if the metric is purely descriptive (no reward attached), Goodhart's law doesn't apply yet.\n3. Elicit the specific metric and the underlying goal: what's being measured? What's the actual outcome you care about?\n> **[WAIT — do not advance until user responds]**\n4. One question at a time: proxy gap? How would a clever agent game this? Which Goodhart category? What countermeasure fits?\n> **[WAIT — do not advance until user responds]**\n5. Close: name the gaming-resistant design (multi-metric, audit, rotation, paired-constraint) + monitoring schedule.\n> **[WAIT — do not advance until user responds]**\n\n## The Process\n\n**Step 1 — State metric and goal:** metric being targeted / underlying goal / current proxy-goal correlation / who is measured / stakes.\n\n**Step 2 — Predict the gaming:** list ≥3 ways to game the metric with minimum effort on the goal. If you can't list 3, you haven't thought hard enough.\n\n**Step 3 — Categorize mechanism:**\n\n| Mechanism | Test |\n|---|---|\n| Regressional | Is there noise that optimization will push into? |\n| Extremal | Does metric-goal correlation break at extremes? |\n| Causal | Is the metric a symptom, not a cause? |\n| Adversarial | Will agents actively game with intelligence? |\n\n**Step 4 — Choose countermeasure:** Regressional → constrain range. Extremal → paired constraint metrics. Causal → closer-to-causation metric + direct audit. Adversarial → multi-metric + randomized audits + rotation.\n**Step 5 — Design the system:** primary metric / constraint metric(s) / audit mechanism (sampled direct goal observation) / rotation schedule / separation of measure-for-control from measure-for-diagnosis / gaming-detection threshold.\n**Step 6 — Schedule re-evaluation:** independent goal measurement (how/when/who) / drift threshold / retirement criteria / owner.\n\n## Output template\n\n```markdown\n# Goodhart-Robust Design: <metric>\nMetric: | Underlying goal: | Correlation: | Who measured: | Stakes:\nGaming vectors (≥3):\nMechanism: Regressional / Extremal / Causal / Adversarial\nPrimary metric: | Constraint metric(s): | Audit: | Rotation: | Separation: | Gaming threshold:\nGoal measurement (independent): | Drift threshold: | Retirement criteria: | Owner:\n```\n\n*→ Method in Action: [Goodhart 1975 (M3) and Strathern 1997 (RAE)](examples/goodhart-1975-m3-and-strathern-1997-rae.md)*\n*→ 2026 lens: [AI benchmarks and engagement metrics as targets (2023–2026)](examples/ai-benchmark-and-engagement-gaming-2023-2026.md)*\n\n## Pack: Goodhart's Law Patterns\n\n| Domain | Common gaming | Defense |\n|---|---|---|\n| Sales quotas | Sandbagging, channel stuffing, end-of-quarter discounts | Multi-period averaging; quality metrics; clawback |\n| Hospital wait targets | Ambulance parking, patient reclassification | Outcome audits; paired metrics; randomized inspection |\n| Standardized testing | Teaching to test, curriculum narrowing | Sample-based assessment; multi-measure; reduce single-test stakes |\n| Algorithmic engagement | Clickbait, outrage, misinformation | Multi-objective optimization; quality + harm constraints |\n\n## Applying It Well\n\n- All metrics are proxies — narrower than the goal. Pre-commit to gap analysis before deployment.\n- Gaming is rational under measurement pressure. Fix the system, not the people.\n- Rotation and audit are the only durable defenses. Plan metric retirement at design time.\n\n*→ Primary sources: [references/sources.md](references/sources.md)*\n\n## Common Rationalizations\n\n| Fake move | Reality |\n|---|---|\n| [D] \"If you can't measure it, you can't manage it\" | Often false. Judgment, trust, and direct observation are also valid management tools. |\n| [D] \"Our metric is well-defined; it won't be gamed\" | Precision invites precise gaming. Basel II capital ratios were well-defined — extensively gamed. |\n| [D] \"Our people wouldn't game the metric\" | Goodhart's law is structural; individual virtue is insufficient in aggregate. |\n| [D] \"We just need a better metric\" | Often the issue is any single metric under pressure; fix is multi-metric + audit. |\n| [D] \"We've used this metric for years\" | Long use = more time for gaming to mature. Tenure is a warning, not an endorsement. |\n| *→ Add [O] entries here after each real use — paste the actual failure pattern* | *What went wrong and why* |\n\n**[D] = designed upfront | [O] = observed in real use. [O] entries are more valuable.**\n\n## Red Flags\n\n- Metric tied to high-stakes rewards or punishments\n- Metric \"improves\" without obvious improvement in the underlying goal\n- People being measured can already articulate ways to game it\n- Single metric is the primary evaluation tool, no audit or paired-constraint\n\n## Verification\n\n- [ ] Goal underlying the metric specifically named\n- [ ] Proxy gap explicitly described; ≥3 gaming vectors listed\n- [ ] Goodhart mechanism category identified\n- [ ] Countermeasure design (multi-metric, audit, rotation) in place\n- [ ] Independent goal measurement scheduled with owner and retirement date\n\n---\n\n*Part of **deciqAI Knowledge Skills** — 227 open-source thinking skills that make rigor executable for AI agents. The same skills power every deciqAI agent, which runs them autonomously to operate your company. **See it run → https://www.deciqai.com/c/goodharts-law** · ⭐ Star the repo → https://github.com/deciqAI/knowledge-skills · Contributions welcome.*\n\n*Agents: latest version & machine-readable metadata → https://www.deciqai.com/s/goodharts-law.json*\n\nFile v1.0.5:_meta.json\n\n{\n  \"ownerId\": \"kn754b8sk22s8c6gjxt02bftbn88q7ye\",\n  \"slug\": \"goodharts-law\",\n  \"version\": \"1.0.5\",\n  \"publishedAt\": 1784224892110\n}\n\nFile v1.0.5:references/sources.md\n\n# Sources — goodharts-law\n\n> *Primary sources for the [goodharts-law](../SKILL.md) skill.*\n\n- Goodhart, C. A. E. (1975). \"Problems of monetary management: The U.K. experience.\" Papers in Monetary Economics, Reserve Bank of Australia. Reprinted in *Monetary Theory and Practice: The U.K. Experience* (1984), Macmillan. The original.\n- Strathern, M. (1997). \"'Improving ratings': Audit in the British university system.\" *European Review*, 5(3), 305-321. The modern aphoristic formulation.\n- Campbell, D. T. (1979). \"Assessing the impact of planned social change.\" *Evaluation and Program Planning*, 2(1), 67-90. Independent formulation (\"Campbell's Law\").\n- Lucas, R. E. (1976). \"Econometric policy evaluation: A critique.\" *Carnegie-Rochester Conference Series on Public Policy*, 1(1), 19-46. The adjacent macro-econometric \"Lucas critique.\"\n- Manheim, D., & Garrabrant, S. (2018). \"Categorizing variants of Goodhart's Law.\" *arXiv:1803.04585*. The four-mechanism taxonomy.\n- Muller, J. Z. (2018). *The Tyranny of Metrics.* Princeton University Press. ISBN 978-0691174952. Comprehensive case-study survey.\n- Doerr, J. (2018). *Measure What Matters: How Google, Bono, and the Gates Foundation Rock the World with OKRs.* Portfolio. ISBN 978-0525536222.\n- Wells Fargo / Office of the Comptroller of the Currency (2016, 2018). Various enforcement actions and consent orders documenting the cross-selling case.\n- Zhou, K., et al. (2023). \"Don't Make Your LLM an Evaluation Benchmark Cheater.\" *arXiv:2311.01964*. Documents benchmark data contamination and how public evaluation scores decay as a capability signal once test data leaks into training — a direct 2020s AI instance of Goodhart's law.\n- Chiang, W.-L., et al. (2024). \"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference.\" *arXiv:2403.04132*. Motivates live, blind, human-preference evaluation as a harder-to-game complement to static leaderboards; illustrates both the countermeasure and its own residual gaming risks.\n\nFile v1.0.5:examples/ai-benchmark-and-engagement-gaming-2023-2026.md\n\n# Method in Action: AI Benchmarks and Engagement Metrics as Targets (2023–2026)\n\n> *Example for the [goodharts-law](../SKILL.md) skill.*\n\nBy the mid-2020s, Goodhart's law had become one of the most-cited frames inside the AI industry itself — because two of its own core metrics visibly decayed under optimization pressure. First, **public benchmark scores** (MMLU, GSM8K, HumanEval, and a proliferation of leaderboards) came to dominate model marketing, funding narratives, and internal go/no-go decisions — and, predictably, models began scoring well without a matching gain in real-world capability. Second, **consumer-app engagement metrics** (watch time, session length, daily active use) continued their long slide from \"signal of user value\" to \"target that no longer measures it.\" This walks both cases through the skill's own six-step Process.\n\n---\n\n## Step 1 — State metric and goal\n\n**Case A — AI benchmarks.**\n- **Metric being targeted:** score on a fixed public benchmark (e.g., a multiple-choice knowledge test like MMLU, a grade-school math set like GSM8K, or a coding pass-rate like HumanEval).\n- **Underlying goal:** general, transferable model capability — does the model actually reason, code, and generalize on tasks users bring that were *not* in the test set?\n- **Current proxy–goal correlation:** initially high on a genuinely held-out test, and a legitimate research signal. It degrades as the benchmark ages, becomes a marketing target, and (critically) as the test's questions leak into training data.\n- **Who is measured:** frontier and open-weight model developers; the score is read by press, investors, enterprise buyers, and internal leadership deciding what to ship.\n- **Stakes:** very high — leaderboard position drives valuations, capex-justification narratives, and enterprise procurement in an intensely competitive, AI-native market.\n\n**Case B — engagement metrics.**\n- **Metric being targeted:** engagement (watch time, session length, DAU/MAU, scroll depth) feeding a recommendation or ranking system.\n- **Underlying goal:** users getting durable value — time well spent, learning, connection, satisfaction they'd endorse on reflection.\n- **Correlation:** engagement is a real proxy for value at low intensity, but the two diverge sharply as the system optimizes hard against engagement.\n- **Who is measured:** the ranking model and the teams whose OKRs it feeds.\n- **Stakes:** very high — engagement drives ad revenue and growth targets.\n\n## Step 2 — Predict the gaming (≥3 vectors each)\n\n**Case A — benchmarks:**\n1. **Train on the test (contamination).** Benchmark questions and answers, published openly on the web, get scraped into pretraining or fine-tuning corpora — accidentally or deliberately. The model then \"knows\" the answers rather than deriving them. Contamination of popular benchmarks was widely documented and discussed across the research community by 2023–2024.\n2. **Overfit the format.** Tune specifically to the benchmark's answer style, prompt template, or few-shot format so the score rises without broader capability gain.\n3. **Cherry-pick and configure.** Report the benchmark and settings (prompt, sampling, best-of-N) that flatter the model; omit the ones that don't.\n4. **Chase saturated tests.** Keep reporting benchmarks that are near-ceiling and no longer discriminate between models.\n\n**Case B — engagement:**\n1. **Amplify outrage / clickbait** — high-arousal content that maximizes watch time regardless of user benefit.\n2. **Exploit autoplay and infinite scroll** to inflate session length without adding value.\n3. **Surface borderline / sensational content** the ranking model learns is \"sticky.\"\n4. **Optimize notifications** to reacquire attention even when it degrades satisfaction.\n\nIf you can list only one or two, you haven't thought hard enough — both cases yield four readily.\n\n## Step 3 — Categorize the mechanism\n\n| Case | Dominant mechanism | Why |\n|---|---|---|\n| Benchmarks — contamination / teaching-to-test | **Adversarial** (with a **Causal** layer) | Developers under competitive pressure actively optimize the score; and a high benchmark score is increasingly a *symptom* correlated with capability, not a *cause* of it — once the test leaks, the score no longer moves capability. |\n| Benchmarks — saturated / near-ceiling tests | **Extremal** | At the top of the range the metric–goal correlation flattens; a 1-point gain near ceiling says little about real capability. |\n| Engagement | **Adversarial** + **Extremal** | The optimizer is a relentless intelligent agent; and past a threshold, more engagement stops tracking (and can invert) genuine user value. |\n\n## Step 4 — Choose the countermeasure\n\n- **Adversarial → multi-metric + randomized/rotating audits + rotation of the metric itself.** For benchmarks: rotate to fresh, held-out, contamination-controlled evaluations; keep some test items private; publish contamination checks. For engagement: multi-objective optimization with explicit quality and harm constraints.\n- **Causal → move closer to causation + direct audit.** Evaluate on tasks that are *provably* unseen and drawn from real usage, not on a fixed leaderboard. Human preference evaluation and live, blind head-to-head comparison (as popularized by community \"arena\"-style rankings in 2023–2024) are harder to pre-train against than a static answer key — though they carry their own gaming risks and are not immune.\n- **Extremal → paired constraint metrics.** Pair a capability score with an out-of-distribution / robustness check; pair engagement with a retention-quality or reported-satisfaction constraint.\n\n## Step 5 — Design the system\n\n**Case A — Goodhart-robust model evaluation:**\n- **Primary metric:** performance on **private, rotating, contamination-controlled** evaluations that approximate real user tasks.\n- **Constraint metric(s):** out-of-distribution / adversarial robustness; held-out task families the model was demonstrably not trained on.\n- **Audit:** run contamination detection (e.g., canary strings, n-gram overlap, and held-out-vs-public score gaps); publish the methodology; sample real user tasks blind.\n- **Rotation:** refresh benchmark sets on a schedule and retire saturated ones; assume any public test decays the moment it is published.\n- **Separation of measure-for-control vs measure-for-diagnosis:** never let a single public leaderboard number be both the shipped-model gate *and* the marketing headline; keep an internal, private eval for go/no-go.\n- **Gaming-detection threshold:** flag when public-benchmark score and private/held-out score diverge beyond a set gap.\n\n**Case B — engagement:**\n- **Primary metric:** engagement, but **capped by** a quality/harm constraint set.\n- **Constraint metric(s):** reported user satisfaction, long-horizon retention quality, harm audits.\n- **Audit:** sampled human review of what the ranking model is actually promoting.\n- **Rotation & separation:** revisit the objective when the engagement number rises while satisfaction stalls or falls.\n\n## Step 6 — Schedule re-evaluation\n\n- **Independent goal measurement:** blind, real-task capability evals (Case A) and honest reported-satisfaction / well-being studies (Case B), owned by a team separate from the one whose targets the metric feeds.\n- **Drift threshold:** trigger review when leaderboard gains stop tracking held-out gains, or when engagement rises without a satisfaction rise.\n- **Retirement criteria:** retire any benchmark once contamination is detected or it saturates near ceiling.\n- **Owner:** an evaluation/trust function with authority independent of the shipping and growth orgs.\n\n---\n\n## The lesson\n\nBoth cases are the same structure as Goodhart's original 1975 M3 collapse: a correlation that was real *before* it carried control weight, then decayed *because* it carried control weight. A published benchmark is a target the whole industry can optimize against — and, uniquely, one whose answer key can end up inside the very system being tested. The durable defenses are the skill's own: **multi-metric, private/rotating held-out evaluation, contamination audits, and separating the number you use to decide from the number you use to sell.**\n\n*Sources: Goodhart, C. A. E. (1975), \"Problems of monetary management: The U.K. experience,\" Reserve Bank of Australia (repr. 1984). Strathern, M. (1997), \"'Improving ratings': Audit in the British university system,\" European Review 5(3). Manheim, D., & Garrabrant, S. (2018), \"Categorizing variants of Goodhart's Law,\" arXiv:1803.04585. On benchmark contamination and the limits of static leaderboards, see the widely reported 2023–2024 discussion in the ML research community around data contamination of public benchmarks (e.g., MMLU/GSM8K/HumanEval) and the rise of community human-preference \"arena\"-style evaluations as a harder-to-game complement; and Frances Haugen / \"Facebook Files\" (2021, Wall Street Journal) documenting internally-known harms of engagement optimization. Specific figures are omitted where exact public values were not confirmable.*\n\nFile v1.0.5:examples/goodhart-1975-m3-and-strathern-1997-rae.md\n\n# Method in Action: Goodhart 1975 (M3) and Strathern 1997 (RAE)\n\n> *Example for the [goodharts-law](../SKILL.md) skill.*\n\nThe empirical foundation has two key moments. The first is **Charles Goodhart's 1975 critique of UK monetary policy**, originally a conference paper for the Reserve Bank of Australia, later expanded in his 1984 book *Monetary Theory and Practice*.\n\nThe context: in the mid-1970s, the Bank of England (and many other central banks) had identified a robust historical correlation between growth in broad money supply (the M3 aggregate) and subsequent inflation. The natural policy implication was straightforward: target M3 growth at a level consistent with low inflation, and inflation would be controlled.\n\nGoodhart was skeptical. His objection was structural, not empirical:\n\n> \"It is not the case that there is some fixed and stable relationship between an observed monetary aggregate, such as M3, and other variables in the economic system. Rather, the relationships we observe statistically are the equilibrium outcomes of behavior by banks, depositors, and borrowers, each of whom is responding to a complex set of incentives. When the regulator targets M3 — and especially when the targeting carries policy weight that will affect interest rates and reserve requirements — these economic actors will reorganize their behavior to operate around the regulation. Liquid funds will be reclassified into categories that fall outside the M3 definition. The pre-targeting M3-inflation correlation will not survive the targeting. Any observed statistical regularity will tend to collapse once pressure is placed upon it for control purposes.\"\n>\n> — Goodhart (1975), as reprinted in Goodhart (1984), pp. 96-98.\n\nGoodhart's prediction was empirical and falsifiable. It came true within a few years: as the Bank of England's M3 targets bit, UK banks began creating money-substitutes that escaped the M3 definition (most notably, the rise of the eurodollar market and certain types of negotiable certificates of deposit). The M3-inflation correlation collapsed; the Bank quietly abandoned M3 targeting by the mid-1980s.\n\nThe principle's second formative moment came two decades later, in **Marilyn Strathern's 1997 ethnographic study of British universities** during the rise of the Research Assessment Exercise (RAE). Strathern, an anthropologist at Cambridge, observed the RAE's effect on academic behavior:\n\n> \"The Research Assessment Exercise sets out to evaluate the quality of research conducted in British universities. The exercise was introduced with the goal of identifying excellent research and directing funding toward it. The instrument: publication count, citation analysis, peer-rated quality. The exercise's effect on the academy has been precisely the inversion of its stated goal. Academics have shifted from publishing fewer, more substantial works to publishing more, smaller, shallower ones. They have shifted topic selection toward areas where rapid publication is easier. They have formed citation circles. The exercise has produced not better research, but research-shaped activity that scores well on the exercise. The principle is general: when a measure becomes a target, it ceases to be a good measure.\"\n>\n> — Strathern, M. (1997). \"'Improving ratings': Audit in the British university system.\" *European Review*, 5(3), pp. 308-309.\n\nThe aphoristic compression of \"when a measure becomes a target, it ceases to be a good measure\" became the popular form of the law. It is sometimes attributed to Goodhart directly; the precise wording is Strathern's, but she was articulating Goodhart's principle in academic-administration context.\n\nThe principle has been documented across an enormous number of empirical cases. A non-exhaustive selection:\n\n**Atlanta Public Schools cheating scandal (2009-2015).** Standardized test scores were used to evaluate schools under No Child Left Behind. Teachers and administrators systematically altered student answer sheets. The scandal involved 178 educators across 44 schools; 11 educators were criminally convicted. The pattern was not unique to Atlanta; subsequent investigations found similar gaming in Houston, Washington D.C., and elsewhere. Same mechanism as Goodhart's M3: pressure on the metric, behavioral response that decoupled metric from goal.\n\n**UK NHS Accident & Emergency 4-hour target (2002-2010s).** Hospitals were required to admit, transfer, or discharge 95% of A&E patients within 4 hours. Observed gaming: parking ambulances outside the A&E entrance (so the clock didn't start), reclassifying patients to delay the count start, discharging patients prematurely just before the 4-hour mark. Studies by the NHS itself documented these behaviors at scale. The metric \"improved\" while underlying care quality stagnated or in some cases declined.\n\n**Wells Fargo cross-selling scandal (2002-2016).** Employees were given quotas for cross-selling products (checking + savings + credit card + brokerage). Sales numbers improved dramatically. Investigation revealed millions of fraudulent accounts opened without customer consent. Wells Fargo paid $3+ billion in fines. The metric (cross-sells per customer) had been targeted; employees had gamed it; the underlying goal (customer wallet share earned through service quality) had degraded.\n\n**Soviet manufacturing under Gosplan (1930s-1980s).** Factory output measured in tons of nails → factories produced enormous single nails. Output measured in number of nails → factories produced tiny useless nails. Output measured in chandeliers → enormous unliftable chandeliers (the famous case of the Moscow lighting factory). The Soviet planning system was one long case study in adversarial Goodhart's law.\n\n**Academic h-index gaming (2010s-2020s).** Self-citation, citation rings, and predatory open-access journals exploded as the h-index became a primary academic evaluation metric. The 2020+ \"paper mill\" industry — services that sell fraudulent paper authorship to academics — is a direct Goodhart's-law product.\n\n**Algorithmic recommendation systems.** YouTube, TikTok, Instagram, Facebook all use engagement metrics (watch time, like rate, share rate, comment rate) to drive recommendations. Optimizing for engagement produces: clickbait, outrage content, misinformation amplification, and (well-documented) youth mental-health harm. The 2021 Facebook Files (whistleblower Frances Haugen) showed Facebook's internal research had documented these effects but the metric-target system continued to drive them.\n\nThe theoretical literature has been refined:\n\n**Manheim & Garrabrant (2018).** \"Categorizing variants of Goodhart's Law.\" *arXiv:1803.04585*. The taxonomy (Regressional / Extremal / Causal / Adversarial) referenced above. Originated in the AI alignment community as a framework for understanding why machine-learning systems optimized on a metric reliably produce dysfunction.\n\n**Hennessy & Goodhart (2023).** \"Goodhart's Law and machine learning: A structural perspective.\" *Working paper*. The contemporary AI extension.\n\n**Campbell, D. T. (1979).** \"Assessing the impact of planned social change.\" *Evaluation and Program Planning*, 2(1), 67-90. An independent formulation predating Goodhart's wider recognition, now called \"Campbell's law\": \"The more any quantitative social indicator is used for social decision-making, the more subject it will be to corruption pressures and the more apt it will be to distort and corrupt the social processes it is intended to monitor.\" Same content, different domain emphasis.\n\n**Lucas, R. E. (1976).** \"Econometric policy evaluation: A critique.\" *Carnegie-Rochester Conference Series on Public Policy*, 1(1), 19-46. The \"Lucas critique\" in macroeconomics — the recognition that econometric relationships estimated under one policy regime are not stable when the regime changes, because actors will respond to the new regime. Adjacent to Goodhart's law in mechanism.\n\nThe framework has reshaped operational design in multiple disciplines:\n\n**Modern bank regulation (Basel III, 2017+).** Capital and liquidity ratios are now paired with stress tests, leverage caps, and additional Tier 1 capital buffers — explicitly to constrain optimization of any single metric. The architecture is multi-metric, audit-augmented, and rotation-enabled (regulators update standards every few years specifically to keep ahead of gaming).\n\n**Modern KPI design.** Best-practice OKR frameworks (Doerr 2018) recommend mixing leading and lagging indicators, qualitative and quantitative metrics, and pairing each KR with an explicit audit / quality check. Single-metric reward systems are increasingly regarded as a known-bad design pattern.\n\n**Educational assessment.** Post-NCLB reforms have shifted toward sampling-based assessment, multi-measure school evaluation, and reduction of single-test stakes. Some states have explicitly abandoned standardized testing as an evaluation metric because the gaming costs exceeded the information benefits.\n\n**Algorithmic system design.** ML systems are increasingly designed with multi-objective optimization, adversarial robustness checks, and explicit decoupling of metric (for training) from outcome (for evaluation). The AI alignment field is essentially a deep study of Goodhart-style failure in increasingly powerful optimization systems.\n\n**Healthcare quality metrics.** UK NHS, U.S. CMS, and others have moved away from single-target measurement (e.g., A&E 4-hour) toward composite quality scores that include patient experience, clinical outcomes, and process metrics that can't be simultaneously gamed.\n\nThree operational lessons from Goodhart and Strathern:\n\n**First, all metrics are proxies. The proxy is always narrower than the goal.** The question is not \"is this a good metric?\" but \"what is the gap between the metric and the goal, and how will agents exploit that gap?\" Pre-committing to the gap analysis before deployment is the single highest-leverage step.\n\n**Second, gaming is rational behavior under measurement pressure, not character failure.** Blaming the people who game metrics is a category error — they are responding to incentives, often appropriately. The fix is the measurement system, not the people.\n\n**Third, rotation and audit are the only durable defenses.** No metric, however carefully designed, survives indefinite optimization pressure without gaming. The structural defense is to plan rotation (replace or refresh metrics before they're fully corrupted) and external audit (sample the underlying goal directly, periodically, to catch drift). Metrics are not assets; they are decaying assets.\n\nFile v1.0.5:skill-card.md\n\n## Description:\n\nGuides agents to recognize when metrics become targets and design Goodhart-robust systems with gaming vectors, mechanism categories, countermeasures, audits, and metric rotation.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[deciqai](https://clawhub.ai/user/deciqai)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nEmployees, external operators, and developers use this skill to diagnose when KPI, benchmark, audit, ranking, or incentive systems are likely to be gamed. It helps produce a practical design with a primary metric, constraint metrics, audits, rotation schedule, independent goal measurement, and retirement criteria.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Advisory reasoning can be misapplied as definitive governance advice for high-stakes metric or incentive systems.\n\nMitigation: Use outputs as decision support and have responsible owners review metric-goal fit, audit design, deployment context, and retirement criteria before relying on them.\n\nRisk: Examples and external links are reference material and may be incomplete, stale, or unavailable.\n\nMitigation: Verify cited examples and external references before using them as evidence for operational, compliance, or procurement decisions.\n\n## Reference(s):\n\n- [Sources - goodharts-law](references/sources.md)\n- [Goodhart 1975 (M3) and Strathern 1997 (RAE) example](examples/goodhart-1975-m3-and-strathern-1997-rae.md)\n- [AI benchmarks and engagement metrics as targets (2023-2026)](examples/ai-benchmark-and-engagement-gaming-2023-2026.md)\n- [ClawHub skill page](https://clawhub.ai/deciqai/skills/goodharts-law)\n- [deciqAI Goodhart's Law page](https://www.deciqai.com/c/goodharts-law)\n- [deciqAI machine-readable metadata](https://www.deciqai.com/s/goodharts-law.json)\n- [deciqAI knowledge-skills repository](https://github.com/deciqAI/knowledge-skills)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, guidance]\n\n**Output Format:** [Markdown response with a structured Goodhart-robust design template]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May ask step-by-step questions in coach mode and stop for user input before completing the design.]\n\n## Skill Version(s):\n\n1.0.5 (source: server release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v1.0.4: 6 files, 16063 bytes\n\nFiles: examples/ai-benchmark-and-engagement-gaming-2023-2026.md (9141b), examples/goodhart-1975-m3-and-strathern-1997-rae.md (10675b), references/sources.md (2000b), skill-card.md (2437b), SKILL.md (7724b), _meta.json (132b)\n\nFile v1.0.4:SKILL.md\n\n---\nname: goodharts-law\ndescription: \"Activate when: our KPI is going up but the real outcome isn't improving; people seem to be gaming the metric; we're about to tie bonuses or promotions to a number; an algorithm is producing results nobody intended; a test or audit system is being designed.\n  Do NOT activate when: the metric IS the goal with no proxy gap; measurement is purely descriptive with zero stakes attached.\"\n---\n\n# Goodhart's Law\n\n## Overview\n\n**Goodhart's Law:** when a metric controls behavior, people optimize the metric rather than the underlying goal. Formulated by economist Charles Goodhart (1975) on UK monetary policy; sharpened by Marilyn Strathern (1997): *\"When a measure becomes a target, it ceases to be a good measure.\"* Four failure mechanisms (Manheim & Garrabrant 2018): **Regressional**, **Extremal**, **Causal**, **Adversarial**. Countermeasure is always multi-metric + audit + rotation.\n\nComposes with `feedback-loops`, `principal-agent`, `okr-goal-setting`, `survivorship-bias`.\n\n## When to Use\n\n- A KPI is being introduced or its weight is increasing in performance evaluation\n- A metric is \"improving\" without corresponding improvement in the underlying goal\n- People are visibly optimizing for a number rather than the work it was meant to track\n- Algorithmic optimization is producing outcomes the designers didn't intend\n- Resource allocation is driven by a single composite score or ranking\n- An AI model, benchmark, or engagement metric is being optimized (or used to justify AI capex / adoption / AI-native competition) and the score is rising faster than real capability or user value\n\n**Not when:** metric and goal are identical; stakes too low for gaming; metric is purely descriptive with no reward/punishment; question is which metric to use, not whether the measurement-reward system is sound.\n\n## Coaching Novices (Adaptive Front Door)\n\n- **Engine mode:** user has a concrete metric or system → run The Process directly.\n- **Coach mode:** user is unfamiliar or has no concrete case → guide step by step.\n\nIn Coach mode, respond one step at a time. Each [WAIT] is a hard stop — output only that step's question, then stop.\n\n1. One-line: before relying on a metric to control behavior, predict how people will game it — choose the system that survives that prediction.\n2. Check fit: if the metric is purely descriptive (no reward attached), Goodhart's law doesn't apply yet.\n3. Elicit the specific metric and the underlying goal: what's being measured? What's the actual outcome you care about?\n> **[WAIT — do not advance until user responds]**\n4. One question at a time: proxy gap? How would a clever agent game this? Which Goodhart category? What countermeasure fits?\n> **[WAIT — do not advance until user responds]**\n5. Close: name the gaming-resistant design (multi-metric, audit, rotation, paired-constraint) + monitoring schedule.\n> **[WAIT — do not advance until user responds]**\n\n## The Process\n\n**Step 1 — State metric and goal:** metric being targeted / underlying goal / current proxy-goal correlation / who is measured / stakes.\n\n**Step 2 — Predict the gaming:** list ≥3 ways to game the metric with minimum effort on the goal. If you can't list 3, you haven't thought hard enough.\n\n**Step 3 — Categorize mechanism:**\n\n| Mechanism | Test |\n|---|---|\n| Regressional | Is there noise that optimization will push into? |\n| Extremal | Does metric-goal correlation break at extremes? |\n| Causal | Is the metric a symptom, not a cause? |\n| Adversarial | Will agents actively game with intelligence? |\n\n**Step 4 — Choose countermeasure:** Regressional → constrain range. Extremal → paired constraint metrics. Causal → closer-to-causation metric + direct audit. Adversarial → multi-metric + randomized audits + rotation.\n**Step 5 — Design the system:** primary metric / constraint metric(s) / audit mechanism (sampled direct goal observation) / rotation schedule / separation of measure-for-control from measure-for-diagnosis / gaming-detection threshold.\n**Step 6 — Schedule re-evaluation:** independent goal measurement (how/when/who) / drift threshold / retirement criteria / owner.\n\n## Output template\n\n```markdown\n# Goodhart-Robust Design: <metric>\nMetric: | Underlying goal: | Correlation: | Who measured: | Stakes:\nGaming vectors (≥3):\nMechanism: Regressional / Extremal / Causal / Adversarial\nPrimary metric: | Constraint metric(s): | Audit: | Rotation: | Separation: | Gaming threshold:\nGoal measurement (independent): | Drift threshold: | Retirement criteria: | Owner:\n```\n\n*→ Method in Action: [Goodhart 1975 (M3) and Strathern 1997 (RAE)](examples/goodhart-1975-m3-and-strathern-1997-rae.md)*\n*→ 2026 lens: [AI benchmarks and engagement metrics as targets (2023–2026)](examples/ai-benchmark-and-engagement-gaming-2023-2026.md)*\n\n## Pack: Goodhart's Law Patterns\n\n| Domain | Common gaming | Defense |\n|---|---|---|\n| Sales quotas | Sandbagging, channel stuffing, end-of-quarter discounts | Multi-period averaging; quality metrics; clawback |\n| Hospital wait targets | Ambulance parking, patient reclassification | Outcome audits; paired metrics; randomized inspection |\n| Standardized testing | Teaching to test, curriculum narrowing | Sample-based assessment; multi-measure; reduce single-test stakes |\n| Algorithmic engagement | Clickbait, outrage, misinformation | Multi-objective optimization; quality + harm constraints |\n\n## Applying It Well\n\n- All metrics are proxies — narrower than the goal. Pre-commit to gap analysis before deployment.\n- Gaming is rational under measurement pressure. Fix the system, not the people.\n- Rotation and audit are the only durable defenses. Plan metric retirement at design time.\n\n*→ Primary sources: [references/sources.md](references/sources.md)*\n\n## Common Rationalizations\n\n| Fake move | Reality |\n|---|---|\n| [D] \"If you can't measure it, you can't manage it\" | Often false. Judgment, trust, and direct observation are also valid management tools. |\n| [D] \"Our metric is well-defined; it won't be gamed\" | Precision invites precise gaming. Basel II capital ratios were well-defined — extensively gamed. |\n| [D] \"Our people wouldn't game the metric\" | Goodhart's law is structural; individual virtue is insufficient in aggregate. |\n| [D] \"We just need a better metric\" | Often the issue is any single metric under pressure; fix is multi-metric + audit. |\n| [D] \"We've used this metric for years\" | Long use = more time for gaming to mature. Tenure is a warning, not an endorsement. |\n| *→ Add [O] entries here after each real use — paste the actual failure pattern* | *What went wrong and why* |\n\n**[D] = designed upfront | [O] = observed in real use. [O] entries are more valuable.**\n\n## Red Flags\n\n- Metric tied to high-stakes rewards or punishments\n- Metric \"improves\" without obvious improvement in the underlying goal\n- People being measured can already articulate ways to game it\n- Single metric is the primary evaluation tool, no audit or paired-constraint\n\n## Verification\n\n- [ ] Goal underlying the metric specifically named\n- [ ] Proxy gap explicitly described; ≥3 gaming vectors listed\n- [ ] Goodhart mechanism category identified\n- [ ] Countermeasure design (multi-metric, audit, rotation) in place\n- [ ] Independent goal measurement scheduled with owner and retirement date\n\n---\n\n*Part of **deciqAI Knowledge Skills** — 189 open-source thinking skills that make rigor executable for AI agents. The same skills power every deciqAI agent, which runs them autonomously to operate your company. **See it run → https://www.deciqai.com/c/goodharts-law** · ⭐ Star the repo → https://github.com/deciqAI/knowledge-skills · Contributions welcome.*\n\nFile v1.0.4:_meta.json\n\n{\n  \"ownerId\": \"kn754b8sk22s8c6gjxt02bftbn88q7ye\",\n  \"slug\": \"goodharts-law\",\n  \"version\": \"1.0.4\",\n  \"publishedAt\": 1783595874034\n}\n\nFile v1.0.4:references/sources.md\n\n# Sources — goodharts-law\n\n> *Primary sources for the [goodharts-law](../SKILL.md) skill.*\n\n- Goodhart, C. A. E. (1975). \"Problems of monetary management: The U.K. experience.\" Papers in Monetary Economics, Reserve Bank of Australia. Reprinted in *Monetary Theory and Practice: The U.K. Experience* (1984), Macmillan. The original.\n- Strathern, M. (1997). \"'Improving ratings': Audit in the British university system.\" *European Review*, 5(3), 305-321. The modern aphoristic formulation.\n- Campbell, D. T. (1979). \"Assessing the impact of planned social change.\" *Evaluation and Program Planning*, 2(1), 67-90. Independent formulation (\"Campbell's Law\").\n- Lucas, R. E. (1976). \"Econometric policy evaluation: A critique.\" *Carnegie-Rochester Conference Series on Public Policy*, 1(1), 19-46. The adjacent macro-econometric \"Lucas critique.\"\n- Manheim, D., & Garrabrant, S. (2018). \"Categorizing variants of Goodhart's Law.\" *arXiv:1803.04585*. The four-mechanism taxonomy.\n- Muller, J. Z. (2018). *The Tyranny of Metrics.* Princeton University Press. ISBN 978-0691174952. Comprehensive case-study survey.\n- Doerr, J. (2018). *Measure What Matters: How Google, Bono, and the Gates Foundation Rock the World with OKRs.* Portfolio. ISBN 978-0525536222.\n- Wells Fargo / Office of the Comptroller of the Currency (2016, 2018). Various enforcement actions and consent orders documenting the cross-selling case.\n- Zhou, K., et al. (2023). \"Don't Make Your LLM an Evaluation Benchmark Cheater.\" *arXiv:2311.01964*. Documents benchmark data contamination and how public evaluation scores decay as a capability signal once test data leaks into training — a direct 2020s AI instance of Goodhart's law.\n- Chiang, W.-L., et al. (2024). \"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference.\" *arXiv:2403.04132*. Motivates live, blind, human-preference evaluation as a harder-to-game complement to static leaderboards; illustrates both the countermeasure and its own residual gaming risks.\n\nFile v1.0.4:examples/ai-benchmark-and-engagement-gaming-2023-2026.md\n\n# Method in Action: AI Benchmarks and Engagement Metrics as Targets (2023–2026)\n\n> *Example for the [goodharts-law](../SKILL.md) skill.*\n\nBy the mid-2020s, Goodhart's law had become one of the most-cited frames inside the AI industry itself — because two of its own core metrics visibly decayed under optimization pressure. First, **public benchmark scores** (MMLU, GSM8K, HumanEval, and a proliferation of leaderboards) came to dominate model marketing, funding narratives, and internal go/no-go decisions — and, predictably, models began scoring well without a matching gain in real-world capability. Second, **consumer-app engagement metrics** (watch time, session length, daily active use) continued their long slide from \"signal of user value\" to \"target that no longer measures it.\" This walks both cases through the skill's own six-step Process.\n\n---\n\n## Step 1 — State metric and goal\n\n**Case A — AI benchmarks.**\n- **Metric being targeted:** score on a fixed public benchmark (e.g., a multiple-choice knowledge test like MMLU, a grade-school math set like GSM8K, or a coding pass-rate like HumanEval).\n- **Underlying goal:** general, transferable model capability — does the model actually reason, code, and generalize on tasks users bring that were *not* in the test set?\n- **Current proxy–goal correlation:** initially high on a genuinely held-out test, and a legitimate research signal. It degrades as the benchmark ages, becomes a marketing target, and (critically) as the test's questions leak into training data.\n- **Who is measured:** frontier and open-weight model developers; the score is read by press, investors, enterprise buyers, and internal leadership deciding what to ship.\n- **Stakes:** very high — leaderboard position drives valuations, capex-justification narratives, and enterprise procurement in an intensely competitive, AI-native market.\n\n**Case B — engagement metrics.**\n- **Metric being targeted:** engagement (watch time, session length, DAU/MAU, scroll depth) feeding a recommendation or ranking system.\n- **Underlying goal:** users getting durable value — time well spent, learning, connection, satisfaction they'd endorse on reflection.\n- **Correlation:** engagement is a real proxy for value at low intensity, but the two diverge sharply as the system optimizes hard against engagement.\n- **Who is measured:** the ranking model and the teams whose OKRs it feeds.\n- **Stakes:** very high — engagement drives ad revenue and growth targets.\n\n## Step 2 — Predict the gaming (≥3 vectors each)\n\n**Case A — benchmarks:**\n1. **Train on the test (contamination).** Benchmark questions and answers, published openly on the web, get scraped into pretraining or fine-tuning corpora — accidentally or deliberately. The model then \"knows\" the answers rather than deriving them. Contamination of popular benchmarks was widely documented and discussed across the research community by 2023–2024.\n2. **Overfit the format.** Tune specifically to the benchmark's answer style, prompt template, or few-shot format so the score rises without broader capability gain.\n3. **Cherry-pick and configure.** Report the benchmark and settings (prompt, sampling, best-of-N) that flatter the model; omit the ones that don't.\n4. **Chase saturated tests.** Keep reporting benchmarks that are near-ceiling and no longer discriminate between models.\n\n**Case B — engagement:**\n1. **Amplify outrage / clickbait** — high-arousal content that maximizes watch time regardless of user benefit.\n2. **Exploit autoplay and infinite scroll** to inflate session length without adding value.\n3. **Surface borderline / sensational content** the ranking model learns is \"sticky.\"\n4. **Optimize notifications** to reacquire attention even when it degrades satisfaction.\n\nIf you can list only one or two, you haven't thought hard enough — both cases yield four readily.\n\n## Step 3 — Categorize the mechanism\n\n| Case | Dominant mechanism | Why |\n|---|---|---|\n| Benchmarks — contamination / teaching-to-test | **Adversarial** (with a **Causal** layer) | Developers under competitive pressure actively optimize the score; and a high benchmark score is increasingly a *symptom* correlated with capability, not a *cause* of it — once the test leaks, the score no longer moves capability. |\n| Benchmarks — saturated / near-ceiling tests | **Extremal** | At the top of the range the metric–goal correlation flattens; a 1-point gain near ceiling says little about real capability. |\n| Engagement | **Adversarial** + **Extremal** | The optimizer is a relentless intelligent agent; and past a threshold, more engagement stops tracking (and can invert) genuine user value. |\n\n## Step 4 — Choose the countermeasure\n\n- **Adversarial → multi-metric + randomized/rotating audits + rotation of the metric itself.** For benchmarks: rotate to fresh, held-out, contamination-controlled evaluations; keep some test items private; publish contamination checks. For engagement: multi-objective optimization with explicit quality and harm constraints.\n- **Causal → move closer to causation + direct audit.** Evaluate on tasks that are *provably* unseen and drawn from real usage, not on a fixed leaderboard. Human preference evaluation and live, blind head-to-head comparison (as popularized by community \"arena\"-style rankings in 2023–2024) are harder to pre-train against than a static answer key — though they carry their own gaming risks and are not immune.\n- **Extremal → paired constraint metrics.** Pair a capability score with an out-of-distribution / robustness check; pair engagement with a retention-quality or reported-satisfaction constraint.\n\n## Step 5 — Design the system\n\n**Case A — Goodhart-robust model evaluation:**\n- **Primary metric:** performance on **private, rotating, contamination-controlled** evaluations that approximate real user tasks.\n- **Constraint metric(s):** out-of-distribution / adversarial robustness; held-out task families the model was demonstrably not trained on.\n- **Audit:** run contamination detection (e.g., canary strings, n-gram overlap, and held-out-vs-public score gaps); publish the methodology; sample real user tasks blind.\n- **Rotation:** refresh benchmark sets on a schedule and retire saturated ones; assume any public test decays the moment it is published.\n- **Separation of measure-for-control vs measure-for-diagnosis:** never let a single public leaderboard number be both the shipped-model gate *and* the marketing headline; keep an internal, private eval for go/no-go.\n- **Gaming-detection threshold:** flag when public-benchmark score and private/held-out score diverge beyond a set gap.\n\n**Case B — engagement:**\n- **Primary metric:** engagement, but **capped by** a quality/harm constraint set.\n- **Constraint metric(s):** reported user satisfaction, long-horizon retention quality, harm audits.\n- **Audit:** sampled human review of what the ranking model is actually promoting.\n- **Rotation & separation:** revisit the objective when the engagement number rises while satisfaction stalls or falls.\n\n## Step 6 — Schedule re-evaluation\n\n- **Independent goal measurement:** blind, real-task capability evals (Case A) and honest reported-satisfaction / well-being studies (Case B), owned by a team separate from the one whose targets the metric feeds.\n- **Drift threshold:** trigger review when leaderboard gains stop tracking held-out gains, or when engagement rises without a satisfaction rise.\n- **Retirement criteria:** retire any benchmark once contamination is detected or it saturates near ceiling.\n- **Owner:** an evaluation/trust function with authority independent of the shipping and growth orgs.\n\n---\n\n## The lesson\n\nBoth cases are the same structure as Goodhart's original 1975 M3 collapse: a correlation that was real *before* it carried control weight, then decayed *because* it carried control weight. A published benchmark is a target the whole industry can optimize against — and, uniquely, one whose answer key can end up inside the very system being tested. The durable defenses are the skill's own: **multi-metric, private/rotating held-out evaluation, contamination audits, and separating the number you use to decide from the number you use to sell.**\n\n*Sources: Goodhart, C. A. E. (1975), \"Problems of monetary management: The U.K. experience,\" Reserve Bank of Australia (repr. 1984). Strathern, M. (1997), \"'Improving ratings': Audit in the British university system,\" European Review 5(3). Manheim, D., & Garrabrant, S. (2018), \"Categorizing variants of Goodhart's Law,\" arXiv:1803.04585. On benchmark contamination and the limits of static leaderboards, see the widely reported 2023–2024 discussion in the ML research community around data contamination of public benchmarks (e.g., MMLU/GSM8K/HumanEval) and the rise of community human-preference \"arena\"-style evaluations as a harder-to-game complement; and Frances Haugen / \"Facebook Files\" (2021, Wall Street Journal) documenting internally-known harms of engagement optimization. Specific figures are omitted where exact public values were not confirmable.*\n\nFile v1.0.4:examples/goodhart-1975-m3-and-strathern-1997-rae.md\n\n# Method in Action: Goodhart 1975 (M3) and Strathern 1997 (RAE)\n\n> *Example for the [goodharts-law](../SKILL.md) skill.*\n\nThe empirical foundation has two key moments. The first is **Charles Goodhart's 1975 critique of UK monetary policy**, originally a conference paper for the Reserve Bank of Australia, later expanded in his 1984 book *Monetary Theory and Practice*.\n\nThe context: in the mid-1970s, the Bank of England (and many other central banks) had identified a robust historical correlation between growth in broad money supply (the M3 aggregate) and subsequent inflation. The natural policy implication was straightforward: target M3 growth at a level consistent with low inflation, and inflation would be controlled.\n\nGoodhart was skeptical. His objection was structural, not empirical:\n\n> \"It is not the case that there is some fixed and stable relationship between an observed monetary aggregate, such as M3, and other variables in the economic system. Rather, the relationships we observe statistically are the equilibrium outcomes of behavior by banks, depositors, and borrowers, each of whom is responding to a complex set of incentives. When the regulator targets M3 — and especially when the targeting carries policy weight that will affect interest rates and reserve requirements — these economic actors will reorganize their behavior to operate around the regulation. Liquid funds will be reclassified into categories that fall outside the M3 definition. The pre-targeting M3-inflation correlation will not survive the targeting. Any observed statistical regularity will tend to collapse once pressure is placed upon it for control purposes.\"\n>\n> — Goodhart (1975), as reprinted in Goodhart (1984), pp. 96-98.\n\nGoodhart's prediction was empirical and falsifiable. It came true within a few years: as the Bank of England's M3 targets bit, UK banks began creating money-substitutes that escaped the M3 definition (most notably, the rise of the eurodollar market and certain types of negotiable certificates of deposit). The M3-inflation correlation collapsed; the Bank quietly abandoned M3 targeting by the mid-1980s.\n\nThe principle's second formative moment came two decades later, in **Marilyn Strathern's 1997 ethnographic study of British universities** during the rise of the Research Assessment Exercise (RAE). Strathern, an anthropologist at Cambridge, observed the RAE's effect on academic behavior:\n\n> \"The Research Assessment Exercise sets out to evaluate the quality of research conducted in British universities. The exercise was introduced with the goal of identifying excellent research and directing funding toward it. The instrument: publication count, citation analysis, peer-rated quality. The exercise's effect on the academy has been precisely the inversion of its stated goal. Academics have shifted from publishing fewer, more substantial works to publishing more, smaller, shallower ones. They have shifted topic selection toward areas where rapid publication is easier. They have formed citation circles. The exercise has produced not better research, but research-shaped activity that scores well on the exercise. The principle is general: when a measure becomes a target, it ceases to be a good measure.\"\n>\n> — Strathern, M. (1997). \"'Improving ratings': Audit in the British university system.\" *European Review*, 5(3), pp. 308-309.\n\nThe aphoristic compression of \"when a measure becomes a target, it ceases to be a good measure\" became the popular form of the law. It is sometimes attributed to Goodhart directly; the precise wording is Strathern's, but she was articulating Goodhart's principle in academic-administration context.\n\nThe principle has been documented across an enormous number of empirical cases. A non-exhaustive selection:\n\n**Atlanta Public Schools cheating scandal (2009-2015).** Standardized test scores were used to evaluate schools under No Child Left Behind. Teachers and administrators systematically altered student answer sheets. The scandal involved 178 educators across 44 schools; 11 educators were criminally convicted. The pattern was not unique to Atlanta; subsequent investigations found similar gaming in Houston, Washington D.C., and elsewhere. Same mechanism as Goodhart's M3: pressure on the metric, behavioral response that decoupled metric from goal.\n\n**UK NHS Accident & Emergency 4-hour target (2002-2010s).** Hospitals were required to admit, transfer, or discharge 95% of A&E patients within 4 hours. Observed gaming: parking ambulances outside the A&E entrance (so the clock didn't start), reclassifying patients to delay the count start, discharging patients prematurely just before the 4-hour mark. Studies by the NHS itself documented these behaviors at scale. The metric \"improved\" while underlying care quality stagnated or in some cases declined.\n\n**Wells Fargo cross-selling scandal (2002-2016).** Employees were given quotas for cross-selling products (checking + savings + credit card + brokerage). Sales numbers improved dramatically. Investigation revealed millions of fraudulent accounts opened without customer consent. Wells Fargo paid $3+ billion in fines. The metric (cross-sells per customer) had been targeted; employees had gamed it; the underlying goal (customer wallet share earned through service quality) had degraded.\n\n**Soviet manufacturing under Gosplan (1930s-1980s).** Factory output measured in tons of nails → factories produced enormous single nails. Output measured in number of nails → factories produced tiny useless nails. Output measured in chandeliers → enormous unliftable chandeliers (the famous case of the Moscow lighting factory). The Soviet planning system was one long case study in adversarial Goodhart's law.\n\n**Academic h-index gaming (2010s-2020s).** Self-citation, citation rings, and predatory open-access journals exploded as the h-index became a primary academic evaluation metric. The 2020+ \"paper mill\" industry — services that sell fraudulent paper authorship to academics — is a direct Goodhart's-law product.\n\n**Algorithmic recommendation systems.** YouTube, TikTok, Instagram, Facebook all use engagement metrics (watch time, like rate, share rate, comment rate) to drive recommendations. Optimizing for engagement produces: clickbait, outrage content, misinformation amplification, and (well-documented) youth mental-health harm. The 2021 Facebook Files (whistleblower Frances Haugen) showed Facebook's internal research had documented these effects but the metric-target system continued to drive them.\n\nThe theoretical literature has been refined:\n\n**Manheim & Garrabrant (2018).** \"Categorizing variants of Goodhart's Law.\" *arXiv:1803.04585*. The taxonomy (Regressional / Extremal / Causal / Adversarial) referenced above. Originated in the AI alignment community as a framework for understanding why machine-learning systems optimized on a metric reliably produce dysfunction.\n\n**Hennessy & Goodhart (2023).** \"Goodhart's Law and machine learning: A structural perspective.\" *Working paper*. The contemporary AI extension.\n\n**Campbell, D. T. (1979).** \"Assessing the impact of planned social change.\" *Evaluation and Program Planning*, 2(1), 67-90. An independent formulation predating Goodhart's wider recognition, now called \"Campbell's law\": \"The more any quantitative social indicator is used for social decision-making, the more subject it will be to corruption pressures and the more apt it will be to distort and corrupt the social processes it is intended to monitor.\" Same content, different domain emphasis.\n\n**Lucas, R. E. (1976).** \"Econometric policy evaluation: A critique.\" *Carnegie-Rochester Conference Series on Public Policy*, 1(1), 19-46. The \"Lucas critique\" in macroeconomics — the recognition that econometric relationships estimated under one policy regime are not stable when the regime changes, because actors will respond to the new regime. Adjacent to Goodhart's law in mechanism.\n\nThe framework has reshaped operational design in multiple disciplines:\n\n**Modern bank regulation (Basel III, 2017+).** Capital and liquidity ratios are now paired with stress tests, leverage caps, and additional Tier 1 capital buffers — explicitly to constrain optimization of any single metric. The architecture is multi-metric, audit-augmented, and rotation-enabled (regulators update standards every few years specifically to keep ahead of gaming).\n\n**Modern KPI design.** Best-practice OKR frameworks (Doerr 2018) recommend mixing leading and lagging indicators, qualitative and quantitative metrics, and pairing each KR with an explicit audit / quality check. Single-metric reward systems are increasingly regarded as a known-bad design pattern.\n\n**Educational assessment.** Post-NCLB reforms have shifted toward sampling-based assessment, multi-measure school evaluation, and reduction of single-test stakes. Some states have explicitly abandoned standardized testing as an evaluation metric because the gaming costs exceeded the information benefits.\n\n**Algorithmic system design.** ML systems are increasingly designed with multi-objective optimization, adversarial robustness checks, and explicit decoupling of metric (for training) from outcome (for evaluation). The AI alignment field is essentially a deep study of Goodhart-style failure in increasingly powerful optimization systems.\n\n**Healthcare quality metrics.** UK NHS, U.S. CMS, and others have moved away from single-target measurement (e.g., A&E 4-hour) toward composite quality scores that include patient experience, clinical outcomes, and process metrics that can't be simultaneously gamed.\n\nThree operational lessons from Goodhart and Strathern:\n\n**First, all metrics are proxies. The proxy is always narrower than the goal.** The question is not \"is this a good metric?\" but \"what is the gap between the metric and the goal, and how will agents exploit that gap?\" Pre-committing to the gap analysis before deployment is the single highest-leverage step.\n\n**Second, gaming is rational behavior under measurement pressure, not character failure.** Blaming the people who game metrics is a category error — they are responding to incentives, often appropriately. The fix is the measurement system, not the people.\n\n**Third, rotation and audit are the only durable defenses.** No metric, however carefully designed, survives indefinite optimization pressure without gaming. The structural defense is to plan rotation (replace or refresh metrics before they're fully corrupted) and external audit (sample the underlying goal directly, periodically, to catch drift). Metrics are not assets; they are decaying assets.\n\nFile v1.0.4:skill-card.md\n\n## Description: <br>\nHelps agents detect when metrics, KPIs, audits, benchmarks, or reward systems are likely to be gamed and design multi-metric, audited, rotating controls. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[deciqai](https://clawhub.ai/user/deciqai) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nEmployees, operators, product leaders, and developers use this skill to analyze metric-driven systems before tying decisions, incentives, rankings, or model evaluations to a number. It guides the agent to identify proxy gaps, predict gaming vectors, classify the Goodhart mechanism, and propose audit, rotation, and paired-metric defenses. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill may be allowed to run in autonomous deciqAI workflows depending on the host agent configuration. <br>\nMitigation: Confirm which deciqAI workflows the skill can run and whether autonomous behavior is enabled before installation. <br>\nRisk: Metric-system recommendations can be misleading if adopted without review of the real organizational incentives and available audit data. <br>\nMitigation: Review the proposed gaming vectors, countermeasures, and audit plan with the system owner before using them for incentive or evaluation decisions. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/deciqai/skills/goodharts-law) <br>\n- [Primary sources for goodharts-law](artifact/references/sources.md) <br>\n- [Goodhart 1975 and Strathern 1997 example](artifact/examples/goodhart-1975-m3-and-strathern-1997-rae.md) <br>\n- [AI benchmarks and engagement metrics example](artifact/examples/ai-benchmark-and-engagement-gaming-2023-2026.md) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, guidance] <br>\n**Output Format:** [Markdown analysis and structured design template] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [May ask step-by-step clarification questions before producing the final design.] <br>\n\n## Skill Version(s): <br>\n1.0.4 (source: server release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.0.3: 5 files, 11189 bytes\n\nFiles: examples/goodhart-1975-m3-and-strathern-1997-rae.md (10675b), references/sources.md (1408b), skill-card.md (2232b), SKILL.md (7388b), _meta.json (132b)\n\nFile v1.0.3:SKILL.md\n\n---\nname: goodharts-law\ndescription: \"Activate when: our KPI is going up but the real outcome isn't improving; people seem to be gaming the metric; we're about to tie bonuses or promotions to a number; an algorithm is producing results nobody intended; a test or audit system is being designed.\n  Do NOT activate when: the metric IS the goal with no proxy gap; measurement is purely descriptive with zero stakes attached.\"\n---\n\n# Goodhart's Law\n\n## Overview\n\n**Goodhart's Law:** when a metric controls behavior, people optimize the metric rather than the underlying goal. Formulated by economist Charles Goodhart (1975) on UK monetary policy; sharpened by Marilyn Strathern (1997): *\"When a measure becomes a target, it ceases to be a good measure.\"* Four failure mechanisms (Manheim & Garrabrant 2018): **Regressional**, **Extremal**, **Causal**, **Adversarial**. Countermeasure is always multi-metric + audit + rotation.\n\nComposes with `feedback-loops`, `principal-agent`, `okr-goal-setting`, `survivorship-bias`.\n\n## When to Use\n\n- A KPI is being introduced or its weight is increasing in performance evaluation\n- A metric is \"improving\" without corresponding improvement in the underlying goal\n- People are visibly optimizing for a number rather than the work it was meant to track\n- Algorithmic optimization is producing outcomes the designers didn't intend\n- Resource allocation is driven by a single composite score or ranking\n\n**Not when:** metric and goal are identical; stakes too low for gaming; metric is purely descriptive with no reward/punishment; question is which metric to use, not whether the measurement-reward system is sound.\n\n## Coaching Novices (Adaptive Front Door)\n\n- **Engine mode:** user has a concrete metric or system → run The Process directly.\n- **Coach mode:** user is unfamiliar or has no concrete case → guide step by step.\n\nIn Coach mode, respond one step at a time. Each [WAIT] is a hard stop — output only that step's question, then stop.\n\n1. One-line: before relying on a metric to control behavior, predict how people will game it — choose the system that survives that prediction.\n2. Check fit: if the metric is purely descriptive (no reward attached), Goodhart's law doesn't apply yet.\n3. Elicit the specific metric and the underlying goal: what's being measured? What's the actual outcome you care about?\n> **[WAIT — do not advance until user responds]**\n4. One question at a time: proxy gap? How would a clever agent game this? Which Goodhart category? What countermeasure fits?\n> **[WAIT — do not advance until user responds]**\n5. Close: name the gaming-resistant design (multi-metric, audit, rotation, paired-constraint) + monitoring schedule.\n> **[WAIT — do not advance until user responds]**\n\n## The Process\n\n**Step 1 — State metric and goal:** metric being targeted / underlying goal / current proxy-goal correlation / who is measured / stakes.\n\n**Step 2 — Predict the gaming:** list ≥3 ways to game the metric with minimum effort on the goal. If you can't list 3, you haven't thought hard enough.\n\n**Step 3 — Categorize mechanism:**\n\n| Mechanism | Test |\n|---|---|\n| Regressional | Is there noise that optimization will push into? |\n| Extremal | Does metric-goal correlation break at extremes? |\n| Causal | Is the metric a symptom, not a cause? |\n| Adversarial | Will agents actively game with intelligence? |\n\n**Step 4 — Choose countermeasure:** Regressional → constrain range. Extremal → paired constraint metrics. Causal → closer-to-causation metric + direct audit. Adversarial → multi-metric + randomized audits + rotation.\n**Step 5 — Design the system:** primary metric / constraint metric(s) / audit mechanism (sampled direct goal observation) / rotation schedule / separation of measure-for-control from measure-for-diagnosis / gaming-detection threshold.\n**Step 6 — Schedule re-evaluation:** independent goal measurement (how/when/who) / drift threshold / retirement criteria / owner.\n\n## Output template\n\n```markdown\n# Goodhart-Robust Design: <metric>\nMetric: | Underlying goal: | Correlation: | Who measured: | Stakes:\nGaming vectors (≥3):\nMechanism: Regressional / Extremal / Causal / Adversarial\nPrimary metric: | Constraint metric(s): | Audit: | Rotation: | Separation: | Gaming threshold:\nGoal measurement (independent): | Drift threshold: | Retirement criteria: | Owner:\n```\n\n*→ Method in Action: [Goodhart 1975 (M3) and Strathern 1997 (RAE)](examples/goodhart-1975-m3-and-strathern-1997-rae.md)*\n\n## Pack: Goodhart's Law Patterns\n\n| Domain | Common gaming | Defense |\n|---|---|---|\n| Sales quotas | Sandbagging, channel stuffing, end-of-quarter discounts | Multi-period averaging; quality metrics; clawback |\n| Hospital wait targets | Ambulance parking, patient reclassification | Outcome audits; paired metrics; randomized inspection |\n| Standardized testing | Teaching to test, curriculum narrowing | Sample-based assessment; multi-measure; reduce single-test stakes |\n| Algorithmic engagement | Clickbait, outrage, misinformation | Multi-objective optimization; quality + harm constraints |\n\n## Applying It Well\n\n- All metrics are proxies — narrower than the goal. Pre-commit to gap analysis before deployment.\n- Gaming is rational under measurement pressure. Fix the system, not the people.\n- Rotation and audit are the only durable defenses. Plan metric retirement at design time.\n\n*→ Primary sources: [references/sources.md](references/sources.md)*\n\n## Common Rationalizations\n\n| Fake move | Reality |\n|---|---|\n| [D] \"If you can't measure it, you can't manage it\" | Often false. Judgment, trust, and direct observation are also valid management tools. |\n| [D] \"Our metric is well-defined; it won't be gamed\" | Precision invites precise gaming. Basel II capital ratios were well-defined — extensively gamed. |\n| [D] \"Our people wouldn't game the metric\" | Goodhart's law is structural; individual virtue is insufficient in aggregate. |\n| [D] \"We just need a better metric\" | Often the issue is any single metric under pressure; fix is multi-metric + audit. |\n| [D] \"We've used this metric for years\" | Long use = more time for gaming to mature. Tenure is a warning, not an endorsement. |\n| *→ Add [O] entries here after each real use — paste the actual failure pattern* | *What went wrong and why* |\n\n**[D] = designed upfront | [O] = observed in real use. [O] entries are more valuable.**\n\n## Red Flags\n\n- Metric tied to high-stakes rewards or punishments\n- Metric \"improves\" without obvious improvement in the underlying goal\n- People being measured can already articulate ways to game it\n- Single metric is the primary evaluation tool, no audit or paired-constraint\n\n## Verification\n\n- [ ] Goal underlying the metric specifically named\n- [ ] Proxy gap explicitly described; ≥3 gaming vectors listed\n- [ ] Goodhart mechanism category identified\n- [ ] Countermeasure design (multi-metric, audit, rotation) in place\n- [ ] Independent goal measurement scheduled with owner and retirement date\n\n---\n\n*Part of **deciqAI Knowledge Skills** — 164 open-source thinking skills that make rigor executable for AI agents. The same skills power every deciqAI agent, which runs them autonomously to operate your company. **See it run → https://www.deciqai.com/c/goodharts-law** · ⭐ Star the repo → https://github.com/deciqAI/knowledge-skills · Contributions welcome.*\n\nFile v1.0.3:_meta.json\n\n{\n  \"ownerId\": \"kn754b8sk22s8c6gjxt02bftbn88q7ye\",\n  \"slug\": \"goodharts-law\",\n  \"version\": \"1.0.3\",\n  \"publishedAt\": 1783508641897\n}\n\nFile v1.0.3:references/sources.md\n\n# Sources — goodharts-law\n\n> *Primary sources for the [goodharts-law](../SKILL.md) skill.*\n\n- Goodhart, C. A. E. (1975). \"Problems of monetary management: The U.K. experience.\" Papers in Monetary Economics, Reserve Bank of Australia. Reprinted in *Monetary Theory and Practice: The U.K. Experience* (1984), Macmillan. The original.\n- Strathern, M. (1997). \"'Improving ratings': Audit in the British university system.\" *European Review*, 5(3), 305-321. The modern aphoristic formulation.\n- Campbell, D. T. (1979). \"Assessing the impact of planned social change.\" *Evaluation and Program Planning*, 2(1), 67-90. Independent formulation (\"Campbell's Law\").\n- Lucas, R. E. (1976). \"Econometric policy evaluation: A critique.\" *Carnegie-Rochester Conference Series on Public Policy*, 1(1), 19-46. The adjacent macro-econometric \"Lucas critique.\"\n- Manheim, D., & Garrabrant, S. (2018). \"Categorizing variants of Goodhart's Law.\" *arXiv:1803.04585*. The four-mechanism taxonomy.\n- Muller, J. Z. (2018). *The Tyranny of Metrics.* Princeton University Press. ISBN 978-0691174952. Comprehensive case-study survey.\n- Doerr, J. (2018). *Measure What Matters: How Google, Bono, and the Gates Foundation Rock the World with OKRs.* Portfolio. ISBN 978-0525536222.\n- Wells Fargo / Office of the Comptroller of the Currency (2016, 2018). Various enforcement actions and consent orders documenting the cross-selling case.\n\nFile v1.0.3:examples/goodhart-1975-m3-and-strathern-1997-rae.md\n\n# Method in Action: Goodhart 1975 (M3) and Strathern 1997 (RAE)\n\n> *Example for the [goodharts-law](../SKILL.md) skill.*\n\nThe empirical foundation has two key moments. The first is **Charles Goodhart's 1975 critique of UK monetary policy**, originally a conference paper for the Reserve Bank of Australia, later expanded in his 1984 book *Monetary Theory and Practice*.\n\nThe context: in the mid-1970s, the Bank of England (and many other central banks) had identified a robust historical correlation between growth in broad money supply (the M3 aggregate) and subsequent inflation. The natural policy implication was straightforward: target M3 growth at a level consistent with low inflation, and inflation would be controlled.\n\nGoodhart was skeptical. His objection was structural, not empirical:\n\n> \"It is not the case that there is some fixed and stable relationship between an observed monetary aggregate, such as M3, and other variables in the economic system. Rather, the relationships we observe statistically are the equilibrium outcomes of behavior by banks, depositors, and borrowers, each of whom is responding to a complex set of incentives. When the regulator targets M3 — and especially when the targeting carries policy weight that will affect interest rates and reserve requirements — these economic actors will reorganize their behavior to operate around the regulation. Liquid funds will be reclassified into categories that fall outside the M3 definition. The pre-targeting M3-inflation correlation will not survive the targeting. Any observed statistical regularity will tend to collapse once pressure is placed upon it for control purposes.\"\n>\n> — Goodhart (1975), as reprinted in Goodhart (1984), pp. 96-98.\n\nGoodhart's prediction was empirical and falsifiable. It came true within a few years: as the Bank of England's M3 targets bit, UK banks began creating money-substitutes that escaped the M3 definition (most notably, the rise of the eurodollar market and certain types of negotiable certificates of deposit). The M3-inflation correlation collapsed; the Bank quietly abandoned M3 targeting by the mid-1980s.\n\nThe principle's second formative moment came two decades later, in **Marilyn Strathern's 1997 ethnographic study of British universities** during the rise of the Research Assessment Exercise (RAE). Strathern, an anthropologist at Cambridge, observed the RAE's effect on academic behavior:\n\n> \"The Research Assessment Exercise sets out to evaluate the quality of research conducted in British universities. The exercise was introduced with the goal of identifying excellent research and directing funding toward it. The instrument: publication count, citation analysis, peer-rated quality. The exercise's effect on the academy has been precisely the inversion of its stated goal. Academics have shifted from publishing fewer, more substantial works to publishing more, smaller, shallower ones. They have shifted topic selection toward areas where rapid publication is easier. They have formed citation circles. The exercise has produced not better research, but research-shaped activity that scores well on the exercise. The principle is general: when a measure becomes a target, it ceases to be a good measure.\"\n>\n> — Strathern, M. (1997). \"'Improving ratings': Audit in the British university system.\" *European Review*, 5(3), pp. 308-309.\n\nThe aphoristic compression of \"when a measure becomes a target, it ceases to be a good measure\" became the popular form of the law. It is sometimes attributed to Goodhart directly; the precise wording is Strathern's, but she was articulating Goodhart's principle in academic-administration context.\n\nThe principle has been documented across an enormous number of empirical cases. A non-exhaustive selection:\n\n**Atlanta Public Schools cheating scandal (2009-2015).** Standardized test scores were used to evaluate schools under No Child Left Behind. Teachers and administrators systematically altered student answer sheets. The scandal involved 178 educators across 44 schools; 11 educators were criminally convicted. The pattern was not unique to Atlanta; subsequent investigations found similar gaming in Houston, Washington D.C., and elsewhere. Same mechanism as Goodhart's M3: pressure on the metric, behavioral response that decoupled metric from goal.\n\n**UK NHS Accident & Emergency 4-hour target (2002-2010s).** Hospitals were required to admit, transfer, or discharge 95% of A&E patients within 4 hours. Observed gaming: parking ambulances outside the A&E entrance (so the clock didn't start), reclassifying patients to delay the count start, discharging patients prematurely just before the 4-hour mark. Studies by the NHS itself documented these behaviors at scale. The metric \"improved\" while underlying care quality stagnated or in some cases declined.\n\n**Wells Fargo cross-selling scandal (2002-2016).** Employees were given quotas for cross-selling products (checking + savings + credit card + brokerage). Sales numbers improved dramatically. Investigation revealed millions of fraudulent accounts opened without customer consent. Wells Fargo paid $3+ billion in fines. The metric (cross-sells per customer) had been targeted; employees had gamed it; the underlying goal (customer wallet share earned through service quality) had degraded.\n\n**Soviet manufacturing under Gosplan (1930s-1980s).** Factory output measured in tons of nails → factories produced enormous single nails. Output measured in number of nails → factories produced tiny useless nails. Output measured in chandeliers → enormous unliftable chandeliers (the famous case of the Moscow lighting factory). The Soviet planning system was one long case study in adversarial Goodhart's law.\n\n**Academic h-index gaming (2010s-2020s).** Self-citation, citation rings, and predatory open-access journals exploded as the h-index became a primary academic evaluation metric. The 2020+ \"paper mill\" industry — services that sell fraudulent paper authorship to academics — is a direct Goodhart's-law product.\n\n**Algorithmic recommendation systems.** YouTube, TikTok, Instagram, Facebook all use engagement metrics (watch time, like rate, share rate, comment rate) to drive recommendations. Optimizing for engagement produces: clickbait, outrage content, misinformation amplification, and (well-documented) youth mental-health harm. The 2021 Facebook Files (whistleblower Frances Haugen) showed Facebook's internal research had documented these effects but the metric-target system continued to drive them.\n\nThe theoretical literature has been refined:\n\n**Manheim & Garrabrant (2018).** \"Categorizing variants of Goodhart's Law.\" *arXiv:1803.04585*. The taxonomy (Regressional / Extremal / Causal / Adversarial) referenced above. Originated in the AI alignment community as a framework for understanding why machine-learning systems optimized on a metric reliably produce dysfunction.\n\n**Hennessy & Goodhart (2023).** \"Goodhart's Law and machine learning: A structural perspective.\" *Working paper*. The contemporary AI extension.\n\n**Campbell, D. T. (1979).** \"Assessing the impact of planned social change.\" *Evaluation and Program Planning*, 2(1), 67-90. An independent formulation predating Goodhart's wider recognition, now called \"Campbell's law\": \"The more any quantitative social indicator is used for social decision-making, the more subject it will be to corruption pressures and the more apt it will be to distort and corrupt the social processes it is intended to monitor.\" Same content, different domain emphasis.\n\n**Lucas, R. E. (1976).** \"Econometric policy evaluation: A critique.\" *Carnegie-Rochester Conference Series on Public Policy*, 1(1), 19-46. The \"Lucas critique\" in macroeconomics — the recognition that econometric relationships estimated under one policy regime are not stable when the regime changes, because actors will respond to the new regime. Adjacent to Goodhart's law in mechanism.\n\nThe framework has reshaped operational design in multiple disciplines:\n\n**Modern bank regulation (Basel III, 2017+).** Capital and liquidity ratios are now paired with stress tests, leverage caps, and additional Tier 1 capital buffers — explicitly to constrain optimization of any single metric. The architecture is multi-metric, audit-augmented, and rotation-enabled (regulators update standards every few years specifically to keep ahead of gaming).\n\n**Modern KPI design.** Best-practice OKR frameworks (Doerr 2018) recommend mixing leading and lagging indicators, qualitative and quantitative metrics, and pairing each KR with an explicit audit / quality check. Single-metric reward systems are increasingly regarded as a known-bad design pattern.\n\n**Educational assessment.** Post-NCLB reforms have shifted toward sampling-based assessment, multi-measure school evaluation, and reduction of single-test stakes. Some states have explicitly abandoned standardized testing as an evaluation metric because the gaming costs exceeded the information benefits.\n\n**Algorithmic system design.** ML systems are increasingly designed with multi-objective optimization, adversarial robustness checks, and explicit decoupling of metric (for training) from outcome (for evaluation). The AI alignment field is essentially a deep study of Goodhart-style failure in increasingly powerful optimization systems.\n\n**Healthcare quality metrics.** UK NHS, U.S. CMS, and others have moved away from single-target measurement (e.g., A&E 4-hour) toward composite quality scores that include patient experience, clinical outcomes, and process metrics that can't be simultaneously gamed.\n\nThree operational lessons from Goodhart and Strathern:\n\n**First, all metrics are proxies. The proxy is always narrower than the goal.** The question is not \"is this a good metric?\" but \"what is the gap between the metric and the goal, and how will agents exploit that gap?\" Pre-committing to the gap analysis before deployment is the single highest-leverage step.\n\n**Second, gaming is rational behavior under measurement pressure, not character failure.** Blaming the people who game metrics is a category error — they are responding to incentives, often appropriately. The fix is the measurement system, not the people.\n\n**Third, rotation and audit are the only durable defenses.** No metric, however carefully designed, survives indefinite optimization pressure without gaming. The structural defense is to plan rotation (replace or refresh metrics before they're fully corrupted) and external audit (sample the underlying goal directly, periodically, to catch drift). Metrics are not assets; they are decaying assets.\n\nFile v1.0.3:skill-card.md\n\n## Description: <br>\nGuides agents through diagnosing Goodhart's Law risks in KPI, incentive, algorithmic, and audit systems and designing multi-metric, audit, and rotation countermeasures. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[deciqai](https://clawhub.ai/user/deciqai) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nEmployees, external users, developers, and operators use this skill to evaluate whether a metric is becoming a target, predict gaming behavior, classify Goodhart failure modes, and design more robust measurement systems. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Users may disclose confidential KPI failures, employee details, or proprietary operating patterns while applying the coaching prompts. <br>\nMitigation: Anonymize sensitive examples and avoid adding confidential operational details to shared skill files or public prompts. <br>\nRisk: Metric-design recommendations can be misleading if accepted without local context, stakeholder review, or independent outcome measurement. <br>\nMitigation: Review proposed gaming vectors and countermeasures with accountable owners, and pair the analysis with direct audits of the underlying goal. <br>\n\n\n## Reference(s): <br>\n- [Goodhart's Law primary sources](references/sources.md) <br>\n- [Method in Action: Goodhart 1975 and Strathern 1997](examples/goodhart-1975-m3-and-strathern-1997-rae.md) <br>\n- [ClawHub skill page](https://clawhub.ai/deciqai/skills/goodharts-law) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [Text, Markdown, Guidance] <br>\n**Output Format:** [Markdown guidance and structured analysis template] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Knowledge-only coaching output; no executable code or automatic system changes.] <br>\n\n## Skill Version(s): <br>\n1.0.3 (source: server release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.0.2: 5 files, 11088 bytes\n\nFiles: examples/goodhart-1975-m3-and-strathern-1997-rae.md (10675b), references/sources.md (1408b), skill-card.md (1884b), SKILL.md (7491b), _meta.json (132b)\n\nFile v1.0.2:SKILL.md\n\n---\nname: goodharts-law\ndescription: \"Activate when: our KPI is going up but the real outcome isn't improving; people seem to be gaming the metric; we're about to tie bonuses or promotions to a number; an algorithm is producing results nobody intended; a test or audit system is being designed.\n  Do NOT activate when: the metric IS the goal with no proxy gap; measurement is purely descriptive with zero stakes attached.\"\n---\n\n# Goodhart's Law\n\n## Overview\n\n**Goodhart's Law:** when a metric controls behavior, people optimize the metric rather than the underlying goal. Formulated by economist Charles Goodhart (1975) on UK monetary policy; sharpened by Marilyn Strathern (1997): *\"When a measure becomes a target, it ceases to be a good measure.\"* Four failure mechanisms (Manheim & Garrabrant 2018): **Regressional**, **Extremal**, **Causal**, **Adversarial**. Countermeasure is always multi-metric + audit + rotation.\n\nComposes with `feedback-loops`, `principal-agent`, `okr-goal-setting`, `survivorship-bias`.\n\n## When to Use\n\n- A KPI is being introduced or its weight is increasing in performance evaluation\n- A metric is \"improving\" without corresponding improvement in the underlying goal\n- People are visibly optimizing for a number rather than the work it was meant to track\n- Algorithmic optimization is producing outcomes the designers didn't intend\n- Resource allocation is driven by a single composite score or ranking\n\n**Not when:** metric and goal are identical; stakes too low for gaming; metric is purely descriptive with no reward/punishment; question is which metric to use, not whether the measurement-reward system is sound.\n\n## Coaching Novices (Adaptive Front Door)\n\n- **Engine mode:** user has a concrete metric or system → run The Process directly.\n- **Coach mode:** user is unfamiliar or has no concrete case → guide step by step.\n\nIn Coach mode, respond one step at a time. Each [WAIT] is a hard stop — output only that step's question, then stop.\n\n1. One-line: before relying on a metric to control behavior, predict how people will game it — choose the system that survives that prediction.\n2. Check fit: if the metric is purely descriptive (no reward attached), Goodhart's law doesn't apply yet.\n3. Elicit the specific metric and the underlying goal: what's being measured? What's the actual outcome you care about?\n> **[WAIT — do not advance until user responds]**\n4. One question at a time: proxy gap? How would a clever agent game this? Which Goodhart category? What countermeasure fits?\n> **[WAIT — do not advance until user responds]**\n5. Close: name the gaming-resistant design (multi-metric, audit, rotation, paired-constraint) + monitoring schedule.\n> **[WAIT — do not advance until user responds]**\n\n## The Process\n\n**Step 1 — State metric and goal:** metric being targeted / underlying goal / current proxy-goal correlation / who is measured / stakes.\n\n**Step 2 — Predict the gaming:** list ≥3 ways to game the metric with minimum effort on the goal. If you can't list 3, you haven't thought hard enough.\n\n**Step 3 — Categorize mechanism:**\n\n| Mechanism | Test |\n|---|---|\n| Regressional | Is there noise that optimization will push into? |\n| Extremal | Does metric-goal correlation break at extremes? |\n| Causal | Is the metric a symptom, not a cause? |\n| Adversarial | Will agents actively game with intelligence? |\n\n**Step 4 — Choose countermeasure:** Regressional → constrain range. Extremal → paired constraint metrics. Causal → closer-to-causation metric + direct audit. Adversarial → multi-metric + randomized audits + rotation.\n**Step 5 — Design the system:** primary metric / constraint metric(s) / audit mechanism (sampled direct goal observation) / rotation schedule / separation of measure-for-control from measure-for-diagnosis / gaming-detection threshold.\n**Step 6 — Schedule re-evaluation:** independent goal measurement (how/when/who) / drift threshold / retirement criteria / owner.\n\n## Output template\n\n```markdown\n# Goodhart-Robust Design: <metric>\nMetric: | Underlying goal: | Correlation: | Who measured: | Stakes:\nGaming vectors (≥3):\nMechanism: Regressional / Extremal / Causal / Adversarial\nPrimary metric: | Constraint metric(s): | Audit: | Rotation: | Separation: | Gaming threshold:\nGoal measurement (independent): | Drift threshold: | Retirement criteria: | Owner:\n```\n\n*→ Method in Action: [Goodhart 1975 (M3) and Strathern 1997 (RAE)](examples/goodhart-1975-m3-and-strathern-1997-rae.md)*\n\n## Pack: Goodhart's Law Patterns\n\n| Domain | Common gaming | Defense |\n|---|---|---|\n| Sales quotas | Sandbagging, channel stuffing, end-of-quarter discounts | Multi-period averaging; quality metrics; clawback |\n| Hospital wait targets | Ambulance parking, patient reclassification | Outcome audits; paired metrics; randomized inspection |\n| Standardized testing | Teaching to test, curriculum narrowing | Sample-based assessment; multi-measure; reduce single-test stakes |\n| Algorithmic engagement | Clickbait, outrage, misinformation | Multi-objective optimization; quality + harm constraints |\n\n## Applying It Well\n\n- All metrics are proxies — narrower than the goal. Pre-commit to gap analysis before deployment.\n- Gaming is rational under measurement pressure. Fix the system, not the people.\n- Rotation and audit are the only durable defenses. Plan metric retirement at design time.\n\n*→ Primary sources: [references/sources.md](references/sources.md)*\n\n## Common Rationalizations\n\n| Fake move | Reality |\n|---|---|\n| [D] \"If you can't measure it, you can't manage it\" | Often false. Judgment, trust, and direct observation are also valid management tools. |\n| [D] \"Our metric is well-defined; it won't be gamed\" | Precision invites precise gaming. Basel II capital ratios were well-defined — extensively gamed. |\n| [D] \"Our people wouldn't game the metric\" | Goodhart's law is structural; individual virtue is insufficient in aggregate. |\n| [D] \"We just need a better metric\" | Often the issue is any single metric under pressure; fix is multi-metric + audit. |\n| [D] \"We've used this metric for years\" | Long use = more time for gaming to mature. Tenure is a warning, not an endorsement. |\n| *→ Add [O] entries here after each real use — paste the actual failure pattern* | *What went wrong and why* |\n\n**[D] = designed upfront | [O] = observed in real use. [O] entries are more valuable.**\n\n## Red Flags\n\n- Metric tied to high-stakes rewards or punishments\n- Metric \"improves\" without obvious improvement in the underlying goal\n- People being measured can already articulate ways to game it\n- Single metric is the primary evaluation tool, no audit or paired-constraint\n\n## Verification\n\n- [ ] Goal underlying the metric specifically named\n- [ ] Proxy gap explicitly described; ≥3 gaming vectors listed\n- [ ] Goodhart mechanism category identified\n- [ ] Countermeasure design (multi-metric, audit, rotation) in place\n- [ ] Independent goal measurement scheduled with owner and retirement date\n\n---\n\n*Part of **deciqAI Knowledge Skills** — 163 open-source thinking skills that make rigor executable for AI agents. The same skills power every deciqAI agent, which runs them autonomously to operate your company. **See it run → https://www.deciqai.com/skills/goodharts-law?utm_source=clawhub&utm_medium=marketplace&utm_campaign=knowledge-skills&utm_content=goodharts-law** · ⭐ Star the repo → https://github.com/deciqAI/knowledge-skills · Contributions welcome.*\n\nFile v1.0.2:_meta.json\n\n{\n  \"ownerId\": \"kn754b8sk22s8c6gjxt02bftbn88q7ye\",\n  \"slug\": \"goodharts-law\",\n  \"version\": \"1.0.2\",\n  \"publishedAt\": 1783471759378\n}\n\nFile v1.0.2:references/sources.md\n\n# Sources — goodharts-law\n\n> *Primary sources for the [goodharts-law](../SKILL.md) skill.*\n\n- Goodhart, C. A. E. (1975). \"Problems of monetary management: The U.K. experience.\" Papers in Monetary Economics, Reserve Bank of Australia. Reprinted in *Monetary Theory and Practice: The U.K. Experience* (1984), Macmillan. The original.\n- Strathern, M. (1997). \"'Improving ratings': Audit in the British university system.\" *European Review*, 5(3), 305-321. The modern aphoristic formulation.\n- Campbell, D. T. (1979). \"Assessing the impact of planned social change.\" *Evaluation and Program Planning*, 2(1), 67-90. Independent formulation (\"Campbell's Law\").\n- Lucas, R. E. (1976). \"Econometric policy evaluation: A critique.\" *Carnegie-Rochester Conference Series on Public Policy*, 1(1), 19-46. The adjacent macro-econometric \"Lucas critique.\"\n- Manheim, D., & Garrabrant, S. (2018). \"Categorizing variants of Goodhart's Law.\" *arXiv:1803.04585*. The four-mechanism taxonomy.\n- Muller, J. Z. (2018). *The Tyranny of Metrics.* Princeton University Press. ISBN 978-0691174952. Comprehensive case-study survey.\n- Doerr, J. (2018). *Measure What Matters: How Google, Bono, and the Gates Foundation Rock the World with OKRs.* Portfolio. ISBN 978-0525536222.\n- Wells Fargo / Office of the Comptroller of the Currency (2016, 2018). Various enforcement actions and consent orders documenting the cross-selling case.\n\nFile v1.0.2:examples/goodhart-1975-m3-and-strathern-1997-rae.md\n\n# Method in Action: Goodhart 1975 (M3) and Strathern 1997 (RAE)\n\n> *Example for the [goodharts-law](../SKILL.md) skill.*\n\nThe empirical foundation has two key moments. The first is **Charles Goodhart's 1975 critique of UK monetary policy**, originally a conference paper for the Reserve Bank of Australia, later expanded in his 1984 book *Monetary Theory and Practice*.\n\nThe context: in the mid-1970s, the Bank of England (and many other central banks) had identified a robust historical correlation between growth in broad money supply (the M3 aggregate) and subsequent inflation. The natural policy implication was straightforward: target M3 growth at a level consistent with low inflation, and inflation would be controlled.\n\nGoodhart was skeptical. His objection was structural, not empirical:\n\n> \"It is not the case that there is some fixed and stable relationship between an observed monetary aggregate, such as M3, and other variables in the economic system. Rather, the relationships we observe statistically are the equilibrium outcomes of behavior by banks, depositors, and borrowers, each of whom is responding to a complex set of incentives. When the regulator targets M3 — and especially when the targeting carries policy weight that will affect interest rates and reserve requirements — these economic actors will reorganize their behavior to operate around the regulation. Liquid funds will be reclassified into categories that fall outside the M3 definition. The pre-targeting M3-inflation correlation will not survive the targeting. Any observed statistical regularity will tend to collapse once pressure is placed upon it for control purposes.\"\n>\n> — Goodhart (1975), as reprinted in Goodhart (1984), pp. 96-98.\n\nGoodhart's prediction was empirical and falsifiable. It came true within a few years: as the Bank of England's M3 targets bit, UK banks began creating money-substitutes that escaped the M3 definition (most notably, the rise of the eurodollar market and certain types of negotiable certificates of deposit). The M3-inflation correlation collapsed; the Bank quietly abandoned M3 targeting by the mid-1980s.\n\nThe principle's second formative moment came two decades later, in **Marilyn Strathern's 1997 ethnographic study of British universities** during the rise of the Research Assessment Exercise (RAE). Strathern, an anthropologist at Cambridge, observed the RAE's effect on academic behavior:\n\n> \"The Research Assessment Exercise sets out to evaluate the quality of research conducted in British universities. The exercise was introduced with the goal of identifying excellent research and directing funding toward it. The instrument: publication count, citation analysis, peer-rated quality. The exercise's effect on the academy has been precisely the inversion of its stated goal. Academics have shifted from publishing fewer, more substantial works to publishing more, smaller, shallower ones. They have shifted topic selection toward areas where rapid publication is easier. They have formed citation circles. The exercise has produced not better research, but research-shaped activity that scores well on the exercise. The principle is general: when a measure becomes a target, it ceases to be a good measure.\"\n>\n> — Strathern, M. (1997). \"'Improving ratings': Audit in the British university system.\" *European Review*, 5(3), pp. 308-309.\n\nThe aphoristic compression of \"when a measure becomes a target, it ceases to be a good measure\" became the popular form of the law. It is sometimes attributed to Goodhart directly; the precise wording is Strathern's, but she was articulating Goodhart's principle in academic-administration context.\n\nThe principle has been documented across an enormous number of empirical cases. A non-exhaustive selection:\n\n**Atlanta Public Schools cheating scandal (2009-2015).** Standardized test scores were used to evaluate schools under No Child Left Behind. Teachers and administrators systematically altered student answer sheets. The scandal involved 178 educators across 44 schools; 11 educators were criminally convicted. The pattern was not unique to Atlanta; subsequent investigations found similar gaming in Houston, Washington D.C., and elsewhere. Same mechanism as Goodhart's M3: pressure on the metric, behavioral response that decoupled metric from goal.\n\n**UK NHS Accident & Emergency 4-hour target (2002-2010s).** Hospitals were required to admit, transfer, or discharge 95% of A&E patients within 4 hours. Observed gaming: parking ambulances outside the A&E entrance (so the clock didn't start), reclassifying patients to delay the count start, discharging patients prematurely just before the 4-hour mark. Studies by the NHS itself documented these behaviors at scale. The metric \"improved\" while underlying care quality stagnated or in some cases declined.\n\n**Wells Fargo cross-selling scandal (2002-2016).** Employees were given quotas for cross-selling products (checking + savings + credit card + brokerage). Sales numbers improved dramatically. Investigation revealed millions of fraudulent accounts opened without customer consent. Wells Fargo paid $3+ billion in fines. The metric (cross-sells per customer) had been targeted; employees had gamed it; the underlying goal (customer wallet share earned through service quality) had degraded.\n\n**Soviet manufacturing under Gosplan (1930s-1980s).** Factory output measured in tons of nails → factories produced enormous single nails. Output measured in number of nails → factories produced tiny useless nails. Output measured in chandeliers → enormous unliftable chandeliers (the famous case of the Moscow lighting factory). The Soviet planning system was one long case study in adversarial Goodhart's law.\n\n**Academic h-index gaming (2010s-2020s).** Self-citation, citation rings, and predatory open-access journals exploded as the h-index became a primary academic evaluation metric. The 2020+ \"paper mill\" industry — services that sell fraudulent paper authorship to academics — is a direct Goodhart's-law product.\n\n**Algorithmic recommendation systems.** YouTube, TikTok, Instagram, Facebook all use engagement metrics (watch time, like rate, share rate, comment rate) to drive recommendations. Optimizing for engagement produces: clickbait, outrage content, misinformation amplification, and (well-documented) youth mental-health harm. The 2021 Facebook Files (whistleblower Frances Haugen) showed Facebook's internal research had documented these effects but the metric-target system continued to drive them.\n\nThe theoretical literature has been refined:\n\n**Manheim & Garrabrant (2018).** \"Categorizing variants of Goodhart's Law.\" *arXiv:1803.04585*. The taxonomy (Regressional / Extremal / Causal / Adversarial) referenced above. Originated in the AI alignment community as a framework for understanding why machine-learning systems optimized on a metric reliably produce dysfunction.\n\n**Hennessy & Goodhart (2023).** \"Goodhart's Law and machine learning: A structural perspective.\" *Working paper*. The contemporary AI extension.\n\n**Campbell, D. T. (1979).** \"Assessing the impact of planned social change.\" *Evaluation and Program Planning*, 2(1), 67-90. An independent formulation predating Goodhart's wider recognition, now called \"Campbell's law\": \"The more any quantitative social indicator is used for social decision-making, the more subject it will be to corruption pressures and the more apt it will be to distort and corrupt the social processes it is intended to monitor.\" Same content, different domain emphasis.\n\n**Lucas, R. E. (1976).** \"Econometric policy evaluation: A critique.\" *Carnegie-Rochester Conference Series on Public Policy*, 1(1), 19-46. The \"Lucas critique\" in macroeconomics — the recognition that econometric relationships estimated under one policy regime are not stable when the regime changes, because actors will respond to the new regime. Adjacent to Goodhart's law in mechanism.\n\nThe framework has reshaped operational design in multiple disciplines:\n\n**Modern bank regulation (Basel III, 2017+).** Capital and liquidity ratios are now paired with stress tests, leverage caps, and additional Tier 1 capital buffers — explicitly to constrain optimization of any single metric. The architecture is multi-metric, audit-augmented, and rotation-enabled (regulators update standards every few years specifically to keep ahead of gaming).\n\n**Modern KPI design.** Best-practice OKR frameworks (Doerr 2018) recommend mixing leading and lagging indicators, qualitative and quantitative metrics, and pairing each KR with an explicit audit / quality check. Single-metric reward systems are increasingly regarded as a known-bad design pattern.\n\n**Educational assessment.** Post-NCLB reforms have shifted toward sampling-based assessment, multi-measure school evaluation, and reduction of single-test stakes. Some states have explicitly abandoned standardized testing as an evaluation metric because the gaming costs exceeded the information benefits.\n\n**Algorithmic system design.** ML systems are increasingly designed with multi-objective optimization, adversarial robustness checks, and explicit decoupling of metric (for training) from outcome (for evaluation). The AI alignment field is essentially a deep study of Goodhart-style failure in increasingly powerful optimization systems.\n\n**Healthcare quality metrics.** UK NHS, U.S. CMS, and others have moved away from single-target measurement (e.g., A&E 4-hour) toward composite quality scores that include patient experience, clinical outcomes, and process metrics that can't be simultaneously gamed.\n\nThree operational lessons from Goodhart and Strathern:\n\n**First, all metrics are proxies. The proxy is always narrower than the goal.** The question is not \"is this a good metric?\" but \"what is the gap between the metric and the goal, and how will agents exploit that gap?\" Pre-committing to the gap analysis before deployment is the single highest-leverage step.\n\n**Second, gaming is rational behavior under measurement pressure, not character failure.** Blaming the people who game metrics is a category error — they are responding to incentives, often appropriately. The fix is the measurement system, not the people.\n\n**Third, rotation and audit are the only durable defenses.** No metric, however carefully designed, survives indefinite optimization pressure without gaming. The structural defense is to plan rotation (replace or refresh metrics before they're fully corrupted) and external audit (sample the underlying goal directly, periodically, to catch drift). Metrics are not assets; they are decaying assets.\n\nFile v1.0.2:skill-card.md\n\n## Description: <br>\nGuides agents through analyzing incentive metrics for Goodhart's Law failure modes and designing multi-metric audits, rotations, and constraints. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[deciqai](https://clawhub.ai/user/deciqai) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nEmployees, external advisors, and developers use this skill to evaluate KPIs, reward systems, audits, rankings, or algorithms where a proxy metric may be gamed or decouple from the real goal. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Historical examples and source attributions may be treated as authoritative in high-stakes incentive or policy decisions. <br>\nMitigation: Verify cited sources and apply domain review before using the skill's recommendations for compensation, audit, regulatory, healthcare, finance, or safety-critical decisions. <br>\n\n\n## Reference(s): <br>\n- [Sources - goodharts-law](references/sources.md) <br>\n- [Goodhart 1975 (M3) and Strathern 1997 (RAE)](examples/goodhart-1975-m3-and-strathern-1997-rae.md) <br>\n- [Goodhart's Law on ClawHub](https://clawhub.ai/deciqai/skills/goodharts-law) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [Text, Markdown, Guidance] <br>\n**Output Format:** [Markdown checklist and structured design template] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Text-only; may pause for user input in coach mode.] <br>\n\n## Skill Version(s): <br>\n1.0.2 (source: server evidence release.version) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.0.1: 5 files, 11171 bytes\n\nFiles: examples/goodhart-1975-m3-and-strathern-1997-rae.md (10675b), references/sources.md (1408b), skill-card.md (2282b), SKILL.md (7363b), _meta.json (132b)\n\nFile v1.0.1:SKILL.md\n\n---\nname: goodharts-law\ndescription: \"Activate when: our KPI is going up but the real outcome isn't improving; people seem to be gaming the metric; we're about to tie bonuses or promotions to a number; an algorithm is producing results nobody intended; a test or audit system is being designed.\n  Do NOT activate when: the metric IS the goal with no proxy gap; measurement is purely descriptive with zero stakes attached.\"\n---\n\n# Goodhart's Law\n\n## Overview\n\n**Goodhart's Law:** when a metric controls behavior, people optimize the metric rather than the underlying goal. Formulated by economist Charles Goodhart (1975) on UK monetary policy; sharpened by Marilyn Strathern (1997): *\"When a measure becomes a target, it ceases to be a good measure.\"* Four failure mechanisms (Manheim & Garrabrant 2018): **Regressional**, **Extremal**, **Causal**, **Adversarial**. Countermeasure is always multi-metric + audit + rotation.\n\nComposes with [`feedback-loops`](../feedback-loops/SKILL.md), [`principal-agent`](../principal-agent/SKILL.md), [`okr-goal-setting`](../okr-goal-setting/SKILL.md), [`survivorship-bias`](../survivorship-bias/SKILL.md).\n\n## When to Use\n\n- A KPI is being introduced or its weight is increasing in performance evaluation\n- A metric is \"improving\" without corresponding improvement in the underlying goal\n- People are visibly optimizing for a number rather than the work it was meant to track\n- Algorithmic optimization is producing outcomes the designers didn't intend\n- Resource allocation is driven by a single composite score or ranking\n\n**Not when:** metric and goal are identical; stakes too low for gaming; metric is purely descriptive with no reward/punishment; question is which metric to use, not whether the measurement-reward system is sound.\n\n## Coaching Novices (Adaptive Front Door)\n\n- **Engine mode:** user has a concrete metric or system → run The Process directly.\n- **Coach mode:** user is unfamiliar or has no concrete case → guide step by step.\n\nIn Coach mode, respond one step at a time. Each [WAIT] is a hard stop — output only that step's question, then stop.\n\n1. One-line: before relying on a metric to control behavior, predict how people will game it — choose the system that survives that prediction.\n2. Check fit: if the metric is purely descriptive (no reward attached), Goodhart's law doesn't apply yet.\n3. Elicit the specific metric and the underlying goal: what's being measured? What's the actual outcome you care about?\n> **[WAIT — do not advance until user responds]**\n4. One question at a time: proxy gap? How would a clever agent game this? Which Goodhart category? What countermeasure fits?\n> **[WAIT — do not advance until user responds]**\n5. Close: name the gaming-resistant design (multi-metric, audit, rotation, paired-constraint) + monitoring schedule.\n> **[WAIT — do not advance until user responds]**\n\n## The Process\n\n**Step 1 — State metric and goal:** metric being targeted / underlying goal / current proxy-goal correlation / who is measured / stakes.\n\n**Step 2 — Predict the gaming:** list ≥3 ways to game the metric with minimum effort on the goal. If you can't list 3, you haven't thought hard enough.\n\n**Step 3 — Categorize mechanism:**\n\n| Mechanism | Test |\n|---|---|\n| Regressional | Is there noise that optimization will push into? |\n| Extremal | Does metric-goal correlation break at extremes? |\n| Causal | Is the metric a symptom, not a cause? |\n| Adversarial | Will agents actively game with intelligence? |\n\n**Step 4 — Choose countermeasure:** Regressional → constrain range. Extremal → paired constraint metrics. Causal → closer-to-causation metric + direct audit. Adversarial → multi-metric + randomized audits + rotation.\n**Step 5 — Design the system:** primary metric / constraint metric(s) / audit mechanism (sampled direct goal observation) / rotation schedule / separation of measure-for-control from measure-for-diagnosis / gaming-detection threshold.\n**Step 6 — Schedule re-evaluation:** independent goal measurement (how/when/who) / drift threshold / retirement criteria / owner.\n\n## Output template\n\n```markdown\n# Goodhart-Robust Design: <metric>\nMetric: | Underlying goal: | Correlation: | Who measured: | Stakes:\nGaming vectors (≥3):\nMechanism: Regressional / Extremal / Causal / Adversarial\nPrimary metric: | Constraint metric(s): | Audit: | Rotation: | Separation: | Gaming threshold:\nGoal measurement (independent): | Drift threshold: | Retirement criteria: | Owner:\n```\n\n*→ Method in Action: [Goodhart 1975 (M3) and Strathern 1997 (RAE)](examples/goodhart-1975-m3-and-strathern-1997-rae.md)*\n\n## Pack: Goodhart's Law Patterns\n\n| Domain | Common gaming | Defense |\n|---|---|---|\n| Sales quotas | Sandbagging, channel stuffing, end-of-quarter discounts | Multi-period averaging; quality metrics; clawback |\n| Hospital wait targets | Ambulance parking, patient reclassification | Outcome audits; paired metrics; randomized inspection |\n| Standardized testing | Teaching to test, curriculum narrowing | Sample-based assessment; multi-measure; reduce single-test stakes |\n| Algorithmic engagement | Clickbait, outrage, misinformation | Multi-objective optimization; quality + harm constraints |\n\n## Applying It Well\n\n- All metrics are proxies — narrower than the goal. Pre-commit to gap analysis before deployment.\n- Gaming is rational under measurement pressure. Fix the system, not the people.\n- Rotation and audit are the only durable defenses. Plan metric retirement at design time.\n\n*→ Primary sources: [references/sources.md](references/sources.md)*\n\n## Common Rationalizations\n\n| Fake move | Reality |\n|---|---|\n| [D] \"If you can't measure it, you can't manage it\" | Often false. Judgment, trust, and direct observation are also valid management tools. |\n| [D] \"Our metric is well-defined; it won't be gamed\" | Precision invites precise gaming. Basel II capital ratios were well-defined — extensively gamed. |\n| [D] \"Our people wouldn't game the metric\" | Goodhart's law is structural; individual virtue is insufficient in aggregate. |\n| [D] \"We just need a better metric\" | Often the issue is any single metric under pressure; fix is multi-metric + audit. |\n| [D] \"We've used this metric for years\" | Long use = more time for gaming to mature. Tenure is a warning, not an endorsement. |\n| *→ Add [O] entries here after each real use — paste the actual failure pattern* | *What went wrong and why* |\n\n**[D] = designed upfront | [O] = observed in real use. [O] entries are more valuable.**\n\n## Red Flags\n\n- Metric tied to high-stakes rewards or punishments\n- Metric \"improves\" without obvious improvement in the underlying goal\n- People being measured can already articulate ways to game it\n- Single metric is the primary evaluation tool, no audit or paired-constraint\n\n## Verification\n\n- [ ] Goal underlying the metric specifically named\n- [ ] Proxy gap explicitly described; ≥3 gaming vectors listed\n- [ ] Goodhart mechanism category identified\n- [ ] Countermeasure design (multi-metric, audit, rotation) in place\n- [ ] Independent goal measurement scheduled with owner and retirement date\n\n---\n\n*Part of **deciqAI Knowledge Skills** — open-source thinking skills that make rigor executable for AI agents. Built by deciqAI · https://deciqai.com · Contributions welcome — see the template at the repo root.*\n\nFile v1.0.1:_meta.json\n\n{\n  \"ownerId\": \"kn754b8sk22s8c6gjxt02bftbn88q7ye\",\n  \"slug\": \"goodharts-law\",\n  \"version\": \"1.0.1\",\n  \"publishedAt\": 1783456396613\n}\n\nFile v1.0.1:references/sources.md\n\n# Sources — goodharts-law\n\n> *Primary sources for the [goodharts-law](../SKILL.md) skill.*\n\n- Goodhart, C. A. E. (1975). \"Problems of monetary management: The U.K. experience.\" Papers in Monetary Economics, Reserve Bank of Australia. Reprinted in *Monetary Theory and Practice: The U.K. Experience* (1984), Macmillan. The original.\n- Strathern, M. (1997). \"'Improving ratings': Audit in the British university system.\" *European Review*, 5(3), 305-321. The modern aphoristic formulation.\n- Campbell, D. T. (1979). \"Assessing the impact of planned social change.\" *Evaluation and Program Planning*, 2(1), 67-90. Independent formulation (\"Campbell's Law\").\n- Lucas, R. E. (1976). \"Econometric policy evaluation: A critique.\" *Carnegie-Rochester Conference Series on Public Policy*, 1(1), 19-46. The adjacent macro-econometric \"Lucas critique.\"\n- Manheim, D., & Garrabrant, S. (2018). \"Categorizing variants of Goodhart's Law.\" *arXiv:1803.04585*. The four-mechanism taxonomy.\n- Muller, J. Z. (2018). *The Tyranny of Metrics.* Princeton University Press. ISBN 978-0691174952. Comprehensive case-study survey.\n- Doerr, J. (2018). *Measure What Matters: How Google, Bono, and the Gates Foundation Rock the World with OKRs.* Portfolio. ISBN 978-0525536222.\n- Wells Fargo / Office of the Comptroller of the Currency (2016, 2018). Various enforcement actions and consent orders documenting the cross-selling case.\n\nFile v1.0.1:examples/goodhart-1975-m3-and-strathern-1997-rae.md\n\n# Method in Action: Goodhart 1975 (M3) and Strathern 1997 (RAE)\n\n> *Example for the [goodharts-law](../SKILL.md) skill.*\n\nThe empirical foundation has two key moments. The first is **Charles Goodhart's 1975 critique of UK monetary policy**, originally a conference paper for the Reserve Bank of Australia, later expanded in his 1984 book *Monetary Theory and Practice*.\n\nThe context: in the mid-1970s, the Bank of England (and many other central banks) had identified a robust historical correlation between growth in broad money supply (the M3 aggregate) and subsequent inflation. The natural policy implication was straightforward: target M3 growth at a level consistent with low inflation, and inflation would be controlled.\n\nGoodhart was skeptical. His objection was structural, not empirical:\n\n> \"It is not the case that there is some fixed and stable relationship between an observed monetary aggregate, such as M3, and other variables in the economic system. Rather, the relationships we observe statistically are the equilibrium outcomes of behavior by banks, depositors, and borrowers, each of whom is responding to a complex set of incentives. When the regulator targets M3 — and especially when the targeting carries policy weight that will affect interest rates and reserve requirements — these economic actors will reorganize their behavior to operate around the regulation. Liquid funds will be reclassified into categories that fall outside the M3 definition. The pre-targeting M3-inflation correlation will not survive the targeting. Any observed statistical regularity will tend to collapse once pressure is placed upon it for control purposes.\"\n>\n> — Goodhart (1975), as reprinted in Goodhart (1984), pp. 96-98.\n\nGoodhart's prediction was empirical and falsifiable. It came true within a few years: as the Bank of England's M3 targets bit, UK banks began creating money-substitutes that escaped the M3 definition (most notably, the rise of the eurodollar market and certain types of negotiable certificates of deposit). The M3-inflation correlation collapsed; the Bank quietly abandoned M3 targeting by the mid-1980s.\n\nThe principle's second formative moment came two decades later, in **Marilyn Strathern's 1997 ethnographic study of British universities** during the rise of the Research Assessment Exercise (RAE). Strathern, an anthropologist at Cambridge, observed the RAE's effect on academic behavior:\n\n> \"The Research Assessment Exercise sets out to evaluate the quality of research conducted in British universities. The exercise was introduced with the goal of identifying excellent research and directing funding toward it. The instrument: publication count, citation analysis, peer-rated quality. The exercise's effect on the academy has been precisely the inversion of its stated goal. Academics have shifted from publishing fewer, more substantial works to publishing more, smaller, shallower ones. They have shifted topic selection toward areas where rapid publication is easier. They have formed citation circles. The exercise has produced not better research, but research-shaped activity that scores well on the exercise. The principle is general: when a measure becomes a target, it ceases to be a good measure.\"\n>\n> — Strathern, M. (1997). \"'Improving ratings': Audit in the British university system.\" *European Review*, 5(3), pp. 308-309.\n\nThe aphoristic compression of \"when a measure becomes a target, it ceases to be a good measure\" became the popular form of the law. It is sometimes attributed to Goodhart directly; the precise wording is Strathern's, but she was articulating Goodhart's principle in academic-administration context.\n\nThe principle has been documented across an enormous number of empirical cases. A non-exhaustive selection:\n\n**Atlanta Public Schools cheating scandal (2009-2015).** Standardized test scores were used to evaluate schools under No Child Left Behind. Teachers and administrators systematically altered student answer sheets. The scandal involved 178 educators across 44 schools; 11 educators were criminally convicted. The pattern was not unique to Atlanta; subsequent investigations found similar gaming in Houston, Washington D.C., and elsewhere. Same mechanism as Goodhart's M3: pressure on the metric, behavioral response that decoupled metric from goal.\n\n**UK NHS Accident & Emergency 4-hour target (2002-2010s).** Hospitals were required to admit, transfer, or discharge 95% of A&E patients within 4 hours. Observed gaming: parking ambulances outside the A&E entrance (so the clock didn't start), reclassifying patients to delay the count start, discharging patients prematurely just before the 4-hour mark. Studies by the NHS itself documented these behaviors at scale. The metric \"improved\" while underlying care quality stagnated or in some cases declined.\n\n**Wells Fargo cross-selling scandal (2002-2016).** Employees were given quotas for cross-selling products (checking + savings + credit card + brokerage). Sales numbers improved dramatically. Investigation revealed millions of fraudulent accounts opened without customer consent. Wells Fargo paid $3+ billion in fines. The metric (cross-sells per customer) had been targeted; employees had gamed it; the underlying goal (customer wallet share earned through service quality) had degraded.\n\n**Soviet manufacturing under Gosplan (1930s-1980s).** Factory output measured in tons of nails → factories produced enormous single nails. Output measured in number of nails → factories produced tiny useless nails. Output measured in chandeliers → enormous unliftable chandeliers (the famous case of the Moscow lighting factory). The Soviet planning system was one long case study in adversarial Goodhart's law.\n\n**Academic h-index gaming (2010s-2020s).** Self-citation, citation rings, and predatory open-access journals exploded as the h-index became a primary academic evaluation metric. The 2020+ \"paper mill\" industry — services that sell fraudulent paper authorship to academics — is a direct Goodhart's-law product.\n\n**Algorithmic recommendation systems.** YouTube, TikTok, Instagram, Facebook all use engagement metrics (watch time, like rate, share rate, comment rate) to drive recommendations. Optimizing for engagement produces: clickbait, outrage content, misinformation amplification, and (well-documented) youth mental-health harm. The 2021 Facebook Files (whistleblower Frances Haugen) showed Facebook's internal research had documented these effects but the metric-target system continued to drive them.\n\nThe theoretical literature has been refined:\n\n**Manheim & Garrabrant (2018).** \"Categorizing variants of Goodhart's Law.\" *arXiv:1803.04585*. The taxonomy (Regressional / Extremal / Causal / Adversarial) referenced above. Originated in the AI alignment community as a framework for understanding why machine-learning systems optimized on a metric reliably produce dysfunction.\n\n**Hennessy & Goodhart (2023).** \"Goodhart's Law and machine learning: A structural perspective.\" *Working paper*. The contemporary AI extension.\n\n**Campbell, D. T. (1979).** \"Assessing the impact of planned social change.\" *Evaluation and Program Planning*, 2(1), 67-90. An independent formulation predating Goodhart's wider recognition, now called \"Campbell's law\": \"The more any quantitative social indicator is used for social decision-making, the more subject it will be to corruption pressures and the more apt it will be to distort and corrupt the social processes it is intended to monitor.\" Same content, different domain emphasis.\n\n**Lucas, R. E. (1976).** \"Econometric policy evaluation: A critique.\" *Carnegie-Rochester Conference Series on Public Policy*, 1(1), 19-46. The \"Lucas critique\" in macroeconomics — the recognition that econometric relationships estimated under one policy regime are not stable when the regime changes, because actors will respond to the new regime. Adjacent to Goodhart's law in mechanism.\n\nThe framework has reshaped operational design in multiple disciplines:\n\n**Modern bank regulation (Basel III, 2017+).** Capital and liquidity ratios are now paired with stress tests, leverage caps, and additional Tier 1 capital buffers — explicitly to constrain optimization of any single metric. The architecture is multi-metric, audit-augmented, and rotation-enabled (regulators update standards every few years specifically to keep ahead of gaming).\n\n**Modern KPI design.** Best-practice OKR frameworks (Doerr 2018) recommend mixing leading and lagging indicators, qualitative and quantitative metrics, and pairing each KR with an explicit audit / quality check. Single-metric reward systems are increasingly regarded as a known-bad design pattern.\n\n**Educational assessment.** Post-NCLB reforms have shifted toward sampling-based assessment, multi-measure school evaluation, and reduction of single-test stakes. Some states have explicitly abandoned standardized testing as an evaluation metric because the gaming costs exceeded the information benefits.\n\n**Algorithmic system design.** ML systems are increasingly designed with multi-objective optimization, adversarial robustness checks, and explicit decoupling of metric (for training) from outcome (for evaluation). The AI alignment field is essentially a deep study of Goodhart-style failure in increasingly powerful optimization systems.\n\n**Healthcare quality metrics.** UK NHS, U.S. CMS, and others have moved away from single-target measurement (e.g., A&E 4-hour) toward composite quality scores that include patient experience, clinical outcomes, and process metrics that can't be simultaneously gamed.\n\nThree operational lessons from Goodhart and Strathern:\n\n**First, all metrics are proxies. The proxy is always narrower than the goal.** The question is not \"is this a good metric?\" but \"what is the gap between the metric and the goal, and how will agents exploit that gap?\" Pre-committing to the gap analysis before deployment is the single highest-leverage step.\n\n**Second, gaming is rational behavior under measurement pressure, not character failure.** Blaming the people who game metrics is a category error — they are responding to incentives, often appropriately. The fix is the measurement system, not the people.\n\n**Third, rotation and audit are the only durable defenses.** No metric, however carefully designed, survives indefinite optimization pressure without gaming. The structural defense is to plan rotation (replace or refresh metrics before they're fully corrupted) and external audit (sample the underlying goal directly, periodically, to catch drift). Metrics are not assets; they are decaying assets.\n\nFile v1.0.1:skill-card.md\n\n## Description: <br>\nHelps agents evaluate KPI and metric-driven systems for Goodhart's Law failure modes and design multi-metric, audited, rotation-aware countermeasures. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[deciqai](https://clawhub.ai/user/deciqai) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nExternal users and developers use this skill to test whether a metric is becoming a target, predict gaming vectors, classify the failure mechanism, and design audits, constraint metrics, rotation schedules, and retirement criteria. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill can influence metric, incentive, and evaluation design, so weak input context could lead to misleading KPI recommendations. <br>\nMitigation: Review outputs with domain owners and apply the skill's independent goal measurement, audit, drift threshold, and retirement checks before adopting recommendations. <br>\nRisk: As with any skill, deployment risk increases if future versions request credentials, network access, or permission to modify important local or account data. <br>\nMitigation: Review artifact files and server scan evidence before deployment; this release has clean scan signals and no artifact-backed evidence of hidden, destructive, or purpose-mismatched behavior. <br>\n\n\n## Reference(s): <br>\n- [Sources - goodharts-law](references/sources.md) <br>\n- [Method in Action: Goodhart 1975 (M3) and Strathern 1997 (RAE)](examples/goodhart-1975-m3-and-strathern-1997-rae.md) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [guidance, markdown, analysis] <br>\n**Output Format:** [Markdown guidance with structured checklists and design templates] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Outputs may pause for user input in coach mode before completing the analysis.] <br>\n\n## Skill Version(s): <br>\n1.0.1 (source: server release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.0.0: 5 files, 11162 bytes\n\nFiles: examples/goodhart-1975-m3-and-strathern-1997-rae.md (10675b), references/sources.md (1408b), skill-card.md (2265b), SKILL.md (7363b), _meta.json (132b)\n\nFile v1.0.0:SKILL.md\n\n---\nname: goodharts-law\ndescription: \"Activate when: our KPI is going up but the real outcome isn't improving; people seem to be gaming the metric; we're about to tie bonuses or promotions to a number; an algorithm is producing results nobody intended; a test or audit system is being designed.\n  Do NOT activate when: the metric IS the goal with no proxy gap; measurement is purely descriptive with zero stakes attached.\"\n---\n\n# Goodhart's Law\n\n## Overview\n\n**Goodhart's Law:** when a metric controls behavior, people optimize the metric rather than the underlying goal. Formulated by economist Charles Goodhart (1975) on UK monetary policy; sharpened by Marilyn Strathern (1997): *\"When a measure becomes a target, it ceases to be a good measure.\"* Four failure mechanisms (Manheim & Garrabrant 2018): **Regressional**, **Extremal**, **Causal**, **Adversarial**. Countermeasure is always multi-metric + audit + rotation.\n\nComposes with [`feedback-loops`](../feedback-loops/SKILL.md), [`principal-agent`](../principal-agent/SKILL.md), [`okr-goal-setting`](../okr-goal-setting/SKILL.md), [`survivorship-bias`](../survivorship-bias/SKILL.md).\n\n## When to Use\n\n- A KPI is being introduced or its weight is increasing in performance evaluation\n- A metric is \"improving\" without corresponding improvement in the underlying goal\n- People are visibly optimizing for a number rather than the work it was meant to track\n- Algorithmic optimization is producing outcomes the designers didn't intend\n- Resource allocation is driven by a single composite score or ranking\n\n**Not when:** metric and goal are identical; stakes too low for gaming; metric is purely descriptive with no reward/punishment; question is which metric to use, not whether the measurement-reward system is sound.\n\n## Coaching Novices (Adaptive Front Door)\n\n- **Engine mode:** user has a concrete metric or system → run The Process directly.\n- **Coach mode:** user is unfamiliar or has no concrete case → guide step by step.\n\nIn Coach mode, respond one step at a time. Each [WAIT] is a hard stop — output only that step's question, then stop.\n\n1. One-line: before relying on a metric to control behavior, predict how people will game it — choose the system that survives that prediction.\n2. Check fit: if the metric is purely descriptive (no reward attached), Goodhart's law doesn't apply yet.\n3. Elicit the specific metric and the underlying goal: what's being measured? What's the actual outcome you care about?\n> **[WAIT — do not advance until user responds]**\n4. One question at a time: proxy gap? How would a clever agent game this? Which Goodhart category? What countermeasure fits?\n> **[WAIT — do not advance until user responds]**\n5. Close: name the gaming-resistant design (multi-metric, audit, rotation, paired-constraint) + monitoring schedule.\n> **[WAIT — do not advance until user responds]**\n\n## The Process\n\n**Step 1 — State metric and goal:** metric being targeted / underlying goal / current proxy-goal correlation / who is measured / stakes.\n\n**Step 2 — Predict the gaming:** list ≥3 ways to game the metric with minimum effort on the goal. If you can't list 3, you haven't thought hard enough.\n\n**Step 3 — Categorize mechanism:**\n\n| Mechanism | Test |\n|---|---|\n| Regressional | Is there noise that optimization will push into? |\n| Extremal | Does metric-goal correlation break at extremes? |\n| Causal | Is the metric a symptom, not a cause? |\n| Adversarial | Will agents actively game with intelligence? |\n\n**Step 4 — Choose countermeasure:** Regressional → constrain range. Extremal → paired constraint metrics. Causal → closer-to-causation metric + direct audit. Adversarial → multi-metric + randomized audits + rotation.\n**Step 5 — Design the system:** primary metric / constraint metric(s) / audit mechanism (sampled direct goal observation) / rotation schedule / separation of measure-for-control from measure-for-diagnosis / gaming-detection threshold.\n**Step 6 — Schedule re-evaluation:** independent goal measurement (how/when/who) / drift threshold / retirement criteria / owner.\n\n## Output template\n\n```markdown\n# Goodhart-Robust Design: <metric>\nMetric: | Underlying goal: | Correlation: | Who measured: | Stakes:\nGaming vectors (≥3):\nMechanism: Regressional / Extremal / Causal / Adversarial\nPrimary metric: | Constraint metric(s): | Audit: | Rotation: | Separation: | Gaming threshold:\nGoal measurement (independent): | Drift threshold: | Retirement criteria: | Owner:\n```\n\n*→ Method in Action: [Goodhart 1975 (M3) and Strathern 1997 (RAE)](examples/goodhart-1975-m3-and-strathern-1997-rae.md)*\n\n## Pack: Goodhart's Law Patterns\n\n| Domain | Common gaming | Defense |\n|---|---|---|\n| Sales quotas | Sandbagging, channel stuffing, end-of-quarter discounts | Multi-period averaging; quality metrics; clawback |\n| Hospital wait targets | Ambulance parking, patient reclassification | Outcome audits; paired metrics; randomized inspection |\n| Standardized testing | Teaching to test, curriculum narrowing | Sample-based assessment; multi-measure; reduce single-test stakes |\n| Algorithmic engagement | Clickbait, outrage, misinformation | Multi-objective optimization; quality + harm constraints |\n\n## Applying It Well\n\n- All metrics are proxies — narrower than the goal. Pre-commit to gap analysis before deployment.\n- Gaming is rational under measurement pressure. Fix the system, not the people.\n- Rotation and audit are the only durable defenses. Plan metric retirement at design time.\n\n*→ Primary sources: [references/sources.md](references/sources.md)*\n\n## Common Rationalizations\n\n| Fake move | Reality |\n|---|---|\n| [D] \"If you can't measure it, you can't manage it\" | Often false. Judgment, trust, and direct observation are also valid management tools. |\n| [D] \"Our metric is well-defined; it won't be gamed\" | Precision invites precise gaming. Basel II capital ratios were well-defined — extensively gamed. |\n| [D] \"Our people wouldn't game the metric\" | Goodhart's law is structural; individual virtue is insufficient in aggregate. |\n| [D] \"We just need a better metric\" | Often the issue is any single metric under pressure; fix is multi-metric + audit. |\n| [D] \"We've used this metric for years\" | Long use = more time for gaming to mature. Tenure is a warning, not an endorsement. |\n| *→ Add [O] entries here after each real use — paste the actual failure pattern* | *What went wrong and why* |\n\n**[D] = designed upfront | [O] = observed in real use. [O] entries are more valuable.**\n\n## Red Flags\n\n- Metric tied to high-stakes rewards or punishments\n- Metric \"improves\" without obvious improvement in the underlying goal\n- People being measured can already articulate ways to game it\n- Single metric is the primary evaluation tool, no audit or paired-constraint\n\n## Verification\n\n- [ ] Goal underlying the metric specifically named\n- [ ] Proxy gap explicitly described; ≥3 gaming vectors listed\n- [ ] Goodhart mechanism category identified\n- [ ] Countermeasure design (multi-metric, audit, rotation) in place\n- [ ] Independent goal measurement scheduled with owner and retirement date\n\n---\n\n*Part of **deciqAI Knowledge Skills** — open-source thinking skills that make rigor executable for AI agents. Built by deciqAI · https://deciqai.com · Contributions welcome — see the template at the repo root.*\n\nFile v1.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn754b8sk22s8c6gjxt02bftbn88q7ye\",\n  \"slug\": \"goodharts-law\",\n  \"version\": \"1.0.0\",\n  \"publishedAt\": 1782645448608\n}\n\nFile v1.0.0:references/sources.md\n\n# Sources — goodharts-law\n\n> *Primary sources for the [goodharts-law](../SKILL.md) skill.*\n\n- Goodhart, C. A. E. (1975). \"Problems of monetary management: The U.K. experience.\" Papers in Monetary Economics, Reserve Bank of Australia. Reprinted in *Monetary Theory and Practice: The U.K. Experience* (1984), Macmillan. The original.\n- Strathern, M. (1997). \"'Improving ratings': Audit in the British university system.\" *European Review*, 5(3), 305-321. The modern aphoristic formulation.\n- Campbell, D. T. (1979). \"Assessing the impact of planned social change.\" *Evaluation and Program Planning*, 2(1), 67-90. Independent formulation (\"Campbell's Law\").\n- Lucas, R. E. (1976). \"Econometric policy evaluation: A critique.\" *Carnegie-Rochester Conference Series on Public Policy*, 1(1), 19-46. The adjacent macro-econometric \"Lucas critique.\"\n- Manheim, D., & Garrabrant, S. (2018). \"Categorizing variants of Goodhart's Law.\" *arXiv:1803.04585*. The four-mechanism taxonomy.\n- Muller, J. Z. (2018). *The Tyranny of Metrics.* Princeton University Press. ISBN 978-0691174952. Comprehensive case-study survey.\n- Doerr, J. (2018). *Measure What Matters: How Google, Bono, and the Gates Foundation Rock the World with OKRs.* Portfolio. ISBN 978-0525536222.\n- Wells Fargo / Office of the Comptroller of the Currency (2016, 2018). Various enforcement actions and consent orders documenting the cross-selling case.\n\nFile v1.0.0:examples/goodhart-1975-m3-and-strathern-1997-rae.md\n\n# Method in Action: Goodhart 1975 (M3) and Strathern 1997 (RAE)\n\n> *Example for the [goodharts-law](../SKILL.md) skill.*\n\nThe empirical foundation has two key moments. The first is **Charles Goodhart's 1975 critique of UK monetary policy**, originally a conference paper for the Reserve Bank of Australia, later expanded in his 1984 book *Monetary Theory and Practice*.\n\nThe context: in the mid-1970s, the Bank of England (and many other central banks) had identified a robust historical correlation between growth in broad money supply (the M3 aggregate) and subsequent inflation. The natural policy implication was straightforward: target M3 growth at a level consistent with low inflation, and inflation would be controlled.\n\nGoodhart was skeptical. His objection was structural, not empirical:\n\n> \"It is not the case that there is some fixed and stable relationship between an observed monetary aggregate, such as M3, and other variables in the economic system. Rather, the relationships we observe statistically are the equilibrium outcomes of behavior by banks, depositors, and borrowers, each of whom is responding to a complex set of incentives. When the regulator targets M3 — and especially when the targeting carries policy weight that will affect interest rates and reserve requirements — these economic actors will reorganize their behavior to operate around the regulation. Liquid funds will be reclassified into categories that fall outside the M3 definition. The pre-targeting M3-inflation correlation will not survive the targeting. Any observed statistical regularity will tend to collapse once pressure is placed upon it for control purposes.\"\n>\n> — Goodhart (1975), as reprinted in Goodhart (1984), pp. 96-98.\n\nGoodhart's prediction was empirical and falsifiable. It came true within a few years: as the Bank of England's M3 targets bit, UK banks began creating money-substitutes that escaped the M3 definition (most notably, the rise of the eurodollar market and certain ty","readmeExcerpt":"Skill: Goodhart's Law Owner: deciqai Summary: Activate when: our KPI is going up but the real outcome isn't improving; people seem to be gaming the metric; we're about to tie bonuses or promotions to a n... Tags: latest:1.0.5 Version history: v1.0.5 | 2026-07-16T18:01:32.110Z | user Description tail link + agents machine-readable metadata line (deciqai.com/s/goodharts-law.json) v1.0.4 | 2026-07-09T11:17:54.034Z | use","codeSnippets":[],"executableExamples":[{"language":"markdown","snippet":"# Goodhart-Robust Design: <metric>\nMetric: | Underlying goal: | Correlation: | Who measured: | Stakes:\nGaming vectors (≥3):\nMechanism: Regressional / Extremal / Causal / Adversarial\nPrimary metric: | Constraint metric(s): | Audit: | Rotation: | Separation: | Gaming threshold:\nGoal measurement (independent): | Drift threshold: | Retirement criteria: | Owner:"},{"language":"markdown","snippet":"# Goodhart-Robust Design: <metric>\nMetric: | Underlying goal: | Correlation: | Who measured: | Stakes:\nGaming vectors (≥3):\nMechanism: Regressional / Extremal / Causal / Adversarial\nPrimary metric: | Constraint metric(s): | Audit: | Rotation: | Separation: | Gaming threshold:\nGoal measurement (independent): | Drift threshold: | Retirement criteria: | Owner:"},{"language":"markdown","snippet":"# Goodhart-Robust Design: <metric>\nMetric: | Underlying goal: | Correlation: | Who measured: | Stakes:\nGaming vectors (≥3):\nMechanism: Regressional / Extremal / Causal / Adversarial\nPrimary metric: | Constraint metric(s): | Audit: | Rotation: | Separation: | Gaming threshold:\nGoal measurement (independent): | Drift threshold: | Retirement criteria: | Owner:"},{"language":"markdown","snippet":"# Goodhart-Robust Design: <metric>\nMetric: | Underlying goal: | Correlation: | Who measured: | Stakes:\nGaming vectors (≥3):\nMechanism: Regressional / Extremal / Causal / Adversarial\nPrimary metric: | Constraint metric(s): | Audit: | Rotation: | Separation: | Gaming threshold:\nGoal measurement (independent): | Drift threshold: | Retirement criteria: | Owner:"},{"language":"markdown","snippet":"# Goodhart-Robust Design: <metric>\nMetric: | Underlying goal: | Correlation: | Who measured: | Stakes:\nGaming vectors (≥3):\nMechanism: Regressional / Extremal / Causal / Adversarial\nPrimary metric: | Constraint metric(s): | Audit: | Rotation: | Separation: | Gaming threshold:\nGoal measurement (independent): | Drift threshold: | Retirement criteria: | Owner:"},{"language":"markdown","snippet":"# Goodhart-Robust Design: <metric>\nMetric: | Underlying goal: | Correlation: | Who measured: | Stakes:\nGaming vectors (≥3):\nMechanism: Regressional / Extremal / Causal / Adversarial\nPrimary metric: | Constraint metric(s): | Audit: | Rotation: | Separation: | Gaming threshold:\nGoal measurement (independent): | Drift threshold: | Retirement criteria: | Owner:"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: goodharts-law\ndescription: \"Activate when: our KPI is going up but the real outcome isn't improving; people seem to be gaming the metric; we're about to tie bonuses or promotions to a number; an algorithm is producing results nobody intended; a test or audit system is being designed.\n  Do NOT activate when: the metric IS the goal with no proxy gap; measurement is purely descriptive with zero stakes attached. More: deciqai.com/c/goodharts-law\"\n---\n\n# Goodhart's Law\n\n## Overview\n\n**Goodhart's Law:** when a metric controls behavior, people optimize the metric rather than the underlying goal. Formulated by economist Charles Goodhart (1975) on UK monetary policy; sharpened by Marilyn Strathern (1997): *\"When a measure becomes a target, it ceases to be a good measure.\"* Four failure mechanisms (Manheim & Garrabrant 2018): **Regressional**, **Extremal**, **Causal**, **Adversarial**. Countermeasure is always multi-metric + audit + rotation.\n\nComposes with `feedback-loops`, `principal-agent`, `okr-goal-setting`, `survivorship-bias`.\n\n## When to Use\n\n- A KPI is being introduced or its weight is increasing in performance evaluation\n- A metric is \"improving\" without corresponding improvement in the underlying goal\n- People are visibly optimizing for a number rather than the work it was meant to track\n- Algorithmic optimization is producing outcomes the designers didn't intend\n- Resource allocation is driven by a single composite score or ranking\n- An AI model, benchmark, or engagement metric is being optimized (or used to justify AI capex / adoption / AI-native competition) and the score is rising faster than real capability or user value\n\n**Not when:** metric and goal are identical; stakes too low for gaming; metric is purely descriptive with no reward/punishment; question is which metric to use, not whether the measurement-reward system is sound.\n\n## Coaching Novices (Adaptive Front Door)\n\n- **Engine mode:** user has a concrete metric or system → run The Process directly.\n- **Coach mode:** user is unfamiliar or has no concrete case → guide step by step.\n\nIn Coach mode, respond one step at a time. Each [WAIT] is a hard stop — output only that step's question, then stop.\n\n1. One-line: before relying on a metric to control behavior, predict how people will game it — choose the system that survives that prediction.\n2. Check fit: if the metric is purely descriptive (no reward attached), Goodhart's law doesn't apply yet.\n3. Elicit the specific metric and the underlying goal: what's being measured? What's the actual outcome you care about?\n> **[WAIT — do not advance until user responds]**\n4. One question at a time: proxy gap? How would a clever agent game this? Which Goodhart category? What countermeasure fits?\n> **[WAIT — do not advance until user responds]**\n5. Close: name the gaming-resistant design (multi-metric, audit, rotation, paired-constraint) + monitoring schedule.\n> **[WAIT — do not advance until user responds]**\n\n## The Process\n\n**Step 1 — S"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn754b8sk22s8c6gjxt02bftbn88q7ye\",\n  \"slug\": \"goodharts-law\",\n  \"version\": \"1.0.5\",\n  \"publishedAt\": 1784224892110\n}"},{"path":"references/sources.md","content":"# Sources — goodharts-law\n\n> *Primary sources for the [goodharts-law](../SKILL.md) skill.*\n\n- Goodhart, C. A. E. (1975). \"Problems of monetary management: The U.K. experience.\" Papers in Monetary Economics, Reserve Bank of Australia. Reprinted in *Monetary Theory and Practice: The U.K. Experience* (1984), Macmillan. The original.\n- Strathern, M. (1997). \"'Improving ratings': Audit in the British university system.\" *European Review*, 5(3), 305-321. The modern aphoristic formulation.\n- Campbell, D. T. (1979). \"Assessing the impact of planned social change.\" *Evaluation and Program Planning*, 2(1), 67-90. Independent formulation (\"Campbell's Law\").\n- Lucas, R. E. (1976). \"Econometric policy evaluation: A critique.\" *Carnegie-Rochester Conference Series on Public Policy*, 1(1), 19-46. The adjacent macro-econometric \"Lucas critique.\"\n- Manheim, D., & Garrabrant, S. (2018). \"Categorizing variants of Goodhart's Law.\" *arXiv:1803.04585*. The four-mechanism taxonomy.\n- Muller, J. Z. (2018). *The Tyranny of Metrics.* Princeton University Press. ISBN 978-0691174952. Comprehensive case-study survey.\n- Doerr, J. (2018). *Measure What Matters: How Google, Bono, and the Gates Foundation Rock the World with OKRs.* Portfolio. ISBN 978-0525536222.\n- Wells Fargo / Office of the Comptroller of the Currency (2016, 2018). Various enforcement actions and consent orders documenting the cross-selling case.\n- Zhou, K., et al. (2023). \"Don't Make Your LLM an Evaluation Benchmark Cheater.\" *arXiv:2311.01964*. Documents benchmark data contamination and how public evaluation scores decay as a capability signal once test data leaks into training — a direct 2020s AI instance of Goodhart's law.\n- Chiang, W.-L., et al. (2024). \"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference.\" *arXiv:2403.04132*. Motivates live, blind, human-preference evaluation as a harder-to-game complement to static leaderboards; illustrates both the countermeasure and its own residual gaming risks."},{"path":"examples/ai-benchmark-and-engagement-gaming-2023-2026.md","content":"# Method in Action: AI Benchmarks and Engagement Metrics as Targets (2023–2026)\n\n> *Example for the [goodharts-law](../SKILL.md) skill.*\n\nBy the mid-2020s, Goodhart's law had become one of the most-cited frames inside the AI industry itself — because two of its own core metrics visibly decayed under optimization pressure. First, **public benchmark scores** (MMLU, GSM8K, HumanEval, and a proliferation of leaderboards) came to dominate model marketing, funding narratives, and internal go/no-go decisions — and, predictably, models began scoring well without a matching gain in real-world capability. Second, **consumer-app engagement metrics** (watch time, session length, daily active use) continued their long slide from \"signal of user value\" to \"target that no longer measures it.\" This walks both cases through the skill's own six-step Process.\n\n---\n\n## Step 1 — State metric and goal\n\n**Case A — AI benchmarks.**\n- **Metric being targeted:** score on a fixed public benchmark (e.g., a multiple-choice knowledge test like MMLU, a grade-school math set like GSM8K, or a coding pass-rate like HumanEval).\n- **Underlying goal:** general, transferable model capability — does the model actually reason, code, and generalize on tasks users bring that were *not* in the test set?\n- **Current proxy–goal correlation:** initially high on a genuinely held-out test, and a legitimate research signal. It degrades as the benchmark ages, becomes a marketing target, and (critically) as the test's questions leak into training data.\n- **Who is measured:** frontier and open-weight model developers; the score is read by press, investors, enterprise buyers, and internal leadership deciding what to ship.\n- **Stakes:** very high — leaderboard position drives valuations, capex-justification narratives, and enterprise procurement in an intensely competitive, AI-native market.\n\n**Case B — engagement metrics.**\n- **Metric being targeted:** engagement (watch time, session length, DAU/MAU, scroll depth) feeding a recommendation or ranking system.\n- **Underlying goal:** users getting durable value — time well spent, learning, connection, satisfaction they'd endorse on reflection.\n- **Correlation:** engagement is a real proxy for value at low intensity, but the two diverge sharply as the system optimizes hard against engagement.\n- **Who is measured:** the ranking model and the teams whose OKRs it feeds.\n- **Stakes:** very high — engagement drives ad revenue and growth targets.\n\n## Step 2 — Predict the gaming (≥3 vectors each)\n\n**Case A — benchmarks:**\n1. **Train on the test (contamination).** Benchmark questions and answers, published openly on the web, get scraped into pretraining or fine-tuning corpora — accidentally or deliberately. The model then \"knows\" the answers rather than deriving them. Contamination of popular benchmarks was widely documented and discussed across the research community by 2023–2024.\n2. **Overfit the format.** Tune specifically to the benchmark's answer style, pr"},{"path":"examples/goodhart-1975-m3-and-strathern-1997-rae.md","content":"# Method in Action: Goodhart 1975 (M3) and Strathern 1997 (RAE)\n\n> *Example for the [goodharts-law](../SKILL.md) skill.*\n\nThe empirical foundation has two key moments. The first is **Charles Goodhart's 1975 critique of UK monetary policy**, originally a conference paper for the Reserve Bank of Australia, later expanded in his 1984 book *Monetary Theory and Practice*.\n\nThe context: in the mid-1970s, the Bank of England (and many other central banks) had identified a robust historical correlation between growth in broad money supply (the M3 aggregate) and subsequent inflation. The natural policy implication was straightforward: target M3 growth at a level consistent with low inflation, and inflation would be controlled.\n\nGoodhart was skeptical. His objection was structural, not empirical:\n\n> \"It is not the case that there is some fixed and stable relationship between an observed monetary aggregate, such as M3, and other variables in the economic system. Rather, the relationships we observe statistically are the equilibrium outcomes of behavior by banks, depositors, and borrowers, each of whom is responding to a complex set of incentives. When the regulator targets M3 — and especially when the targeting carries policy weight that will affect interest rates and reserve requirements — these economic actors will reorganize their behavior to operate around the regulation. Liquid funds will be reclassified into categories that fall outside the M3 definition. The pre-targeting M3-inflation correlation will not survive the targeting. Any observed statistical regularity will tend to collapse once pressure is placed upon it for control purposes.\"\n>\n> — Goodhart (1975), as reprinted in Goodhart (1984), pp. 96-98.\n\nGoodhart's prediction was empirical and falsifiable. It came true within a few years: as the Bank of England's M3 targets bit, UK banks began creating money-substitutes that escaped the M3 definition (most notably, the rise of the eurodollar market and certain types of negotiable certificates of deposit). The M3-inflation correlation collapsed; the Bank quietly abandoned M3 targeting by the mid-1980s.\n\nThe principle's second formative moment came two decades later, in **Marilyn Strathern's 1997 ethnographic study of British universities** during the rise of the Research Assessment Exercise (RAE). Strathern, an anthropologist at Cambridge, observed the RAE's effect on academic behavior:\n\n> \"The Research Assessment Exercise sets out to evaluate the quality of research conducted in British universities. The exercise was introduced with the goal of identifying excellent research and directing funding toward it. The instrument: publication count, citation analysis, peer-rated quality. The exercise's effect on the academy has been precisely the inversion of its stated goal. Academics have shifted from publishing fewer, more substantial works to publishing more, smaller, shallower ones. They have shifted topic selection toward areas where rapid publication"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Activate when: our KPI is going up but the real outcome isn't improving; people seem to be gaming the metric; we're about to tie bonuses or promotions to a n... Skill: Goodhart's Law Owner: deciqai Summary: Activate when: our KPI is going up but the real outcome isn't improving; people seem to be gaming the metric; we're about to tie bonuses or promotions to a n... Tags: latest:1.0.5 Version history: v1.0.5 | 2026-07-16T18:01:32.110Z | user Description tail link + agents machine-readable metadata line (deciqai.com/s/goodharts-law.json) v1.0.4 | 2026-07-09T11:17:54.034Z | use","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":2009,"uniquenessScore":53,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T00:08:25.475Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T00:08:25.475Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T03:56:43.688Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}