{"id":"f0921a83-6395-4f80-8784-f59f7177c42a","entityType":"agent","slug":"clawhub-athola-nm-imbue-karpathy-principles","name":"karpathy-principles","canonicalUrl":"https://www.xpersona.co/agent/clawhub-athola-nm-imbue-karpathy-principles","canonicalPath":"/agent/clawhub-athola-nm-imbue-karpathy-principles","generatedAt":"2026-10-11T21:00:25.862Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-11T17:54:19.397Z","emptyReason":null},"description":"Pre-implementation gate covering think-first, simplicity, surgical edits, and verifiable goals Skill: karpathy-principles Owner: athola Summary: Pre-implementation gate covering think-first, simplicity, surgical edits, and verifiable goals Tags: latest:1.9.19 Version history: v1.9.19 | 2026-08-26T13:12:45.526Z | user Release v1.9.19 v1.9.17 | 2026-07-30T05:34:19.068Z | user Release v1.9.17 v1.9.16 | 2026-07-14T19:51:02.878Z | user Release v1.9.16 v1.9.14 | 2026-06-30T18:00:06.428Z | user Release v1.9.14 v1.9.1","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s17emme0e2m3cpf7k2jvp3a84984b8z9:nm-imbue-karpathy-principles","sourceUrl":"https://clawhub.ai/athola/nm-imbue-karpathy-principles","homepage":"https://clawhub.ai/athola/skills/nm-imbue-karpathy-principles","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/athola/nm-imbue-karpathy-principles","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/athola/skills/nm-imbue-karpathy-principles","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":60,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Pre-implementation gate covering think-first, simplicity, surgical edits, and verifiable goals Skill: karpathy-principles Owner: athola Summary: Pre-implementat"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T17:54:19.397Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T17:54:19.397Z","emptyReason":null},"stars":null,"forks":null,"downloads":1014,"likes":null,"task":null,"library":null,"packageName":null,"latestVersion":"1.9.19","tractionLabel":"1K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T17:54:19.383Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T17:54:19.397Z","lastCrawledAt":"2026-10-11T17:54:19.383Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T17:54:19.383Z","lastVerifiedAt":null,"highlights":[{"version":"1.9.19","createdAt":"2026-08-26T13:12:45.526Z","changelog":"Release v1.9.19","fileCount":7,"zipByteSize":15167},{"version":"1.9.17","createdAt":"2026-07-30T05:34:19.068Z","changelog":"Release v1.9.17","fileCount":7,"zipByteSize":15067},{"version":"1.9.16","createdAt":"2026-07-14T19:51:02.878Z","changelog":"Release v1.9.16","fileCount":7,"zipByteSize":14968},{"version":"1.9.14","createdAt":"2026-06-30T18:00:06.428Z","changelog":"Release v1.9.14","fileCount":7,"zipByteSize":14978},{"version":"1.9.13","createdAt":"2026-06-27T16:18:50.722Z","changelog":"Release v1.9.13","fileCount":7,"zipByteSize":14991},{"version":"1.9.12","createdAt":"2026-06-19T03:12:59.042Z","changelog":"Release v1.9.12","fileCount":7,"zipByteSize":14902},{"version":"1.0.0","createdAt":"2026-06-18T14:08:13.515Z","changelog":"- Initial release of the Karpathy Principles skill as a pre-implementation quality gate. - Introduces four core principles: think-first, simplicity, surgical edits, and verifiable goals. - Provides guidance and checklists to reduce common LLM and coding pitfalls before shipping code. - Includes references to supporting skills and modules for deeper dives and practical self-checks. - Defines concrete exit criteria and required pre-flight artifacts to ensure rigorous implementation habits.","fileCount":7,"zipByteSize":15027}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17emme0e2m3cpf7k2jvp3a84984b8z9:nm-imbue-karpathy-principles","setupComplexity":"low","setupSteps":["Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-imbue-karpathy-principles/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-imbue-karpathy-principles/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-imbue-karpathy-principles/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-imbue-karpathy-principles/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-imbue-karpathy-principles/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-imbue-karpathy-principles/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T21:00:25.859Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-imbue-karpathy-principles/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-imbue-karpathy-principles/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-imbue-karpathy-principles/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-imbue-karpathy-principles/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-11T17:54:19.397Z","emptyReason":null},"readme":"Skill: karpathy-principles\n\nOwner: athola\n\nSummary: Pre-implementation gate covering think-first, simplicity, surgical edits, and verifiable goals\n\nTags: latest:1.9.19\n\nVersion history:\n\nv1.9.19 | 2026-08-26T13:12:45.526Z | user\n\nRelease v1.9.19\n\nv1.9.17 | 2026-07-30T05:34:19.068Z | user\n\nRelease v1.9.17\n\nv1.9.16 | 2026-07-14T19:51:02.878Z | user\n\nRelease v1.9.16\n\nv1.9.14 | 2026-06-30T18:00:06.428Z | user\n\nRelease v1.9.14\n\nv1.9.13 | 2026-06-27T16:18:50.722Z | user\n\nRelease v1.9.13\n\nv1.9.12 | 2026-06-19T03:12:59.042Z | user\n\nRelease v1.9.12\n\nv1.0.0 | 2026-06-18T14:08:13.515Z | auto\n\n- Initial release of the Karpathy Principles skill as a pre-implementation quality gate.\n- Introduces four core principles: think-first, simplicity, surgical edits, and verifiable goals.\n- Provides guidance and checklists to reduce common LLM and coding pitfalls before shipping code.\n- Includes references to supporting skills and modules for deeper dives and practical self-checks.\n- Defines concrete exit criteria and required pre-flight artifacts to ensure rigorous implementation habits.\n\nArchive index:\n\nArchive v1.9.19: 7 files, 15167 bytes\n\nFiles: modules/anti-patterns.md (8298b), modules/senior-engineer-test.md (3378b), modules/tradeoff-acknowledgment.md (3757b), modules/verifiable-goals.md (4501b), skill-card.md (2566b), SKILL.md (7513b), _meta.json (148b)\n\nFile v1.9.19:SKILL.md\n\n---\nname: karpathy-principles\ndescription: |\n  Pre-implementation gate covering think-first, simplicity, surgical edits, and verifiable goals\nversion: 1.9.8\ntriggers:\n  - karpathy\n  - coding-pitfalls\n  - synthesis\n  - entry-point\n  - discipline\n  - anti-overengineering\n  - TDD\n  - starting implementation to verify the approach\nmetadata: {\"openclaw\": {\"homepage\": \"https://github.com/athola/claude-night-market/tree/master/plugins/imbue\", \"emoji\": \"\\ud83e\\udd9e\", \"requires\": {\"config\": [\"night-market.imbue:scope-guard\", \"night-market.imbue:proof-of-work\", \"night-market.imbue:rigorous-reasoning\", \"night-market.leyline:additive-bias-defense\", \"night-market.conserve:code-quality-principles\"]}}}\nsource: claude-night-market\nsource_plugin: imbue\n---\n\n> **Night Market Skill** — ported from [claude-night-market/imbue](https://github.com/athola/claude-night-market/tree/master/plugins/imbue). For the full experience with agents, hooks, and commands, install the Claude Code plugin.\n\n\n> The models make wrong assumptions on your behalf and\n> just run along with them without checking. They don't\n> manage their confusion, don't seek clarifications,\n> don't surface inconsistencies, don't present\n> tradeoffs, don't push back when they should.\n>\n> -- Andrej Karpathy, on agentic coding failure modes\n\n## What This Is\n\nA four-principle contract for reducing the most common\nLLM coding pitfalls. Compact entry-point. Each\nprinciple has a deeper-dive skill in night-market;\nthis skill is the index, not the encyclopedia.\n\nDerivation: distilled by Forrest Chang\n(forrestchang/andrej-karpathy-skills, MIT) from\nKarpathy's observations. Full attribution in\n`references/source-attribution.md`.\n\n## When to Use\n\n- Before starting any coding task larger than a typo\n- During code review, to name the failure mode you see\n- After writing a diff, to self-audit before claiming\n  done\n- When training a junior engineer to read agent diffs\n\n## When NOT to Use\n\nThese principles bias toward caution over speed. For\ncases listed in `modules/tradeoff-acknowledgment.md`,\nuse judgment: trivial fixes, exploratory spikes,\ndocumentation-only edits, and time-boxed prototypes.\n\n## The Four Principles\n\n### 1. Think Before Coding\n\n**State assumptions. Surface confusion. Match tone to\nevidence.**\n\n- If multiple interpretations of the request exist,\n  list them. Do not silently pick.\n- If a simpler approach exists, name it. Push back\n  when the simpler path is correct.\n- If something is unclear, stop and ask. Hidden\n  assumptions are the cheapest bug to prevent and the\n  most expensive to find later.\n- Make claims no stronger than the evidence supports.\n  Calibrated tone beats confident hand-waving.\n\nDeep dives: `Skill(imbue:rigorous-reasoning)` for the\nsycophancy guard, `Skill(superpowers:brainstorming)`\nfor option generation, `/spec-kit:speckit-clarify`\ncommand for ambiguity drilldown.\n\n### 2. Simplicity First\n\n**Minimum code that solves the problem. Nothing\nspeculative.**\n\n> They really like to overcomplicate code and APIs,\n> bloat abstractions.\n>\n> -- Andrej Karpathy, on the same agentic-coding thread\n\n\n\n- No features beyond what was asked\n- No abstractions for single-use code\n- No flexibility or configurability that wasn't\n  requested\n- No error handling for impossible scenarios\n- If you wrote 200 lines and it could be 50, rewrite\n  it\n\nSelf-check: would a senior engineer say this is\novercomplicated? See `modules/senior-engineer-test.md`.\n\nDeep dives: `Skill(imbue:scope-guard)` for the\nworthiness formula and branch budgets,\n`Skill(leyline:additive-bias-defense)` for burden of\nproof on every addition,\n`Skill(conserve:code-quality-principles)` for the\nKISS / YAGNI / SOLID foundation.\n\n### 3. Surgical Changes\n\n**Touch only what you must. Clean up only your own\nmess.**\n\n- Do not improve adjacent code, comments, or\n  formatting\n- Do not refactor things that aren't broken\n- Match existing style even when you would do it\n  differently\n- If you notice unrelated dead code, mention it; do\n  not delete it\n- When your changes orphan imports or variables,\n  remove the orphans you created. Pre-existing dead\n  code stays unless asked.\n\nThe trace-back test: every changed line should trace\ndirectly to the user's request.\n\nDeep dives: `Skill(imbue:justify)` for additive-bias\naudits on diffs, `Skill(leyline:additive-bias-defense)`\nfor the burden-of-proof contract, the\n`bounded-discovery.md` rule for read-budget caps.\n\n### 4. Goal-Driven Execution\n\n**Define verifiable success criteria. Loop until\nverified.**\n\nTransform vague tasks into checkable goals:\n\n- \"Add validation\" becomes \"tests for invalid inputs\n  pass\"\n- \"Fix the bug\" becomes \"test reproducing the bug,\n  then make it pass\"\n- \"Refactor X\" becomes \"tests pass before and after\"\n- \"Make it faster\" becomes \"benchmark Y under N ms\"\n\nFor multi-step tasks, state a brief plan with\nverification per step. Strong success criteria let\nyou loop independently. Weak criteria require\nconstant clarification.\n\nSee `modules/verifiable-goals.md` for the full\nreformulation template.\n\nDeep dives: `Skill(imbue:proof-of-work)` for the Iron\nLaw (no implementation without a failing test first),\n`Skill(superpowers:test-driven-development)` for the\nRED-GREEN-REFACTOR loop.\n\n## The Karpathy Self-Check\n\nBefore you ship, four questions:\n\n| Principle | Question |\n|-----------|----------|\n| Think Before Coding | Did I list assumptions, or did I guess silently? |\n| Simplicity First | Would a senior engineer call this overcomplicated? |\n| Surgical Changes | Does every changed line trace to the request? |\n| Goal-Driven Execution | Can I prove this is done with a check, not a feeling? |\n\nFour \"yes\" answers means ship. Anything else means\niterate.\n\n## Modules\n\n- `modules/anti-patterns.md` - eight named drift rails\n  with before/after diffs\n- `modules/senior-engineer-test.md` - the\n  three-question self-check battery\n- `modules/verifiable-goals.md` - vague-to-verifiable\n  reformulation template with worked examples\n- `modules/tradeoff-acknowledgment.md` - when the four\n  principles do not apply\n\n## References\n\n- `references/source-attribution.md` - Karpathy\n  primary citation, Forrest Chang derivation, license,\n  adjacent prior art\n\n## Related Skills\n\n- `Skill(imbue:scope-guard)` - worthiness formula and\n  branch budgets\n- `Skill(imbue:proof-of-work)` - Iron Law TDD gate\n- `Skill(imbue:rigorous-reasoning)` - sycophancy and\n  hidden-assumption guard\n- `Skill(imbue:justify)` - additive-bias diff audit\n- `Skill(leyline:additive-bias-defense)` - burden of\n  proof on every addition\n- `Skill(conserve:code-quality-principles)` - KISS,\n  YAGNI, SOLID\n- `Skill(superpowers:test-driven-development)` -\n  RED-GREEN-REFACTOR\n- `Skill(superpowers:brainstorming)` - generate\n  options before committing\n- See `docs/quality-gates.md#skill-level-quality-gate-composition`\n  for the full gate-skill federation graph (this skill\n  is the synthesis hub)\n\n## Required TodoWrite Items\n\nWhen invoked as a pre-flight gate, create:\n\n- `karpathy:assumptions-listed` - principle 1 satisfied\n- `karpathy:simplicity-checked` - principle 2 satisfied\n- `karpathy:trace-back-verified` - principle 3 satisfied\n- `karpathy:success-criteria-defined` - principle 4\n  satisfied\n\n## Exit Criteria\n\n- Each of the four principles has been answered with a\n  concrete artifact (assumption list, scope rationale,\n  diff trace, verification plan).\n- The senior-engineer test was applied at least once.\n- Verifiable success criteria are written down before\n  the implementation begins.\n\nFile v1.9.19:_meta.json\n\n{\n  \"ownerId\": \"kn7d107jg9jv602h9ytsegydq184a42s\",\n  \"slug\": \"nm-imbue-karpathy-principles\",\n  \"version\": \"1.9.19\",\n  \"publishedAt\": 1787749965526\n}\n\nFile v1.9.19:modules/anti-patterns.md\n\n# Anti-Patterns: Eight Named Drift Rails\n\nEach drift rail names a specific way LLM-driven coding\ngoes wrong, with a trigger pattern, a short bad\nexample, a short good example, and a one-sentence\ntest you can apply to your own diff. Naming the rail\nhelps recall.\n\nAdapted from observations distilled by Forrest Chang\nfrom Karpathy's notes; see\n`references/source-attribution.md` for the citation\nchain. Examples are rewritten in our own prose to\nillustrate the failure shape, not lifted verbatim.\n\n## AP-1: Hidden Assumptions\n\n**Maps to**: Principle 1 (Think Before Coding)\n\n**Trigger**: User asks for a feature with multiple\nunstated dimensions (scope, format, volume, privacy).\nThe agent picks one silently and ships.\n\n**Bad shape**\n\n```python\ndef export_users(format='json'):\n    users = User.query.all()\n    if format == 'json':\n        with open('users.json', 'w') as f:\n            json.dump([u.to_dict() for u in users], f)\n```\n\nAssumed: all users (privacy?), file path, fields,\nvolume.\n\n**Good shape**\n\n> Before implementing, I want to confirm:\n>\n> - Scope: all users, or filtered?\n> - Format: download, background job, or API endpoint?\n> - Fields: which? Some are sensitive.\n> - Volume: thousands or millions?\n>\n> Default proposal: a paginated JSON endpoint. Adjust?\n\n**Test**: count silent assumptions in the diff. Three\nor more is a rail violation.\n\n## AP-2: Multiple Interpretations Picked Silently\n\n**Maps to**: Principle 1 (Think Before Coding)\n\n**Trigger**: A request like \"make the search faster\"\nhas at least three valid readings (response time,\nthroughput, perceived speed). The agent picks one and\nimplements 200 lines without flagging the choice.\n\n**Bad shape**\n\n```python\n@lru_cache(maxsize=1000)\nasync def search(query: str) -> List[Result]:\n    # 200 lines of caching, async, indexes, all picked\n    # without confirming what \"faster\" means\n    ...\n```\n\n**Good shape**\n\n> \"Faster\" could mean response time, throughput, or\n> perceived speed. Each is a different fix:\n>\n> 1. Response time: indexes plus query caching (~2h)\n> 2. Throughput: async path plus pooling (~4h)\n> 3. Perceived: progressive load and partial results\n>    (~3h)\n>\n> Current: ~500ms typical. Which axis matters?\n\n**Test**: when the verb in the request is ambiguous\n(faster, better, cleaner, simpler), did the agent name\nthe alternatives or pick one?\n\n## AP-3: Strategy Pattern for One Function\n\n**Maps to**: Principle 2 (Simplicity First)\n\n**Trigger**: User asks for a single function. The\nagent ships an abstract base class, two implementing\nclasses, a config dataclass, and a coordinator class\nfor ten lines of arithmetic.\n\n**Bad shape**\n\n```python\nclass DiscountStrategy(ABC):\n    @abstractmethod\n    def calculate(self, amount: float) -> float: ...\n\nclass PercentageDiscount(DiscountStrategy):\n    def __init__(self, p): self.p = p\n    def calculate(self, a): return a * (self.p / 100)\n\n# Plus FixedDiscount, DiscountConfig, DiscountCalculator\n# for what should be one function\n```\n\n**Good shape**\n\n```python\ndef calculate_discount(amount: float, percent: float) -> float:\n    return amount * (percent / 100)\n```\n\n**Test**: count types and classes added per actual use\ncase. If types-added exceeds use-cases-served, the\npattern is premature.\n\n## AP-4: Speculative Features\n\n**Maps to**: Principle 2 (Simplicity First)\n\n**Trigger**: \"Save user preferences to database\"\nbecomes a class with optional caching, validation,\nnotification hooks, and merge semantics. None were\nasked for.\n\n**Bad shape**\n\n```python\nclass PreferenceManager:\n    def __init__(self, db, cache=None, validator=None):\n        ...\n    def save(self, user_id, prefs,\n             merge=True, validate=True, notify=False):\n        # 60 lines of optional behavior\n```\n\n**Good shape**\n\n```python\ndef save_preferences(db, user_id: int, preferences: dict):\n    db.execute(\n        \"UPDATE users SET preferences = ? WHERE id = ?\",\n        (json.dumps(preferences), user_id),\n    )\n```\n\n**Test**: list every parameter that was not in the\nrequest. If you cannot point at a sentence in the\nrequest that demanded it, delete the parameter.\n\n## AP-5: Drive-by Refactoring\n\n**Maps to**: Principle 3 (Surgical Changes)\n\n**Trigger**: User reports a single bug. The agent\nfixes the bug, then \"improves\" three other functions,\nadds docstrings, and tightens validation logic that\nnobody asked about.\n\n**Bad shape**: a 90-line diff to fix a 4-line bug,\nwith related but unrequested cleanups across two more\nfiles.\n\n**Good shape**: a 4-line diff that fixes only the\nreported bug. If you noticed unrelated issues, list\nthem in the response and ask before touching them.\n\n**Test**: read the diff line by line. For each\nchanged line, ask \"which sentence in the user's\nrequest demanded this?\" Lines without an answer are\ncandidates for removal from the diff.\n\n## AP-6: Style Drift During Edit\n\n**Maps to**: Principle 3 (Surgical Changes)\n\n**Trigger**: User asks for one logging line in an\nupload function. The agent ships type hints,\ndocstrings, single-quote-to-double-quote conversion,\nand a flattened control flow.\n\n**Bad shape**\n\n```diff\n- def upload_file(file_path, destination):\n+ def upload_file(file_path: str, destination: str) -> bool:\n+     \"\"\"Upload file to destination.\"\"\"\n      try:\n-         with open(file_path, 'rb') as f:\n+         with open(file_path, \"rb\") as f:\n              ...\n```\n\n**Good shape**\n\n```diff\n+ logger = logging.getLogger(__name__)\n+\n  def upload_file(file_path, destination):\n+     logger.info(f'Starting upload: {file_path}')\n      try:\n          with open(file_path, 'rb') as f:\n              ...\n```\n\n**Test**: the diff should not change quote style,\ntype hint presence, docstring presence, or whitespace\npatterns unless the request named them.\n\n## AP-7: Vague Success Criteria\n\n**Maps to**: Principle 4 (Goal-Driven Execution)\n\n**Trigger**: \"Fix the authentication system.\" The\nagent responds \"I will review the code, identify\nissues, make improvements, and test.\" That is a plan\nshape, not a verification plan.\n\n**Bad shape**: a four-step list of verbs (review,\nidentify, improve, test) with no checkable signal per\nstep.\n\n**Good shape**\n\n> \"Fix authentication\" is too broad to verify. Best\n> guess at the actual problem: sessions stay alive\n> after password change. Plan:\n>\n> 1. Write failing test: change password, prior\n>    session is rejected.\n> 2. Implement: invalidate sessions on change.\n> 3. Edge cases: concurrent change, multi-session.\n> 4. Regression: full auth suite green.\n>\n> Is the password-change-stale-session bug the right\n> target?\n\n**Test**: each step in the plan should have a one-\nsentence \"verify by ...\" attached. If a step lacks\none, the step is ill-specified.\n\n## AP-8: Multi-Step Plan Without Verification Gates\n\n**Maps to**: Principle 4 (Goal-Driven Execution)\n\n**Trigger**: \"Add rate limiting.\" The agent ships 300\nlines with Redis backend, configuration system, and\nmonitoring in one commit, with no per-step\nverification.\n\n**Bad shape**: one large commit covering basic limits,\nmiddleware extraction, Redis integration, and\nconfiguration. Nothing is independently shippable or\nrevertible.\n\n**Good shape**\n\n> Plan, each step independently verifiable:\n>\n> 1. In-memory limit on one endpoint. Verify: 11\n>    requests, the 11th gets 429.\n> 2. Extract to middleware, apply broadly. Verify:\n>    /users and /posts both rate-limit; existing tests\n>    pass.\n> 3. Redis backend. Verify: limits persist across\n>    restarts; two instances share counters.\n> 4. Per-endpoint config. Verify: /search 10/min,\n>    /users 100/min, parsed config tested.\n>\n> Start with step 1?\n\n**Test**: each step in the plan should be revertible\non its own. If reverting step 3 breaks step 4, the\nsteps are not independent and the plan needs a redraw.\n\n## How to Use This Module\n\nWhen reviewing your own diff or someone else's, name\nthe rail you see. \"This is AP-5: drive-by\nrefactoring\" travels faster than \"this could be\nsimpler somehow.\" Naming the rail is the first half\nof fixing the rail.\n\nCross-references for the rails:\n\n- AP-3, AP-4 connect to `Skill(imbue:scope-guard)` and\n  `Skill(leyline:additive-bias-defense)`\n- AP-5, AP-6 connect to `Skill(imbue:justify)` and\n  the `bounded-discovery.md` rule\n- AP-7, AP-8 connect to `Skill(imbue:proof-of-work)`\n  and `Skill(superpowers:test-driven-development)`\n\nFile v1.9.19:modules/senior-engineer-test.md\n\n# The Senior Engineer Test\n\nA three-question battery to apply to your own code\nbefore claiming it is done. The questions stand in\nfor the senior engineer who is not in the room.\n\nAdapted from a self-check Karpathy calls out for\nagentic coding: ask whether a senior engineer would\nsay this is overcomplicated. We expand the question\ninto three concrete sub-questions that map to common\nLLM coding failures.\n\n## The Question\n\n> Would a senior engineer who is busy and a little\n> grumpy say this code is overcomplicated?\n\nIf yes, the diff is not ready.\n\n## The Three Sub-Questions\n\n### Q1: Could this be 50% shorter without losing meaning?\n\nMost LLM-written code can be cut by a third to a\nhalf. If the diff is 200 lines, ask: which 100 lines\nexist because the agent felt clever, not because the\nproblem demanded them?\n\nCommon 50% wins:\n\n- Replace abstract base class plus two subclasses\n  with one function plus a parameter.\n- Replace try-except wrapping every call with a\n  single boundary handler.\n- Replace explicit getter and setter with direct\n  attribute access.\n- Replace nested conditionals with a flat early-return\n  pattern.\n\n### Q2: Are abstractions earning their weight?\n\nAn abstraction earns its weight when it is used three\nor more times, or when it isolates a real boundary\n(network, disk, locale). A class with one consumer is\nceremony. A factory with one product is ceremony.\n\nTest: for every type, class, or helper added, count\nthe call sites. One call site means the abstraction\ncosts more than it saves.\n\n### Q3: Could a junior dev follow this in six months?\n\nSix months means: docs may have rotted, original\ncontext is gone, the original author is on another\nteam. The code has to carry its own meaning.\n\nFailure signals:\n\n- Names that mean something only if you remember the\n  ticket\n- Comments that describe what the code does (the code\n  shows that) instead of why\n- Indirection that requires three jumps to find the\n  actual logic\n- Implicit invariants that nothing checks and nothing\n  documents\n\n## The Decision Tree\n\n```\nFor each of Q1, Q2, Q3:\n  - Yes -> next question\n  - No  -> stop and address before shipping\n\nIf three Yes -> ship\nIf any No   -> rework that dimension first\n```\n\nA No answer is not a failure of the agent; it is the\nagent doing its job. Catching the violation before\nthe senior engineer catches it is the entire point.\n\n## Worked Example\n\nDiff under review: a class hierarchy for a single\ndiscount calculation.\n\n- Q1 (50% shorter)? Yes obviously: one function\n  replaces five classes.\n- Q2 (abstractions earning weight)? No: zero\n  additional call sites for the strategy pattern.\n- Q3 (junior in six months)? No: two indirection hops\n  to find the multiplication.\n\nTwo No answers means rework. Replace the hierarchy\nwith the function. Now Q1, Q2, Q3 are all yes.\n\n## When the Test Does Not Apply\n\nThe senior-engineer test assumes the code will be\nread again. For a one-shot data migration that runs\nonce and is deleted, the test is too strict. See\n`tradeoff-acknowledgment.md` for the boundary cases.\n\n## Cross-References\n\n- `Skill(imbue:scope-guard)` formalizes Q2 (does the\n  abstraction earn its weight) into the Worthiness\n  formula.\n- `Skill(conserve:code-quality-principles)` is the\n  KISS / YAGNI / SOLID foundation that Q1 leans on.\n- `Skill(leyline:additive-bias-defense)` is the\n  burden-of-proof contract that backs Q2.\n\nFile v1.9.19:modules/tradeoff-acknowledgment.md\n\n# Tradeoff Acknowledgment: When Not to Apply These\n\nThe four principles bias toward caution. That bias\ncosts speed. For a substantial portion of coding work\nthe cost is worth paying. For a non-trivial minority,\nthe cost is wrong. This module names the boundary\nhonestly.\n\nThe upstream framing puts it as: \"These guidelines\nbias toward caution over speed. For trivial tasks,\nuse judgment.\" That sentence does the same work as\nthis module, just compressed.\n\n## When the Principles Do Not Apply\n\n### Trivial One-Line Fixes\n\nAsking three clarifying questions before fixing a\ntypo in a comment is a parody of caution. For diffs\nunder five lines with obvious intent, ship and move\non. Principle 1 (Think Before Coding) is for\nambiguous requests, not unambiguous ones.\n\n### Exploratory Spikes and Throwaway Scripts\n\nA 50-line script that runs once, produces a CSV, and\ngets deleted does not need the senior-engineer test.\nIt does not need TDD. It does not need careful\nabstraction analysis. The artifact's lifetime caps\nthe time worth investing in its quality.\n\nTest: if the script will run again next week, treat\nit like real code. If you will throw it away in an\nhour, do not over-invest.\n\n### Documentation-Only Changes\n\nStyle drift in docs is often the point. Rewriting a\nparagraph for clarity touches every line by design.\nPrinciple 3 (Surgical Changes) was written for code\ndiffs, where adjacent edits hide intent. Prose is\ndifferent.\n\n### Time-Boxed Prototypes\n\nA \"by Friday or we move on\" prototype is a different\nartifact from a feature. Verifiable success criteria\nfor a prototype look like \"the demo runs end to\nend,\" not \"the test suite is green.\" Calibrate\nambition to the deadline.\n\n### Production Fires\n\nWhen the database is on fire, \"let's write a failing\ntest first\" is the wrong move. Stop the fire, then\nwrite the test that prevents the next fire. The Iron\nLaw assumes a normal-operations context.\n\n## Contrarian Voices Worth Engaging\n\nThree voices push back on rigorous-by-default LLM\ncoding rules. Their critiques sharpen the boundary.\n\n**Simon Willison** (\"Not all AI-assisted programming\nis vibe coding,\" March 2025) defends throwaway\nprototyping as legitimate. His golden rule: do not\ncommit code you cannot explain. That rule is\ncompatible with everything in this skill, but it\nmakes the throwaway-prototype boundary explicit.\n\n**Mastering Product HQ** (\"What Karpathy's CLAUDE.md\nmisses\") argues code simplicity does not equal scope\nsimplicity. A 50-line solution to the wrong problem\nis still waste. The principles help with how to\nbuild; they do not help with what to build. For\n\"what,\" see `Skill(imbue:scope-guard)` and\n`Skill(imbue:feature-review)`.\n\n**NMN.gl** (\"Vibe Coding Considered Harmful,\" March\n2025) warns that vibed black boxes compound. This is\nadjacent support for the principles, not pushback,\nbut it names the real cost of skipping them at scale:\neach black box you accept becomes a future debugging\nexpense.\n\n## The Honest Bottom Line\n\nThese principles solve a specific class of problem:\nLLM agents shipping wrong-shape code on tasks they\ncould have shipped right with five minutes of\nupfront thought. That class is large. It is not\nuniversal.\n\nIf you find yourself about to invoke these\nprinciples on a task that fits in a sticky note,\nstop. The principles are the heavier path. Use the\nheavier path when the cost of getting it wrong is\nlarger than the cost of slowing down. Otherwise, ship\nand move on.\n\n## Cross-References\n\n- `Skill(imbue:scope-guard)` for \"should we build\n  this at all\" (the scope question this module\n  punts on).\n- `Skill(imbue:feature-review)` for prioritization\n  using RICE / WSJF / Kano scoring.\n- `Skill(conserve:decisive-action)` for guidance on\n  when to skip clarification and proceed.\n\nFile v1.9.19:modules/verifiable-goals.md\n\n# Verifiable Goals: A Reformulation Template\n\nVague tasks generate vague work. The fix is a\nmechanical reformulation: rewrite the request as a\ngoal that has an unambiguous \"done\" signal. Then loop\nuntil the signal fires.\n\nThis module makes the reformulation explicit, with a\ntemplate and worked examples.\n\n## The Template\n\n```\nOriginal request: <user's words>\n\nReformulated goal:\n  Success signal: <something a script or test can check>\n  Test that proves the signal fires: <how>\n  Out-of-scope cleanups noticed: <list, do not fix>\n```\n\nThe success signal must be checkable without human\njudgment. \"It feels faster\" is not a signal.\n\"p95 under 200ms on the seed dataset\" is.\n\n## Worked Examples\n\n### Example 1: \"Add validation\"\n\n```\nOriginal: Add validation to the user signup endpoint.\n\nReformulated:\n  Success signal: requests with invalid email,\n    missing username, or password under 8 chars\n    return HTTP 400 with a JSON error.\n  Test: three pytest cases, one per failure mode,\n    asserting status 400 and a specific error key.\n  Out of scope: rate limiting, password complexity\n    rules, captcha. Mention but do not implement.\n```\n\n### Example 2: \"Fix the bug\"\n\n```\nOriginal: Fix the bug where empty emails crash the\n  validator.\n\nReformulated:\n  Success signal: validate_user with email '' or\n    None raises ValueError, not AttributeError or\n    TypeError.\n  Test: test_validate_user_empty_email and\n    test_validate_user_none_email, both asserting\n    ValueError before the fix lands.\n  Out of scope: improving username validation, adding\n    docstrings, refactoring quote style.\n```\n\nThis pattern is the heart of `Skill(imbue:proof-of-work)`\nand the Iron Law: write the failing test first.\n\n### Example 3: \"Refactor X\"\n\n```\nOriginal: Refactor the upload service.\n\nReformulated:\n  Success signal: the existing test suite for\n    upload (12 tests) is green before the refactor,\n    green after, with no test changes.\n  Test: pytest tests/upload/ both before and after\n    the diff, with diff capture.\n  Out of scope: anything that requires changing a\n    test. If a test must change, the request is\n    behavior change, not refactor, and needs a new\n    spec.\n```\n\n### Example 4: \"Make it faster\"\n\n```\nOriginal: Make the search faster.\n\nReformulated:\n  Success signal: median latency on the seed query\n    set drops from current N ms to under M ms (M\n    chosen with the user).\n  Test: a benchmark script that runs 100 queries\n    against the seed dataset and reports median.\n    Captured before and after the change.\n  Out of scope: throughput optimization, perceived\n    speed, frontend caching. These are different\n    \"faster\" axes. Confirm which one before starting.\n```\n\nThis example also illustrates AP-2 (Multiple\nInterpretations): when \"faster\" is ambiguous, name\nthe axis before reformulating.\n\n### Example 5: \"Improve UX\"\n\n```\nOriginal: Improve the checkout UX.\n\nReformulated:\n  Success signal: a test user completes the checkout\n    flow in 4 clicks or fewer (current: 7), with no\n    blocking validation surprises.\n  Test: a Playwright or manual click-through script\n    that records click count and timestamp per step.\n  Out of scope: visual redesign, copy revisions,\n    accessibility audit. Mention but do not bundle.\n```\n\n### Example 6: \"Add rate limiting\"\n\n```\nOriginal: Add rate limiting to the API.\n\nReformulated:\n  Success signal (step 1): the 11th request to\n    /signup in 60 seconds returns 429.\n  Test: a curl loop in CI plus a pytest that\n    simulates 11 sequential calls.\n  Out of scope (this step): Redis backend,\n    per-endpoint configuration, monitoring. Each\n    is a separate reformulation.\n```\n\nThis example illustrates AP-8 (Multi-Step Plan\nWithout Verification Gates): each step gets its own\nreformulation, its own success signal, its own test.\n\n## Why This Works\n\nWhen the success signal is a script or test, three\nthings become true:\n\n1. The agent can loop independently. No need to ask\n   \"is it good now?\" The test answers.\n2. The user can review by running the test, not by\n   reading 300 lines of diff.\n3. The work is self-documenting. The next person can\n   see what \"done\" meant for this task.\n\n## Cross-References\n\n- `Skill(imbue:proof-of-work)` is the contract that\n  enforces this template under the Iron Law.\n- `Skill(superpowers:test-driven-development)` is the\n  RED-GREEN-REFACTOR loop this template feeds.\n- `/spec-kit:speckit-clarify` command helps when the\n  reformulation surfaces an ambiguity that needs the\n  user's input first.\n\nFile v1.9.19:skill-card.md\n\n## Description:\n\nPre-implementation gate covering think-first, simplicity, surgical edits, and verifiable goals.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[athola](https://clawhub.ai/user/athola)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and coding agents use this skill before, during, or after non-trivial coding work to surface assumptions, keep changes simple and scoped, and define verifiable success criteria.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill can add planning, assumptions, checklist items, and verification criteria to coding tasks even when a user expects a faster path.\n\nMitigation: Narrow broad triggers or invoke the skill explicitly when teams want this pre-flight gate only for non-trivial work.\n\nRisk: The skill references an external plugin and related skills that are separate from the reviewed artifact.\n\nMitigation: Review the referenced plugin and related skills independently before relying on them.\n\nRisk: The artifact is markdown-only guidance, so its value depends on agent interpretation and may over-slow trivial fixes or emergency work.\n\nMitigation: Apply the included tradeoff guidance for trivial fixes, throwaway scripts, documentation-only changes, prototypes, and production incidents.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/athola/skills/nm-imbue-karpathy-principles)\n- [Night Market imbue plugin](https://github.com/athola/claude-night-market/tree/master/plugins/imbue)\n- [Anti-Patterns: Eight Named Drift Rails](artifact/modules/anti-patterns.md)\n- [The Senior Engineer Test](artifact/modules/senior-engineer-test.md)\n- [Tradeoff Acknowledgment: When Not to Apply These](artifact/modules/tradeoff-acknowledgment.md)\n- [Verifiable Goals: A Reformulation Template](artifact/modules/verifiable-goals.md)\n\n## Skill Output:\n\n**Output Type(s):** [Guidance, Markdown, Text, Code review criteria]\n\n**Output Format:** [Markdown guidance with checklists, examples, and concise planning text]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May create planning checklist items and verification criteria around coding tasks.]\n\n## Skill Version(s):\n\n1.9.19 (source: server release metadata; artifact frontmatter lists 1.9.8)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v1.9.17: 7 files, 15067 bytes\n\nFiles: modules/anti-patterns.md (8298b), modules/senior-engineer-test.md (3378b), modules/tradeoff-acknowledgment.md (3757b), modules/verifiable-goals.md (4501b), skill-card.md (2357b), SKILL.md (7513b), _meta.json (148b)\n\nFile v1.9.17:SKILL.md\n\n---\nname: karpathy-principles\ndescription: |\n  Pre-implementation gate covering think-first, simplicity, surgical edits, and verifiable goals\nversion: 1.9.8\ntriggers:\n  - karpathy\n  - coding-pitfalls\n  - synthesis\n  - entry-point\n  - discipline\n  - anti-overengineering\n  - TDD\n  - starting implementation to verify the approach\nmetadata: {\"openclaw\": {\"homepage\": \"https://github.com/athola/claude-night-market/tree/master/plugins/imbue\", \"emoji\": \"\\ud83e\\udd9e\", \"requires\": {\"config\": [\"night-market.imbue:scope-guard\", \"night-market.imbue:proof-of-work\", \"night-market.imbue:rigorous-reasoning\", \"night-market.leyline:additive-bias-defense\", \"night-market.conserve:code-quality-principles\"]}}}\nsource: claude-night-market\nsource_plugin: imbue\n---\n\n> **Night Market Skill** — ported from [claude-night-market/imbue](https://github.com/athola/claude-night-market/tree/master/plugins/imbue). For the full experience with agents, hooks, and commands, install the Claude Code plugin.\n\n\n> The models make wrong assumptions on your behalf and\n> just run along with them without checking. They don't\n> manage their confusion, don't seek clarifications,\n> don't surface inconsistencies, don't present\n> tradeoffs, don't push back when they should.\n>\n> -- Andrej Karpathy, on agentic coding failure modes\n\n## What This Is\n\nA four-principle contract for reducing the most common\nLLM coding pitfalls. Compact entry-point. Each\nprinciple has a deeper-dive skill in night-market;\nthis skill is the index, not the encyclopedia.\n\nDerivation: distilled by Forrest Chang\n(forrestchang/andrej-karpathy-skills, MIT) from\nKarpathy's observations. Full attribution in\n`references/source-attribution.md`.\n\n## When to Use\n\n- Before starting any coding task larger than a typo\n- During code review, to name the failure mode you see\n- After writing a diff, to self-audit before claiming\n  done\n- When training a junior engineer to read agent diffs\n\n## When NOT to Use\n\nThese principles bias toward caution over speed. For\ncases listed in `modules/tradeoff-acknowledgment.md`,\nuse judgment: trivial fixes, exploratory spikes,\ndocumentation-only edits, and time-boxed prototypes.\n\n## The Four Principles\n\n### 1. Think Before Coding\n\n**State assumptions. Surface confusion. Match tone to\nevidence.**\n\n- If multiple interpretations of the request exist,\n  list them. Do not silently pick.\n- If a simpler approach exists, name it. Push back\n  when the simpler path is correct.\n- If something is unclear, stop and ask. Hidden\n  assumptions are the cheapest bug to prevent and the\n  most expensive to find later.\n- Make claims no stronger than the evidence supports.\n  Calibrated tone beats confident hand-waving.\n\nDeep dives: `Skill(imbue:rigorous-reasoning)` for the\nsycophancy guard, `Skill(superpowers:brainstorming)`\nfor option generation, `/spec-kit:speckit-clarify`\ncommand for ambiguity drilldown.\n\n### 2. Simplicity First\n\n**Minimum code that solves the problem. Nothing\nspeculative.**\n\n> They really like to overcomplicate code and APIs,\n> bloat abstractions.\n>\n> -- Andrej Karpathy, on the same agentic-coding thread\n\n\n\n- No features beyond what was asked\n- No abstractions for single-use code\n- No flexibility or configurability that wasn't\n  requested\n- No error handling for impossible scenarios\n- If you wrote 200 lines and it could be 50, rewrite\n  it\n\nSelf-check: would a senior engineer say this is\novercomplicated? See `modules/senior-engineer-test.md`.\n\nDeep dives: `Skill(imbue:scope-guard)` for the\nworthiness formula and branch budgets,\n`Skill(leyline:additive-bias-defense)` for burden of\nproof on every addition,\n`Skill(conserve:code-quality-principles)` for the\nKISS / YAGNI / SOLID foundation.\n\n### 3. Surgical Changes\n\n**Touch only what you must. Clean up only your own\nmess.**\n\n- Do not improve adjacent code, comments, or\n  formatting\n- Do not refactor things that aren't broken\n- Match existing style even when you would do it\n  differently\n- If you notice unrelated dead code, mention it; do\n  not delete it\n- When your changes orphan imports or variables,\n  remove the orphans you created. Pre-existing dead\n  code stays unless asked.\n\nThe trace-back test: every changed line should trace\ndirectly to the user's request.\n\nDeep dives: `Skill(imbue:justify)` for additive-bias\naudits on diffs, `Skill(leyline:additive-bias-defense)`\nfor the burden-of-proof contract, the\n`bounded-discovery.md` rule for read-budget caps.\n\n### 4. Goal-Driven Execution\n\n**Define verifiable success criteria. Loop until\nverified.**\n\nTransform vague tasks into checkable goals:\n\n- \"Add validation\" becomes \"tests for invalid inputs\n  pass\"\n- \"Fix the bug\" becomes \"test reproducing the bug,\n  then make it pass\"\n- \"Refactor X\" becomes \"tests pass before and after\"\n- \"Make it faster\" becomes \"benchmark Y under N ms\"\n\nFor multi-step tasks, state a brief plan with\nverification per step. Strong success criteria let\nyou loop independently. Weak criteria require\nconstant clarification.\n\nSee `modules/verifiable-goals.md` for the full\nreformulation template.\n\nDeep dives: `Skill(imbue:proof-of-work)` for the Iron\nLaw (no implementation without a failing test first),\n`Skill(superpowers:test-driven-development)` for the\nRED-GREEN-REFACTOR loop.\n\n## The Karpathy Self-Check\n\nBefore you ship, four questions:\n\n| Principle | Question |\n|-----------|----------|\n| Think Before Coding | Did I list assumptions, or did I guess silently? |\n| Simplicity First | Would a senior engineer call this overcomplicated? |\n| Surgical Changes | Does every changed line trace to the request? |\n| Goal-Driven Execution | Can I prove this is done with a check, not a feeling? |\n\nFour \"yes\" answers means ship. Anything else means\niterate.\n\n## Modules\n\n- `modules/anti-patterns.md` - eight named drift rails\n  with before/after diffs\n- `modules/senior-engineer-test.md` - the\n  three-question self-check battery\n- `modules/verifiable-goals.md` - vague-to-verifiable\n  reformulation template with worked examples\n- `modules/tradeoff-acknowledgment.md` - when the four\n  principles do not apply\n\n## References\n\n- `references/source-attribution.md` - Karpathy\n  primary citation, Forrest Chang derivation, license,\n  adjacent prior art\n\n## Related Skills\n\n- `Skill(imbue:scope-guard)` - worthiness formula and\n  branch budgets\n- `Skill(imbue:proof-of-work)` - Iron Law TDD gate\n- `Skill(imbue:rigorous-reasoning)` - sycophancy and\n  hidden-assumption guard\n- `Skill(imbue:justify)` - additive-bias diff audit\n- `Skill(leyline:additive-bias-defense)` - burden of\n  proof on every addition\n- `Skill(conserve:code-quality-principles)` - KISS,\n  YAGNI, SOLID\n- `Skill(superpowers:test-driven-development)` -\n  RED-GREEN-REFACTOR\n- `Skill(superpowers:brainstorming)` - generate\n  options before committing\n- See `docs/quality-gates.md#skill-level-quality-gate-composition`\n  for the full gate-skill federation graph (this skill\n  is the synthesis hub)\n\n## Required TodoWrite Items\n\nWhen invoked as a pre-flight gate, create:\n\n- `karpathy:assumptions-listed` - principle 1 satisfied\n- `karpathy:simplicity-checked` - principle 2 satisfied\n- `karpathy:trace-back-verified` - principle 3 satisfied\n- `karpathy:success-criteria-defined` - principle 4\n  satisfied\n\n## Exit Criteria\n\n- Each of the four principles has been answered with a\n  concrete artifact (assumption list, scope rationale,\n  diff trace, verification plan).\n- The senior-engineer test was applied at least once.\n- Verifiable success criteria are written down before\n  the implementation begins.\n\nFile v1.9.17:_meta.json\n\n{\n  \"ownerId\": \"kn7d107jg9jv602h9ytsegydq184a42s\",\n  \"slug\": \"nm-imbue-karpathy-principles\",\n  \"version\": \"1.9.17\",\n  \"publishedAt\": 1785389659068\n}\n\nFile v1.9.17:modules/anti-patterns.md\n\n# Anti-Patterns: Eight Named Drift Rails\n\nEach drift rail names a specific way LLM-driven coding\ngoes wrong, with a trigger pattern, a short bad\nexample, a short good example, and a one-sentence\ntest you can apply to your own diff. Naming the rail\nhelps recall.\n\nAdapted from observations distilled by Forrest Chang\nfrom Karpathy's notes; see\n`references/source-attribution.md` for the citation\nchain. Examples are rewritten in our own prose to\nillustrate the failure shape, not lifted verbatim.\n\n## AP-1: Hidden Assumptions\n\n**Maps to**: Principle 1 (Think Before Coding)\n\n**Trigger**: User asks for a feature with multiple\nunstated dimensions (scope, format, volume, privacy).\nThe agent picks one silently and ships.\n\n**Bad shape**\n\n```python\ndef export_users(format='json'):\n    users = User.query.all()\n    if format == 'json':\n        with open('users.json', 'w') as f:\n            json.dump([u.to_dict() for u in users], f)\n```\n\nAssumed: all users (privacy?), file path, fields,\nvolume.\n\n**Good shape**\n\n> Before implementing, I want to confirm:\n>\n> - Scope: all users, or filtered?\n> - Format: download, background job, or API endpoint?\n> - Fields: which? Some are sensitive.\n> - Volume: thousands or millions?\n>\n> Default proposal: a paginated JSON endpoint. Adjust?\n\n**Test**: count silent assumptions in the diff. Three\nor more is a rail violation.\n\n## AP-2: Multiple Interpretations Picked Silently\n\n**Maps to**: Principle 1 (Think Before Coding)\n\n**Trigger**: A request like \"make the search faster\"\nhas at least three valid readings (response time,\nthroughput, perceived speed). The agent picks one and\nimplements 200 lines without flagging the choice.\n\n**Bad shape**\n\n```python\n@lru_cache(maxsize=1000)\nasync def search(query: str) -> List[Result]:\n    # 200 lines of caching, async, indexes, all picked\n    # without confirming what \"faster\" means\n    ...\n```\n\n**Good shape**\n\n> \"Faster\" could mean response time, throughput, or\n> perceived speed. Each is a different fix:\n>\n> 1. Response time: indexes plus query caching (~2h)\n> 2. Throughput: async path plus pooling (~4h)\n> 3. Perceived: progressive load and partial results\n>    (~3h)\n>\n> Current: ~500ms typical. Which axis matters?\n\n**Test**: when the verb in the request is ambiguous\n(faster, better, cleaner, simpler), did the agent name\nthe alternatives or pick one?\n\n## AP-3: Strategy Pattern for One Function\n\n**Maps to**: Principle 2 (Simplicity First)\n\n**Trigger**: User asks for a single function. The\nagent ships an abstract base class, two implementing\nclasses, a config dataclass, and a coordinator class\nfor ten lines of arithmetic.\n\n**Bad shape**\n\n```python\nclass DiscountStrategy(ABC):\n    @abstractmethod\n    def calculate(self, amount: float) -> float: ...\n\nclass PercentageDiscount(DiscountStrategy):\n    def __init__(self, p): self.p = p\n    def calculate(self, a): return a * (self.p / 100)\n\n# Plus FixedDiscount, DiscountConfig, DiscountCalculator\n# for what should be one function\n```\n\n**Good shape**\n\n```python\ndef calculate_discount(amount: float, percent: float) -> float:\n    return amount * (percent / 100)\n```\n\n**Test**: count types and classes added per actual use\ncase. If types-added exceeds use-cases-served, the\npattern is premature.\n\n## AP-4: Speculative Features\n\n**Maps to**: Principle 2 (Simplicity First)\n\n**Trigger**: \"Save user preferences to database\"\nbecomes a class with optional caching, validation,\nnotification hooks, and merge semantics. None were\nasked for.\n\n**Bad shape**\n\n```python\nclass PreferenceManager:\n    def __init__(self, db, cache=None, validator=None):\n        ...\n    def save(self, user_id, prefs,\n             merge=True, validate=True, notify=False):\n        # 60 lines of optional behavior\n```\n\n**Good shape**\n\n```python\ndef save_preferences(db, user_id: int, preferences: dict):\n    db.execute(\n        \"UPDATE users SET preferences = ? WHERE id = ?\",\n        (json.dumps(preferences), user_id),\n    )\n```\n\n**Test**: list every parameter that was not in the\nrequest. If you cannot point at a sentence in the\nrequest that demanded it, delete the parameter.\n\n## AP-5: Drive-by Refactoring\n\n**Maps to**: Principle 3 (Surgical Changes)\n\n**Trigger**: User reports a single bug. The agent\nfixes the bug, then \"improves\" three other functions,\nadds docstrings, and tightens validation logic that\nnobody asked about.\n\n**Bad shape**: a 90-line diff to fix a 4-line bug,\nwith related but unrequested cleanups across two more\nfiles.\n\n**Good shape**: a 4-line diff that fixes only the\nreported bug. If you noticed unrelated issues, list\nthem in the response and ask before touching them.\n\n**Test**: read the diff line by line. For each\nchanged line, ask \"which sentence in the user's\nrequest demanded this?\" Lines without an answer are\ncandidates for removal from the diff.\n\n## AP-6: Style Drift During Edit\n\n**Maps to**: Principle 3 (Surgical Changes)\n\n**Trigger**: User asks for one logging line in an\nupload function. The agent ships type hints,\ndocstrings, single-quote-to-double-quote conversion,\nand a flattened control flow.\n\n**Bad shape**\n\n```diff\n- def upload_file(file_path, destination):\n+ def upload_file(file_path: str, destination: str) -> bool:\n+     \"\"\"Upload file to destination.\"\"\"\n      try:\n-         with open(file_path, 'rb') as f:\n+         with open(file_path, \"rb\") as f:\n              ...\n```\n\n**Good shape**\n\n```diff\n+ logger = logging.getLogger(__name__)\n+\n  def upload_file(file_path, destination):\n+     logger.info(f'Starting upload: {file_path}')\n      try:\n          with open(file_path, 'rb') as f:\n              ...\n```\n\n**Test**: the diff should not change quote style,\ntype hint presence, docstring presence, or whitespace\npatterns unless the request named them.\n\n## AP-7: Vague Success Criteria\n\n**Maps to**: Principle 4 (Goal-Driven Execution)\n\n**Trigger**: \"Fix the authentication system.\" The\nagent responds \"I will review the code, identify\nissues, make improvements, and test.\" That is a plan\nshape, not a verification plan.\n\n**Bad shape**: a four-step list of verbs (review,\nidentify, improve, test) with no checkable signal per\nstep.\n\n**Good shape**\n\n> \"Fix authentication\" is too broad to verify. Best\n> guess at the actual problem: sessions stay alive\n> after password change. Plan:\n>\n> 1. Write failing test: change password, prior\n>    session is rejected.\n> 2. Implement: invalidate sessions on change.\n> 3. Edge cases: concurrent change, multi-session.\n> 4. Regression: full auth suite green.\n>\n> Is the password-change-stale-session bug the right\n> target?\n\n**Test**: each step in the plan should have a one-\nsentence \"verify by ...\" attached. If a step lacks\none, the step is ill-specified.\n\n## AP-8: Multi-Step Plan Without Verification Gates\n\n**Maps to**: Principle 4 (Goal-Driven Execution)\n\n**Trigger**: \"Add rate limiting.\" The agent ships 300\nlines with Redis backend, configuration system, and\nmonitoring in one commit, with no per-step\nverification.\n\n**Bad shape**: one large commit covering basic limits,\nmiddleware extraction, Redis integration, and\nconfiguration. Nothing is independently shippable or\nrevertible.\n\n**Good shape**\n\n> Plan, each step independently verifiable:\n>\n> 1. In-memory limit on one endpoint. Verify: 11\n>    requests, the 11th gets 429.\n> 2. Extract to middleware, apply broadly. Verify:\n>    /users and /posts both rate-limit; existing tests\n>    pass.\n> 3. Redis backend. Verify: limits persist across\n>    restarts; two instances share counters.\n> 4. Per-endpoint config. Verify: /search 10/min,\n>    /users 100/min, parsed config tested.\n>\n> Start with step 1?\n\n**Test**: each step in the plan should be revertible\non its own. If reverting step 3 breaks step 4, the\nsteps are not independent and the plan needs a redraw.\n\n## How to Use This Module\n\nWhen reviewing your own diff or someone else's, name\nthe rail you see. \"This is AP-5: drive-by\nrefactoring\" travels faster than \"this could be\nsimpler somehow.\" Naming the rail is the first half\nof fixing the rail.\n\nCross-references for the rails:\n\n- AP-3, AP-4 connect to `Skill(imbue:scope-guard)` and\n  `Skill(leyline:additive-bias-defense)`\n- AP-5, AP-6 connect to `Skill(imbue:justify)` and\n  the `bounded-discovery.md` rule\n- AP-7, AP-8 connect to `Skill(imbue:proof-of-work)`\n  and `Skill(superpowers:test-driven-development)`\n\nFile v1.9.17:modules/senior-engineer-test.md\n\n# The Senior Engineer Test\n\nA three-question battery to apply to your own code\nbefore claiming it is done. The questions stand in\nfor the senior engineer who is not in the room.\n\nAdapted from a self-check Karpathy calls out for\nagentic coding: ask whether a senior engineer would\nsay this is overcomplicated. We expand the question\ninto three concrete sub-questions that map to common\nLLM coding failures.\n\n## The Question\n\n> Would a senior engineer who is busy and a little\n> grumpy say this code is overcomplicated?\n\nIf yes, the diff is not ready.\n\n## The Three Sub-Questions\n\n### Q1: Could this be 50% shorter without losing meaning?\n\nMost LLM-written code can be cut by a third to a\nhalf. If the diff is 200 lines, ask: which 100 lines\nexist because the agent felt clever, not because the\nproblem demanded them?\n\nCommon 50% wins:\n\n- Replace abstract base class plus two subclasses\n  with one function plus a parameter.\n- Replace try-except wrapping every call with a\n  single boundary handler.\n- Replace explicit getter and setter with direct\n  attribute access.\n- Replace nested conditionals with a flat early-return\n  pattern.\n\n### Q2: Are abstractions earning their weight?\n\nAn abstraction earns its weight when it is used three\nor more times, or when it isolates a real boundary\n(network, disk, locale). A class with one consumer is\nceremony. A factory with one product is ceremony.\n\nTest: for every type, class, or helper added, count\nthe call sites. One call site means the abstraction\ncosts more than it saves.\n\n### Q3: Could a junior dev follow this in six months?\n\nSix months means: docs may have rotted, original\ncontext is gone, the original author is on another\nteam. The code has to carry its own meaning.\n\nFailure signals:\n\n- Names that mean something only if you remember the\n  ticket\n- Comments that describe what the code does (the code\n  shows that) instead of why\n- Indirection that requires three jumps to find the\n  actual logic\n- Implicit invariants that nothing checks and nothing\n  documents\n\n## The Decision Tree\n\n```\nFor each of Q1, Q2, Q3:\n  - Yes -> next question\n  - No  -> stop and address before shipping\n\nIf three Yes -> ship\nIf any No   -> rework that dimension first\n```\n\nA No answer is not a failure of the agent; it is the\nagent doing its job. Catching the violation before\nthe senior engineer catches it is the entire point.\n\n## Worked Example\n\nDiff under review: a class hierarchy for a single\ndiscount calculation.\n\n- Q1 (50% shorter)? Yes obviously: one function\n  replaces five classes.\n- Q2 (abstractions earning weight)? No: zero\n  additional call sites for the strategy pattern.\n- Q3 (junior in six months)? No: two indirection hops\n  to find the multiplication.\n\nTwo No answers means rework. Replace the hierarchy\nwith the function. Now Q1, Q2, Q3 are all yes.\n\n## When the Test Does Not Apply\n\nThe senior-engineer test assumes the code will be\nread again. For a one-shot data migration that runs\nonce and is deleted, the test is too strict. See\n`tradeoff-acknowledgment.md` for the boundary cases.\n\n## Cross-References\n\n- `Skill(imbue:scope-guard)` formalizes Q2 (does the\n  abstraction earn its weight) into the Worthiness\n  formula.\n- `Skill(conserve:code-quality-principles)` is the\n  KISS / YAGNI / SOLID foundation that Q1 leans on.\n- `Skill(leyline:additive-bias-defense)` is the\n  burden-of-proof contract that backs Q2.\n\nFile v1.9.17:modules/tradeoff-acknowledgment.md\n\n# Tradeoff Acknowledgment: When Not to Apply These\n\nThe four principles bias toward caution. That bias\ncosts speed. For a substantial portion of coding work\nthe cost is worth paying. For a non-trivial minority,\nthe cost is wrong. This module names the boundary\nhonestly.\n\nThe upstream framing puts it as: \"These guidelines\nbias toward caution over speed. For trivial tasks,\nuse judgment.\" That sentence does the same work as\nthis module, just compressed.\n\n## When the Principles Do Not Apply\n\n### Trivial One-Line Fixes\n\nAsking three clarifying questions before fixing a\ntypo in a comment is a parody of caution. For diffs\nunder five lines with obvious intent, ship and move\non. Principle 1 (Think Before Coding) is for\nambiguous requests, not unambiguous ones.\n\n### Exploratory Spikes and Throwaway Scripts\n\nA 50-line script that runs once, produces a CSV, and\ngets deleted does not need the senior-engineer test.\nIt does not need TDD. It does not need careful\nabstraction analysis. The artifact's lifetime caps\nthe time worth investing in its quality.\n\nTest: if the script will run again next week, treat\nit like real code. If you will throw it away in an\nhour, do not over-invest.\n\n### Documentation-Only Changes\n\nStyle drift in docs is often the point. Rewriting a\nparagraph for clarity touches every line by design.\nPrinciple 3 (Surgical Changes) was written for code\ndiffs, where adjacent edits hide intent. Prose is\ndifferent.\n\n### Time-Boxed Prototypes\n\nA \"by Friday or we move on\" prototype is a different\nartifact from a feature. Verifiable success criteria\nfor a prototype look like \"the demo runs end to\nend,\" not \"the test suite is green.\" Calibrate\nambition to the deadline.\n\n### Production Fires\n\nWhen the database is on fire, \"let's write a failing\ntest first\" is the wrong move. Stop the fire, then\nwrite the test that prevents the next fire. The Iron\nLaw assumes a normal-operations context.\n\n## Contrarian Voices Worth Engaging\n\nThree voices push back on rigorous-by-default LLM\ncoding rules. Their critiques sharpen the boundary.\n\n**Simon Willison** (\"Not all AI-assisted programming\nis vibe coding,\" March 2025) defends throwaway\nprototyping as legitimate. His golden rule: do not\ncommit code you cannot explain. That rule is\ncompatible with everything in this skill, but it\nmakes the throwaway-prototype boundary explicit.\n\n**Mastering Product HQ** (\"What Karpathy's CLAUDE.md\nmisses\") argues code simplicity does not equal scope\nsimplicity. A 50-line solution to the wrong problem\nis still waste. The principles help with how to\nbuild; they do not help with what to build. For\n\"what,\" see `Skill(imbue:scope-guard)` and\n`Skill(imbue:feature-review)`.\n\n**NMN.gl** (\"Vibe Coding Considered Harmful,\" March\n2025) warns that vibed black boxes compound. This is\nadjacent support for the principles, not pushback,\nbut it names the real cost of skipping them at scale:\neach black box you accept becomes a future debugging\nexpense.\n\n## The Honest Bottom Line\n\nThese principles solve a specific class of problem:\nLLM agents shipping wrong-shape code on tasks they\ncould have shipped right with five minutes of\nupfront thought. That class is large. It is not\nuniversal.\n\nIf you find yourself about to invoke these\nprinciples on a task that fits in a sticky note,\nstop. The principles are the heavier path. Use the\nheavier path when the cost of getting it wrong is\nlarger than the cost of slowing down. Otherwise, ship\nand move on.\n\n## Cross-References\n\n- `Skill(imbue:scope-guard)` for \"should we build\n  this at all\" (the scope question this module\n  punts on).\n- `Skill(imbue:feature-review)` for prioritization\n  using RICE / WSJF / Kano scoring.\n- `Skill(conserve:decisive-action)` for guidance on\n  when to skip clarification and proceed.\n\nFile v1.9.17:modules/verifiable-goals.md\n\n# Verifiable Goals: A Reformulation Template\n\nVague tasks generate vague work. The fix is a\nmechanical reformulation: rewrite the request as a\ngoal that has an unambiguous \"done\" signal. Then loop\nuntil the signal fires.\n\nThis module makes the reformulation explicit, with a\ntemplate and worked examples.\n\n## The Template\n\n```\nOriginal request: <user's words>\n\nReformulated goal:\n  Success signal: <something a script or test can check>\n  Test that proves the signal fires: <how>\n  Out-of-scope cleanups noticed: <list, do not fix>\n```\n\nThe success signal must be checkable without human\njudgment. \"It feels faster\" is not a signal.\n\"p95 under 200ms on the seed dataset\" is.\n\n## Worked Examples\n\n### Example 1: \"Add validation\"\n\n```\nOriginal: Add validation to the user signup endpoint.\n\nReformulated:\n  Success signal: requests with invalid email,\n    missing username, or password under 8 chars\n    return HTTP 400 with a JSON error.\n  Test: three pytest cases, one per failure mode,\n    asserting status 400 and a specific error key.\n  Out of scope: rate limiting, password complexity\n    rules, captcha. Mention but do not implement.\n```\n\n### Example 2: \"Fix the bug\"\n\n```\nOriginal: Fix the bug where empty emails crash the\n  validator.\n\nReformulated:\n  Success signal: validate_user with email '' or\n    None raises ValueError, not AttributeError or\n    TypeError.\n  Test: test_validate_user_empty_email and\n    test_validate_user_none_email, both asserting\n    ValueError before the fix lands.\n  Out of scope: improving username validation, adding\n    docstrings, refactoring quote style.\n```\n\nThis pattern is the heart of `Skill(imbue:proof-of-work)`\nand the Iron Law: write the failing test first.\n\n### Example 3: \"Refactor X\"\n\n```\nOriginal: Refactor the upload service.\n\nReformulated:\n  Success signal: the existing test suite for\n    upload (12 tests) is green before the refactor,\n    green after, with no test changes.\n  Test: pytest tests/upload/ both before and after\n    the diff, with diff capture.\n  Out of scope: anything that requires changing a\n    test. If a test must change, the request is\n    behavior change, not refactor, and needs a new\n    spec.\n```\n\n### Example 4: \"Make it faster\"\n\n```\nOriginal: Make the search faster.\n\nReformulated:\n  Success signal: median latency on the seed query\n    set drops from current N ms to under M ms (M\n    chosen with the user).\n  Test: a benchmark script that runs 100 queries\n    against the seed dataset and reports median.\n    Captured before and after the change.\n  Out of scope: throughput optimization, perceived\n    speed, frontend caching. These are different\n    \"faster\" axes. Confirm which one before starting.\n```\n\nThis example also illustrates AP-2 (Multiple\nInterpretations): when \"faster\" is ambiguous, name\nthe axis before reformulating.\n\n### Example 5: \"Improve UX\"\n\n```\nOriginal: Improve the checkout UX.\n\nReformulated:\n  Success signal: a test user completes the checkout\n    flow in 4 clicks or fewer (current: 7), with no\n    blocking validation surprises.\n  Test: a Playwright or manual click-through script\n    that records click count and timestamp per step.\n  Out of scope: visual redesign, copy revisions,\n    accessibility audit. Mention but do not bundle.\n```\n\n### Example 6: \"Add rate limiting\"\n\n```\nOriginal: Add rate limiting to the API.\n\nReformulated:\n  Success signal (step 1): the 11th request to\n    /signup in 60 seconds returns 429.\n  Test: a curl loop in CI plus a pytest that\n    simulates 11 sequential calls.\n  Out of scope (this step): Redis backend,\n    per-endpoint configuration, monitoring. Each\n    is a separate reformulation.\n```\n\nThis example illustrates AP-8 (Multi-Step Plan\nWithout Verification Gates): each step gets its own\nreformulation, its own success signal, its own test.\n\n## Why This Works\n\nWhen the success signal is a script or test, three\nthings become true:\n\n1. The agent can loop independently. No need to ask\n   \"is it good now?\" The test answers.\n2. The user can review by running the test, not by\n   reading 300 lines of diff.\n3. The work is self-documenting. The next person can\n   see what \"done\" meant for this task.\n\n## Cross-References\n\n- `Skill(imbue:proof-of-work)` is the contract that\n  enforces this template under the Iron Law.\n- `Skill(superpowers:test-driven-development)` is the\n  RED-GREEN-REFACTOR loop this template feeds.\n- `/spec-kit:speckit-clarify` command helps when the\n  reformulation surfaces an ambiguity that needs the\n  user's input first.\n\nFile v1.9.17:skill-card.md\n\n## Description: <br>\nPre-implementation gate covering think-first, simplicity, surgical edits, and verifiable goals. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[athola](https://clawhub.ai/user/athola) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and coding agents use this skill as a pre-flight and self-review gate for non-trivial coding work, especially when a task needs assumptions surfaced, scope constrained, changes kept surgical, and success criteria made verifiable. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill can slow down trivial fixes, exploratory spikes, documentation-only changes, time-boxed prototypes, or production-fire response. <br>\nMitigation: Apply the documented tradeoff guidance and skip or lighten the gate when the cost of extra deliberation is higher than the risk of a wrong-shaped change. <br>\nRisk: Generic triggers may invoke the skill more often than intended. <br>\nMitigation: Review the trigger set during installation and narrow triggers if the gate interrupts normal workflow. <br>\nRisk: The skill's recommendations can shape code changes even though they are guidance rather than guarantees. <br>\nMitigation: Review proposed assumptions, scope choices, diffs, and verification steps before relying on them for production work. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/athola/skills/nm-imbue-karpathy-principles) <br>\n- [OpenClaw homepage](https://github.com/athola/claude-night-market/tree/master/plugins/imbue) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [guidance, text, markdown] <br>\n**Output Format:** [Markdown guidance with checklists, review questions, and worked examples] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Documentation-only skill; no hidden execution, data access, or persistence was identified in the server security evidence.] <br>\n\n## Skill Version(s): <br>\n1.9.17 (source: server release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.9.16: 7 files, 14968 bytes\n\nFiles: modules/anti-patterns.md (8298b), modules/senior-engineer-test.md (3378b), modules/tradeoff-acknowledgment.md (3757b), modules/verifiable-goals.md (4501b), skill-card.md (2159b), SKILL.md (7513b), _meta.json (148b)\n\nFile v1.9.16:SKILL.md\n\n---\nname: karpathy-principles\ndescription: |\n  Pre-implementation gate covering think-first, simplicity, surgical edits, and verifiable goals\nversion: 1.9.8\ntriggers:\n  - karpathy\n  - coding-pitfalls\n  - synthesis\n  - entry-point\n  - discipline\n  - anti-overengineering\n  - TDD\n  - starting implementation to verify the approach\nmetadata: {\"openclaw\": {\"homepage\": \"https://github.com/athola/claude-night-market/tree/master/plugins/imbue\", \"emoji\": \"\\ud83e\\udd9e\", \"requires\": {\"config\": [\"night-market.imbue:scope-guard\", \"night-market.imbue:proof-of-work\", \"night-market.imbue:rigorous-reasoning\", \"night-market.leyline:additive-bias-defense\", \"night-market.conserve:code-quality-principles\"]}}}\nsource: claude-night-market\nsource_plugin: imbue\n---\n\n> **Night Market Skill** — ported from [claude-night-market/imbue](https://github.com/athola/claude-night-market/tree/master/plugins/imbue). For the full experience with agents, hooks, and commands, install the Claude Code plugin.\n\n\n> The models make wrong assumptions on your behalf and\n> just run along with them without checking. They don't\n> manage their confusion, don't seek clarifications,\n> don't surface inconsistencies, don't present\n> tradeoffs, don't push back when they should.\n>\n> -- Andrej Karpathy, on agentic coding failure modes\n\n## What This Is\n\nA four-principle contract for reducing the most common\nLLM coding pitfalls. Compact entry-point. Each\nprinciple has a deeper-dive skill in night-market;\nthis skill is the index, not the encyclopedia.\n\nDerivation: distilled by Forrest Chang\n(forrestchang/andrej-karpathy-skills, MIT) from\nKarpathy's observations. Full attribution in\n`references/source-attribution.md`.\n\n## When to Use\n\n- Before starting any coding task larger than a typo\n- During code review, to name the failure mode you see\n- After writing a diff, to self-audit before claiming\n  done\n- When training a junior engineer to read agent diffs\n\n## When NOT to Use\n\nThese principles bias toward caution over speed. For\ncases listed in `modules/tradeoff-acknowledgment.md`,\nuse judgment: trivial fixes, exploratory spikes,\ndocumentation-only edits, and time-boxed prototypes.\n\n## The Four Principles\n\n### 1. Think Before Coding\n\n**State assumptions. Surface confusion. Match tone to\nevidence.**\n\n- If multiple interpretations of the request exist,\n  list them. Do not silently pick.\n- If a simpler approach exists, name it. Push back\n  when the simpler path is correct.\n- If something is unclear, stop and ask. Hidden\n  assumptions are the cheapest bug to prevent and the\n  most expensive to find later.\n- Make claims no stronger than the evidence supports.\n  Calibrated tone beats confident hand-waving.\n\nDeep dives: `Skill(imbue:rigorous-reasoning)` for the\nsycophancy guard, `Skill(superpowers:brainstorming)`\nfor option generation, `/spec-kit:speckit-clarify`\ncommand for ambiguity drilldown.\n\n### 2. Simplicity First\n\n**Minimum code that solves the problem. Nothing\nspeculative.**\n\n> They really like to overcomplicate code and APIs,\n> bloat abstractions.\n>\n> -- Andrej Karpathy, on the same agentic-coding thread\n\n\n\n- No features beyond what was asked\n- No abstractions for single-use code\n- No flexibility or configurability that wasn't\n  requested\n- No error handling for impossible scenarios\n- If you wrote 200 lines and it could be 50, rewrite\n  it\n\nSelf-check: would a senior engineer say this is\novercomplicated? See `modules/senior-engineer-test.md`.\n\nDeep dives: `Skill(imbue:scope-guard)` for the\nworthiness formula and branch budgets,\n`Skill(leyline:additive-bias-defense)` for burden of\nproof on every addition,\n`Skill(conserve:code-quality-principles)` for the\nKISS / YAGNI / SOLID foundation.\n\n### 3. Surgical Changes\n\n**Touch only what you must. Clean up only your own\nmess.**\n\n- Do not improve adjacent code, comments, or\n  formatting\n- Do not refactor things that aren't broken\n- Match existing style even when you would do it\n  differently\n- If you notice unrelated dead code, mention it; do\n  not delete it\n- When your changes orphan imports or variables,\n  remove the orphans you created. Pre-existing dead\n  code stays unless asked.\n\nThe trace-back test: every changed line should trace\ndirectly to the user's request.\n\nDeep dives: `Skill(imbue:justify)` for additive-bias\naudits on diffs, `Skill(leyline:additive-bias-defense)`\nfor the burden-of-proof contract, the\n`bounded-discovery.md` rule for read-budget caps.\n\n### 4. Goal-Driven Execution\n\n**Define verifiable success criteria. Loop until\nverified.**\n\nTransform vague tasks into checkable goals:\n\n- \"Add validation\" becomes \"tests for invalid inputs\n  pass\"\n- \"Fix the bug\" becomes \"test reproducing the bug,\n  then make it pass\"\n- \"Refactor X\" becomes \"tests pass before and after\"\n- \"Make it faster\" becomes \"benchmark Y under N ms\"\n\nFor multi-step tasks, state a brief plan with\nverification per step. Strong success criteria let\nyou loop independently. Weak criteria require\nconstant clarification.\n\nSee `modules/verifiable-goals.md` for the full\nreformulation template.\n\nDeep dives: `Skill(imbue:proof-of-work)` for the Iron\nLaw (no implementation without a failing test first),\n`Skill(superpowers:test-driven-development)` for the\nRED-GREEN-REFACTOR loop.\n\n## The Karpathy Self-Check\n\nBefore you ship, four questions:\n\n| Principle | Question |\n|-----------|----------|\n| Think Before Coding | Did I list assumptions, or did I guess silently? |\n| Simplicity First | Would a senior engineer call this overcomplicated? |\n| Surgical Changes | Does every changed line trace to the request? |\n| Goal-Driven Execution | Can I prove this is done with a check, not a feeling? |\n\nFour \"yes\" answers means ship. Anything else means\niterate.\n\n## Modules\n\n- `modules/anti-patterns.md` - eight named drift rails\n  with before/after diffs\n- `modules/senior-engineer-test.md` - the\n  three-question self-check battery\n- `modules/verifiable-goals.md` - vague-to-verifiable\n  reformulation template with worked examples\n- `modules/tradeoff-acknowledgment.md` - when the four\n  principles do not apply\n\n## References\n\n- `references/source-attribution.md` - Karpathy\n  primary citation, Forrest Chang derivation, license,\n  adjacent prior art\n\n## Related Skills\n\n- `Skill(imbue:scope-guard)` - worthiness formula and\n  branch budgets\n- `Skill(imbue:proof-of-work)` - Iron Law TDD gate\n- `Skill(imbue:rigorous-reasoning)` - sycophancy and\n  hidden-assumption guard\n- `Skill(imbue:justify)` - additive-bias diff audit\n- `Skill(leyline:additive-bias-defense)` - burden of\n  proof on every addition\n- `Skill(conserve:code-quality-principles)` - KISS,\n  YAGNI, SOLID\n- `Skill(superpowers:test-driven-development)` -\n  RED-GREEN-REFACTOR\n- `Skill(superpowers:brainstorming)` - generate\n  options before committing\n- See `docs/quality-gates.md#skill-level-quality-gate-composition`\n  for the full gate-skill federation graph (this skill\n  is the synthesis hub)\n\n## Required TodoWrite Items\n\nWhen invoked as a pre-flight gate, create:\n\n- `karpathy:assumptions-listed` - principle 1 satisfied\n- `karpathy:simplicity-checked` - principle 2 satisfied\n- `karpathy:trace-back-verified` - principle 3 satisfied\n- `karpathy:success-criteria-defined` - principle 4\n  satisfied\n\n## Exit Criteria\n\n- Each of the four principles has been answered with a\n  concrete artifact (assumption list, scope rationale,\n  diff trace, verification plan).\n- The senior-engineer test was applied at least once.\n- Verifiable success criteria are written down before\n  the implementation begins.\n\nFile v1.9.16:_meta.json\n\n{\n  \"ownerId\": \"kn7d107jg9jv602h9ytsegydq184a42s\",\n  \"slug\": \"nm-imbue-karpathy-principles\",\n  \"version\": \"1.9.16\",\n  \"publishedAt\": 1784058662878\n}\n\nFile v1.9.16:modules/anti-patterns.md\n\n# Anti-Patterns: Eight Named Drift Rails\n\nEach drift rail names a specific way LLM-driven coding\ngoes wrong, with a trigger pattern, a short bad\nexample, a short good example, and a one-sentence\ntest you can apply to your own diff. Naming the rail\nhelps recall.\n\nAdapted from observations distilled by Forrest Chang\nfrom Karpathy's notes; see\n`references/source-attribution.md` for the citation\nchain. Examples are rewritten in our own prose to\nillustrate the failure shape, not lifted verbatim.\n\n## AP-1: Hidden Assumptions\n\n**Maps to**: Principle 1 (Think Before Coding)\n\n**Trigger**: User asks for a feature with multiple\nunstated dimensions (scope, format, volume, privacy).\nThe agent picks one silently and ships.\n\n**Bad shape**\n\n```python\ndef export_users(format='json'):\n    users = User.query.all()\n    if format == 'json':\n        with open('users.json', 'w') as f:\n            json.dump([u.to_dict() for u in users], f)\n```\n\nAssumed: all users (privacy?), file path, fields,\nvolume.\n\n**Good shape**\n\n> Before implementing, I want to confirm:\n>\n> - Scope: all users, or filtered?\n> - Format: download, background job, or API endpoint?\n> - Fields: which? Some are sensitive.\n> - Volume: thousands or millions?\n>\n> Default proposal: a paginated JSON endpoint. Adjust?\n\n**Test**: count silent assumptions in the diff. Three\nor more is a rail violation.\n\n## AP-2: Multiple Interpretations Picked Silently\n\n**Maps to**: Principle 1 (Think Before Coding)\n\n**Trigger**: A request like \"make the search faster\"\nhas at least three valid readings (response time,\nthroughput, perceived speed). The agent picks one and\nimplements 200 lines without flagging the choice.\n\n**Bad shape**\n\n```python\n@lru_cache(maxsize=1000)\nasync def search(query: str) -> List[Result]:\n    # 200 lines of caching, async, indexes, all picked\n    # without confirming what \"faster\" means\n    ...\n```\n\n**Good shape**\n\n> \"Faster\" could mean response time, throughput, or\n> perceived speed. Each is a different fix:\n>\n> 1. Response time: indexes plus query caching (~2h)\n> 2. Throughput: async path plus pooling (~4h)\n> 3. Perceived: progressive load and partial results\n>    (~3h)\n>\n> Current: ~500ms typical. Which axis matters?\n\n**Test**: when the verb in the request is ambiguous\n(faster, better, cleaner, simpler), did the agent name\nthe alternatives or pick one?\n\n## AP-3: Strategy Pattern for One Function\n\n**Maps to**: Principle 2 (Simplicity First)\n\n**Trigger**: User asks for a single function. The\nagent ships an abstract base class, two implementing\nclasses, a config dataclass, and a coordinator class\nfor ten lines of arithmetic.\n\n**Bad shape**\n\n```python\nclass DiscountStrategy(ABC):\n    @abstractmethod\n    def calculate(self, amount: float) -> float: ...\n\nclass PercentageDiscount(DiscountStrategy):\n    def __init__(self, p): self.p = p\n    def calculate(self, a): return a * (self.p / 100)\n\n# Plus FixedDiscount, DiscountConfig, DiscountCalculator\n# for what should be one function\n```\n\n**Good shape**\n\n```python\ndef calculate_discount(amount: float, percent: float) -> float:\n    return amount * (percent / 100)\n```\n\n**Test**: count types and classes added per actual use\ncase. If types-added exceeds use-cases-served, the\npattern is premature.\n\n## AP-4: Speculative Features\n\n**Maps to**: Principle 2 (Simplicity First)\n\n**Trigger**: \"Save user preferences to database\"\nbecomes a class with optional caching, validation,\nnotification hooks, and merge semantics. None were\nasked for.\n\n**Bad shape**\n\n```python\nclass PreferenceManager:\n    def __init__(self, db, cache=None, validator=None):\n        ...\n    def save(self, user_id, prefs,\n             merge=True, validate=True, notify=False):\n        # 60 lines of optional behavior\n```\n\n**Good shape**\n\n```python\ndef save_preferences(db, user_id: int, preferences: dict):\n    db.execute(\n        \"UPDATE users SET preferences = ? WHERE id = ?\",\n        (json.dumps(preferences), user_id),\n    )\n```\n\n**Test**: list every parameter that was not in the\nrequest. If you cannot point at a sentence in the\nrequest that demanded it, delete the parameter.\n\n## AP-5: Drive-by Refactoring\n\n**Maps to**: Principle 3 (Surgical Changes)\n\n**Trigger**: User reports a single bug. The agent\nfixes the bug, then \"improves\" three other functions,\nadds docstrings, and tightens validation logic that\nnobody asked about.\n\n**Bad shape**: a 90-line diff to fix a 4-line bug,\nwith related but unrequested cleanups across two more\nfiles.\n\n**Good shape**: a 4-line diff that fixes only the\nreported bug. If you noticed unrelated issues, list\nthem in the response and ask before touching them.\n\n**Test**: read the diff line by line. For each\nchanged line, ask \"which sentence in the user's\nrequest demanded this?\" Lines without an answer are\ncandidates for removal from the diff.\n\n## AP-6: Style Drift During Edit\n\n**Maps to**: Principle 3 (Surgical Changes)\n\n**Trigger**: User asks for one logging line in an\nupload function. The agent ships type hints,\ndocstrings, single-quote-to-double-quote conversion,\nand a flattened control flow.\n\n**Bad shape**\n\n```diff\n- def upload_file(file_path, destination):\n+ def upload_file(file_path: str, destination: str) -> bool:\n+     \"\"\"Upload file to destination.\"\"\"\n      try:\n-         with open(file_path, 'rb') as f:\n+         with open(file_path, \"rb\") as f:\n              ...\n```\n\n**Good shape**\n\n```diff\n+ logger = logging.getLogger(__name__)\n+\n  def upload_file(file_path, destination):\n+     logger.info(f'Starting upload: {file_path}')\n      try:\n          with open(file_path, 'rb') as f:\n              ...\n```\n\n**Test**: the diff should not change quote style,\ntype hint presence, docstring presence, or whitespace\npatterns unless the request named them.\n\n## AP-7: Vague Success Criteria\n\n**Maps to**: Principle 4 (Goal-Driven Execution)\n\n**Trigger**: \"Fix the authentication system.\" The\nagent responds \"I will review the code, identify\nissues, make improvements, and test.\" That is a plan\nshape, not a verification plan.\n\n**Bad shape**: a four-step list of verbs (review,\nidentify, improve, test) with no checkable signal per\nstep.\n\n**Good shape**\n\n> \"Fix authentication\" is too broad to verify. Best\n> guess at the actual problem: sessions stay alive\n> after password change. Plan:\n>\n> 1. Write failing test: change password, prior\n>    session is rejected.\n> 2. Implement: invalidate sessions on change.\n> 3. Edge cases: concurrent change, multi-session.\n> 4. Regression: full auth suite green.\n>\n> Is the password-change-stale-session bug the right\n> target?\n\n**Test**: each step in the plan should have a one-\nsentence \"verify by ...\" attached. If a step lacks\none, the step is ill-specified.\n\n## AP-8: Multi-Step Plan Without Verification Gates\n\n**Maps to**: Principle 4 (Goal-Driven Execution)\n\n**Trigger**: \"Add rate limiting.\" The agent ships 300\nlines with Redis backend, configuration system, and\nmonitoring in one commit, with no per-step\nverification.\n\n**Bad shape**: one large commit covering basic limits,\nmiddleware extraction, Redis integration, and\nconfiguration. Nothing is independently shippable or\nrevertible.\n\n**Good shape**\n\n> Plan, each step independently verifiable:\n>\n> 1. In-memory limit on one endpoint. Verify: 11\n>    requests, the 11th gets 429.\n> 2. Extract to middleware, apply broadly. Verify:\n>    /users and /posts both rate-limit; existing tests\n>    pass.\n> 3. Redis backend. Verify: limits persist across\n>    restarts; two instances share counters.\n> 4. Per-endpoint config. Verify: /search 10/min,\n>    /users 100/min, parsed config tested.\n>\n> Start with step 1?\n\n**Test**: each step in the plan should be revertible\non its own. If reverting step 3 breaks step 4, the\nsteps are not independent and the plan needs a redraw.\n\n## How to Use This Module\n\nWhen reviewing your own diff or someone else's, name\nthe rail you see. \"This is AP-5: drive-by\nrefactoring\" travels faster than \"this could be\nsimpler somehow.\" Naming the rail is the first half\nof fixing the rail.\n\nCross-references for the rails:\n\n- AP-3, AP-4 connect to `Skill(imbue:scope-guard)` and\n  `Skill(leyline:additive-bias-defense)`\n- AP-5, AP-6 connect to `Skill(imbue:justify)` and\n  the `bounded-discovery.md` rule\n- AP-7, AP-8 connect to `Skill(imbue:proof-of-work)`\n  and `Skill(superpowers:test-driven-development)`\n\nFile v1.9.16:modules/senior-engineer-test.md\n\n# The Senior Engineer Test\n\nA three-question battery to apply to your own code\nbefore claiming it is done. The questions stand in\nfor the senior engineer who is not in the room.\n\nAdapted from a self-check Karpathy calls out for\nagentic coding: ask whether a senior engineer would\nsay this is overcomplicated. We expand the question\ninto three concrete sub-questions that map to common\nLLM coding failures.\n\n## The Question\n\n> Would a senior engineer who is busy and a little\n> grumpy say this code is overcomplicated?\n\nIf yes, the diff is not ready.\n\n## The Three Sub-Questions\n\n### Q1: Could this be 50% shorter without losing meaning?\n\nMost LLM-written code can be cut by a third to a\nhalf. If the diff is 200 lines, ask: which 100 lines\nexist because the agent felt clever, not because the\nproblem demanded them?\n\nCommon 50% wins:\n\n- Replace abstract base class plus two subclasses\n  with one function plus a parameter.\n- Replace try-except wrapping every call with a\n  single boundary handler.\n- Replace explicit getter and setter with direct\n  attribute access.\n- Replace nested conditionals with a flat early-return\n  pattern.\n\n### Q2: Are abstractions earning their weight?\n\nAn abstraction earns its weight when it is used three\nor more times, or when it isolates a real boundary\n(network, disk, locale). A class with one consumer is\nceremony. A factory with one product is ceremony.\n\nTest: for every type, class, or helper added, count\nthe call sites. One call site means the abstraction\ncosts more than it saves.\n\n### Q3: Could a junior dev follow this in six months?\n\nSix months means: docs may have rotted, original\ncontext is gone, the original author is on another\nteam. The code has to carry its own meaning.\n\nFailure signals:\n\n- Names that mean something only if you remember the\n  ticket\n- Comments that describe what the code does (the code\n  shows that) instead of why\n- Indirection that requires three jumps to find the\n  actual logic\n- Implicit invariants that nothing checks and nothing\n  documents\n\n## The Decision Tree\n\n```\nFor each of Q1, Q2, Q3:\n  - Yes -> next question\n  - No  -> stop and address before shipping\n\nIf three Yes -> ship\nIf any No   -> rework that dimension first\n```\n\nA No answer is not a failure of the agent; it is the\nagent doing its job. Catching the violation before\nthe senior engineer catches it is the entire point.\n\n## Worked Example\n\nDiff under review: a class hierarchy for a single\ndiscount calculation.\n\n- Q1 (50% shorter)? Yes obviously: one function\n  replaces five classes.\n- Q2 (abstractions earning weight)? No: zero\n  additional call sites for the strategy pattern.\n- Q3 (junior in six months)? No: two indirection hops\n  to find the multiplication.\n\nTwo No answers means rework. Replace the hierarchy\nwith the function. Now Q1, Q2, Q3 are all yes.\n\n## When the Test Does Not Apply\n\nThe senior-engineer test assumes the code will be\nread again. For a one-shot data migration that runs\nonce and is deleted, the test is too strict. See\n`tradeoff-acknowledgment.md` for the boundary cases.\n\n## Cross-References\n\n- `Skill(imbue:scope-guard)` formalizes Q2 (does the\n  abstraction earn its weight) into the Worthiness\n  formula.\n- `Skill(conserve:code-quality-principles)` is the\n  KISS / YAGNI / SOLID foundation that Q1 leans on.\n- `Skill(leyline:additive-bias-defense)` is the\n  burden-of-proof contract that backs Q2.\n\nFile v1.9.16:modules/tradeoff-acknowledgment.md\n\n# Tradeoff Acknowledgment: When Not to Apply These\n\nThe four principles bias toward caution. That bias\ncosts speed. For a substantial portion of coding work\nthe cost is worth paying. For a non-trivial minority,\nthe cost is wrong. This module names the boundary\nhonestly.\n\nThe upstream framing puts it as: \"These guidelines\nbias toward caution over speed. For trivial tasks,\nuse judgment.\" That sentence does the same work as\nthis module, just compressed.\n\n## When the Principles Do Not Apply\n\n### Trivial One-Line Fixes\n\nAsking three clarifying questions before fixing a\ntypo in a comment is a parody of caution. For diffs\nunder five lines with obvious intent, ship and move\non. Principle 1 (Think Before Coding) is for\nambiguous requests, not unambiguous ones.\n\n### Exploratory Spikes and Throwaway Scripts\n\nA 50-line script that runs once, produces a CSV, and\ngets deleted does not need the senior-engineer test.\nIt does not need TDD. It does not need careful\nabstraction analysis. The artifact's lifetime caps\nthe time worth investing in its quality.\n\nTest: if the script will run again next week, treat\nit like real code. If you will throw it away in an\nhour, do not over-invest.\n\n### Documentation-Only Changes\n\nStyle drift in docs is often the point. Rewriting a\nparagraph for clarity touches every line by design.\nPrinciple 3 (Surgical Changes) was written for code\ndiffs, where adjacent edits hide intent. Prose is\ndifferent.\n\n### Time-Boxed Prototypes\n\nA \"by Friday or we move on\" prototype is a different\nartifact from a feature. Verifiable success criteria\nfor a prototype look like \"the demo runs end to\nend,\" not \"the test suite is green.\" Calibrate\nambition to the deadline.\n\n### Production Fires\n\nWhen the database is on fire, \"let's write a failing\ntest first\" is the wrong move. Stop the fire, then\nwrite the test that prevents the next fire. The Iron\nLaw assumes a normal-operations context.\n\n## Contrarian Voices Worth Engaging\n\nThree voices push back on rigorous-by-default LLM\ncoding rules. Their critiques sharpen the boundary.\n\n**Simon Willison** (\"Not all AI-assisted programming\nis vibe coding,\" March 2025) defends throwaway\nprototyping as legitimate. His golden rule: do not\ncommit code you cannot explain. That rule is\ncompatible with everything in this skill, but it\nmakes the throwaway-prototype boundary explicit.\n\n**Mastering Product HQ** (\"What Karpathy's CLAUDE.md\nmisses\") argues code simplicity does not equal scope\nsimplicity. A 50-line solution to the wrong problem\nis still waste. The principles help with how to\nbuild; they do not help with what to build. For\n\"what,\" see `Skill(imbue:scope-guard)` and\n`Skill(imbue:feature-review)`.\n\n**NMN.gl** (\"Vibe Coding Considered Harmful,\" March\n2025) warns that vibed black boxes compound. This is\nadjacent support for the principles, not pushback,\nbut it names the real cost of skipping them at scale:\neach black box you accept becomes a future debugging\nexpense.\n\n## The Honest Bottom Line\n\nThese principles solve a specific class of problem:\nLLM agents shipping wrong-shape code on tasks they\ncould have shipped right with five minutes of\nupfront thought. That class is large. It is not\nuniversal.\n\nIf you find yourself about to invoke these\nprinciples on a task that fits in a sticky note,\nstop. The principles are the heavier path. Use the\nheavier path when the cost of getting it wrong is\nlarger than the cost of slowing down. Otherwise, ship\nand move on.\n\n## Cross-References\n\n- `Skill(imbue:scope-guard)` for \"should we build\n  this at all\" (the scope question this module\n  punts on).\n- `Skill(imbue:feature-review)` for prioritization\n  using RICE / WSJF / Kano scoring.\n- `Skill(conserve:decisive-action)` for guidance on\n  when to skip clarification and proceed.\n\nFile v1.9.16:modules/verifiable-goals.md\n\n# Verifiable Goals: A Reformulation Template\n\nVague tasks generate vague work. The fix is a\nmechanical reformulation: rewrite the request as a\ngoal that has an unambiguous \"done\" signal. Then loop\nuntil the signal fires.\n\nThis module makes the reformulation explicit, with a\ntemplate and worked examples.\n\n## The Template\n\n```\nOriginal request: <user's words>\n\nReformulated goal:\n  Success signal: <something a script or test can check>\n  Test that proves the signal fires: <how>\n  Out-of-scope cleanups noticed: <list, do not fix>\n```\n\nThe success signal must be checkable without human\njudgment. \"It feels faster\" is not a signal.\n\"p95 under 200ms on the seed dataset\" is.\n\n## Worked Examples\n\n### Example 1: \"Add validation\"\n\n```\nOriginal: Add validation to the user signup endpoint.\n\nReformulated:\n  Success signal: requests with invalid email,\n    missing username, or password under 8 chars\n    return HTTP 400 with a JSON error.\n  Test: three pytest cases, one per failure mode,\n    asserting status 400 and a specific error key.\n  Out of scope: rate limiting, password complexity\n    rules, captcha. Mention but do not implement.\n```\n\n### Example 2: \"Fix the bug\"\n\n```\nOriginal: Fix the bug where empty emails crash the\n  validator.\n\nReformulated:\n  Success signal: validate_user with email '' or\n    None raises ValueError, not AttributeError or\n    TypeError.\n  Test: test_validate_user_empty_email and\n    test_validate_user_none_email, both asserting\n    ValueError before the fix lands.\n  Out of scope: improving username validation, adding\n    docstrings, refactoring quote style.\n```\n\nThis pattern is the heart of `Skill(imbue:proof-of-work)`\nand the Iron Law: write the failing test first.\n\n### Example 3: \"Refactor X\"\n\n```\nOriginal: Refactor the upload service.\n\nReformulated:\n  Success signal: the existing test suite for\n    upload (12 tests) is green before the refactor,\n    green after, with no test changes.\n  Test: pytest tests/upload/ both before and after\n    the diff, with diff capture.\n  Out of scope: anything that requires changing a\n    test. If a test must change, the request is\n    behavior change, not refactor, and needs a new\n    spec.\n```\n\n### Example 4: \"Make it faster\"\n\n```\nOriginal: Make the search faster.\n\nReformulated:\n  Success signal: median latency on the seed query\n    set drops from current N ms to under M ms (M\n    chosen with the user).\n  Test: a benchmark script that runs 100 queries\n    against the seed dataset and reports median.\n    Captured before and after the change.\n  Out of scope: throughput optimization, perceived\n    speed, frontend caching. These are different\n    \"faster\" axes. Confirm which one before starting.\n```\n\nThis example also illustrates AP-2 (Multiple\nInterpretations): when \"faster\" is ambiguous, name\nthe axis before reformulating.\n\n### Example 5: \"Improve UX\"\n\n```\nOriginal: Improve the checkout UX.\n\nReformulated:\n  Success signal: a test user completes the checkout\n    flow in 4 clicks or fewer (current: 7), with no\n    blocking validation surprises.\n  Test: a Playwright or manual click-through script\n    that records click count and timestamp per step.\n  Out of scope: visual redesign, copy revisions,\n    accessibility audit. Mention but do not bundle.\n```\n\n### Example 6: \"Add rate limiting\"\n\n```\nOriginal: Add rate limiting to the API.\n\nReformulated:\n  Success signal (step 1): the 11th request to\n    /signup in 60 seconds returns 429.\n  Test: a curl loop in CI plus a pytest that\n    simulates 11 sequential calls.\n  Out of scope (this step): Redis backend,\n    per-endpoint configuration, monitoring. Each\n    is a separate reformulation.\n```\n\nThis example illustrates AP-8 (Multi-Step Plan\nWithout Verification Gates): each step gets its own\nreformulation, its own success signal, its own test.\n\n## Why This Works\n\nWhen the success signal is a script or test, three\nthings become true:\n\n1. The agent can loop independently. No need to ask\n   \"is it good now?\" The test answers.\n2. The user can review by running the test, not by\n   reading 300 lines of diff.\n3. The work is self-documenting. The next person can\n   see what \"done\" meant for this task.\n\n## Cross-References\n\n- `Skill(imbue:proof-of-work)` is the contract that\n  enforces this template under the Iron Law.\n- `Skill(superpowers:test-driven-development)` is the\n  RED-GREEN-REFACTOR loop this template feeds.\n- `/spec-kit:speckit-clarify` command helps when the\n  reformulation surfaces an ambiguity that needs the\n  user's input first.\n\nFile v1.9.16:skill-card.md\n\n## Description: <br>\nPre-implementation gate covering think-first, simplicity, surgical edits, and verifiable goals. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[athola](https://clawhub.ai/user/athola) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and coding agents use this skill as a pre-implementation and review gate for non-trivial coding tasks. It prompts assumption checks, scope control, surgical diffs, and verifiable success criteria before and after implementation. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill may slow or over-constrain trivial coding tasks because some triggers are broad. <br>\nMitigation: Apply the tradeoff guidance and keep the gate lightweight for obvious typo fixes, throwaway spikes, documentation-only edits, time-boxed prototypes, and urgent production fixes. <br>\nRisk: Companion Night Market or Claude Code plugin components may introduce behavior outside this skill's markdown guidance. <br>\nMitigation: Review and scan any companion plugin agents, hooks, or commands before enabling them. <br>\n\n\n## Reference(s): <br>\n- [ClawHub Skill Page](https://clawhub.ai/athola/skills/nm-imbue-karpathy-principles) <br>\n- [Claude Night Market imbue plugin](https://github.com/athola/claude-night-market/tree/master/plugins/imbue) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [guidance, markdown, code] <br>\n**Output Format:** [Markdown guidance with checklists, scope rationale, and verification criteria] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [May include assumptions, tradeoff notes, diff-trace checks, and test or command suggestions for verification.] <br>\n\n## Skill Version(s): <br>\n1.9.16 (source: server release metadata; artifact frontmatter says 1.9.8) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.9.14: 7 files, 14978 bytes\n\nFiles: modules/anti-patterns.md (8298b), modules/senior-engineer-test.md (3378b), modules/tradeoff-acknowledgment.md (3757b), modules/verifiable-goals.md (4501b), skill-card.md (2196b), SKILL.md (7513b), _meta.json (148b)\n\nFile v1.9.14:SKILL.md\n\n---\nname: karpathy-principles\ndescription: |\n  Pre-implementation gate covering think-first, simplicity, surgical edits, and verifiable goals\nversion: 1.9.8\ntriggers:\n  - karpathy\n  - coding-pitfalls\n  - synthesis\n  - entry-point\n  - discipline\n  - anti-overengineering\n  - TDD\n  - starting implementation to verify the approach\nmetadata: {\"openclaw\": {\"homepage\": \"https://github.com/athola/claude-night-market/tree/master/plugins/imbue\", \"emoji\": \"\\ud83e\\udd9e\", \"requires\": {\"config\": [\"night-market.imbue:scope-guard\", \"night-market.imbue:proof-of-work\", \"night-market.imbue:rigorous-reasoning\", \"night-market.leyline:additive-bias-defense\", \"night-market.conserve:code-quality-principles\"]}}}\nsource: claude-night-market\nsource_plugin: imbue\n---\n\n> **Night Market Skill** — ported from [claude-night-market/imbue](https://github.com/athola/claude-night-market/tree/master/plugins/imbue). For the full experience with agents, hooks, and commands, install the Claude Code plugin.\n\n\n> The models make wrong assumptions on your behalf and\n> just run along with them without checking. They don't\n> manage their confusion, don't seek clarifications,\n> don't surface inconsistencies, don't present\n> tradeoffs, don't push back when they should.\n>\n> -- Andrej Karpathy, on agentic coding failure modes\n\n## What This Is\n\nA four-principle contract for reducing the most common\nLLM coding pitfalls. Compact entry-point. Each\nprinciple has a deeper-dive skill in night-market;\nthis skill is the index, not the encyclopedia.\n\nDerivation: distilled by Forrest Chang\n(forrestchang/andrej-karpathy-skills, MIT) from\nKarpathy's observations. Full attribution in\n`references/source-attribution.md`.\n\n## When to Use\n\n- Before starting any coding task larger than a typo\n- During code review, to name the failure mode you see\n- After writing a diff, to self-audit before claiming\n  done\n- When training a junior engineer to read agent diffs\n\n## When NOT to Use\n\nThese principles bias toward caution over speed. For\ncases listed in `modules/tradeoff-acknowledgment.md`,\nuse judgment: trivial fixes, exploratory spikes,\ndocumentation-only edits, and time-boxed prototypes.\n\n## The Four Principles\n\n### 1. Think Before Coding\n\n**State assumptions. Surface confusion. Match tone to\nevidence.**\n\n- If multiple interpretations of the request exist,\n  list them. Do not silently pick.\n- If a simpler approach exists, name it. Push back\n  when the simpler path is correct.\n- If something is unclear, stop and ask. Hidden\n  assumptions are the cheapest bug to prevent and the\n  most expensive to find later.\n- Make claims no stronger than the evidence supports.\n  Calibrated tone beats confident hand-waving.\n\nDeep dives: `Skill(imbue:rigorous-reasoning)` for the\nsycophancy guard, `Skill(superpowers:brainstorming)`\nfor option generation, `/spec-kit:speckit-clarify`\ncommand for ambiguity drilldown.\n\n### 2. Simplicity First\n\n**Minimum code that solves the problem. Nothing\nspeculative.**\n\n> They really like to overcomplicate code and APIs,\n> bloat abstractions.\n>\n> -- Andrej Karpathy, on the same agentic-coding thread\n\n\n\n- No features beyond what was asked\n- No abstractions for single-use code\n- No flexibility or configurability that wasn't\n  requested\n- No error handling for impossible scenarios\n- If you wrote 200 lines and it could be 50, rewrite\n  it\n\nSelf-check: would a senior engineer say this is\novercomplicated? See `modules/senior-engineer-test.md`.\n\nDeep dives: `Skill(imbue:scope-guard)` for the\nworthiness formula and branch budgets,\n`Skill(leyline:additive-bias-defense)` for burden of\nproof on every addition,\n`Skill(conserve:code-quality-principles)` for the\nKISS / YAGNI / SOLID foundation.\n\n### 3. Surgical Changes\n\n**Touch only what you must. Clean up only your own\nmess.**\n\n- Do not improve adjacent code, comments, or\n  formatting\n- Do not refactor things that aren't broken\n- Match existing style even when you would do it\n  differently\n- If you notice unrelated dead code, mention it; do\n  not delete it\n- When your changes orphan imports or variables,\n  remove the orphans you created. Pre-existing dead\n  code stays unless asked.\n\nThe trace-back test: every changed line should trace\ndirectly to the user's request.\n\nDeep dives: `Skill(imbue:justify)` for additive-bias\naudits on diffs, `Skill(leyline:additive-bias-defense)`\nfor the burden-of-proof contract, the\n`bounded-discovery.md` rule for read-budget caps.\n\n### 4. Goal-Driven Execution\n\n**Define verifiable success criteria. Loop until\nverified.**\n\nTransform vague tasks into checkable goals:\n\n- \"Add validation\" becomes \"tests for invalid inputs\n  pass\"\n- \"Fix the bug\" becomes \"test reproducing the bug,\n  then make it pass\"\n- \"Refactor X\" becomes \"tests pass before and after\"\n- \"Make it faster\" becomes \"benchmark Y under N ms\"\n\nFor multi-step tasks, state a brief plan with\nverification per step. Strong success criteria let\nyou loop independently. Weak criteria require\nconstant clarification.\n\nSee `modules/verifiable-goals.md` for the full\nreformulation template.\n\nDeep dives: `Skill(imbue:proof-of-work)` for the Iron\nLaw (no implementation without a failing test first),\n`Skill(superpowers:test-driven-development)` for the\nRED-GREEN-REFACTOR loop.\n\n## The Karpathy Self-Check\n\nBefore you ship, four questions:\n\n| Principle | Question |\n|-----------|----------|\n| Think Before Coding | Did I list assumptions, or did I guess silently? |\n| Simplicity First | Would a senior engineer call this overcomplicated? |\n| Surgical Changes | Does every changed line trace to the request? |\n| Goal-Driven Execution | Can I prove this is done with a check, not a feeling? |\n\nFour \"yes\" answers means ship. Anything else means\niterate.\n\n## Modules\n\n- `modules/anti-patterns.md` - eight named drift rails\n  with before/after diffs\n- `modules/senior-engineer-test.md` - the\n  three-question self-check battery\n- `modules/verifiable-goals.md` - vague-to-verifiable\n  reformulation template with worked examples\n- `modules/tradeoff-acknowledgment.md` - when the four\n  principles do not apply\n\n## References\n\n- `references/source-attribution.md` - Karpathy\n  primary citation, Forrest Chang derivation, license,\n  adjacent prior art\n\n## Related Skills\n\n- `Skill(imbue:scope-guard)` - worthiness formula and\n  branch budgets\n- `Skill(imbue:proof-of-work)` - Iron Law TDD gate\n- `Skill(imbue:rigorous-reasoning)` - sycophancy and\n  hidden-assumption guard\n- `Skill(imbue:justify)` - additive-bias diff audit\n- `Skill(leyline:additive-bias-defense)` - burden of\n  proof on every addition\n- `Skill(conserve:code-quality-principles)` - KISS,\n  YAGNI, SOLID\n- `Skill(superpowers:test-driven-development)` -\n  RED-GREEN-REFACTOR\n- `Skill(superpowers:brainstorming)` - generate\n  options before committing\n- See `docs/quality-gates.md#skill-level-quality-gate-composition`\n  for the full gate-skill federation graph (this skill\n  is the synthesis hub)\n\n## Required TodoWrite Items\n\nWhen invoked as a pre-flight gate, create:\n\n- `karpathy:assumptions-listed` - principle 1 satisfied\n- `karpathy:simplicity-checked` - principle 2 satisfied\n- `karpathy:trace-back-verified` - principle 3 satisfied\n- `karpathy:success-criteria-defined` - principle 4\n  satisfied\n\n## Exit Criteria\n\n- Each of the four principles has been answered with a\n  concrete artifact (assumption list, scope rationale,\n  diff trace, verification plan).\n- The senior-engineer test was applied at least once.\n- Verifiable success criteria are written down before\n  the implementation begins.\n\nFile v1.9.14:_meta.json\n\n{\n  \"ownerId\": \"kn7d107jg9jv602h9ytsegydq184a42s\",\n  \"slug\": \"nm-imbue-karpathy-principles\",\n  \"version\": \"1.9.14\",\n  \"publishedAt\": 1782842406428\n}\n\nFile v1.9.14:modules/anti-patterns.md\n\n# Anti-Patterns: Eight Named Drift Rails\n\nEach drift rail names a specific way LLM-driven coding\ngoes wrong, with a trigger pattern, a short bad\nexample, a short good example, and a one-sentence\ntest you can apply to your own diff. Naming the rail\nhelps recall.\n\nAdapted from observations distilled by Forrest Chang\nfrom Karpathy's notes; see\n`references/source-attribution.md` for the citation\nchain. Examples are rewritten in our own prose to\nillustrate the failure shape, not lifted verbatim.\n\n## AP-1: Hidden Assumptions\n\n**Maps to**: Principle 1 (Think Before Coding)\n\n**Trigger**: User asks for a feature with multiple\nunstated dimensions (scope, format, volume, privacy).\nThe agent picks one silently and ships.\n\n**Bad shape**\n\n```python\ndef export_users(format='json'):\n    users = User.query.all()\n    if format == 'json':\n        with open('users.json', 'w') as f:\n            json.dump([u.to_dict() for u in users], f)\n```\n\nAssumed: all users (privacy?), file path, fields,\nvolume.\n\n**Good shape**\n\n> Before implementing, I want to confirm:\n>\n> - Scope: all users, or filtered?\n> - Format: download, background job, or API endpoint?\n> - Fields: which? Some are sensitive.\n> - Volume: thousands or millions?\n>\n> Default proposal: a paginated JSON endpoint. Adjust?\n\n**Test**: count silent assumptions in the diff. Three\nor more is a rail violation.\n\n## AP-2: Multiple Interpretations Picked Silently\n\n**Maps to**: Principle 1 (Think Before Coding)\n\n**Trigger**: A request like \"make the search faster\"\nhas at least three valid readings (response time,\nthroughput, perceived speed). The agent picks one and\nimplements 200 lines without flagging the choice.\n\n**Bad shape**\n\n```python\n@lru_cache(maxsize=1000)\nasync def search(query: str) -> List[Result]:\n    # 200 lines of caching, async, indexes, all picked\n    # without confirming what \"faster\" means\n    ...\n```\n\n**Good shape**\n\n> \"Faster\" could mean response time, throughput, or\n> perceived speed. Each is a different fix:\n>\n> 1. Response time: indexes plus query caching (~2h)\n> 2. Throughput: async path plus pooling (~4h)\n> 3. Perceived: progressive load and partial results\n>    (~3h)\n>\n> Current: ~500ms typical. Which axis matters?\n\n**Test**: when the verb in the request is ambiguous\n(faster, better, cleaner, simpler), did the agent name\nthe alternatives or pick one?\n\n## AP-3: Strategy Pattern for One Function\n\n**Maps to**: Principle 2 (Simplicity First)\n\n**Trigger**: User asks for a single function. The\nagent ships an abstract base class, two implementing\nclasses, a config dataclass, and a coordinator class\nfor ten lines of arithmetic.\n\n**Bad shape**\n\n```python\nclass DiscountStrategy(ABC):\n    @abstractmethod\n    def calculate(self, amount: float) -> float: ...\n\nclass PercentageDiscount(DiscountStrategy):\n    def __init__(self, p): self.p = p\n    def calculate(self, a): return a * (self.p / 100)\n\n# Plus FixedDiscount, DiscountConfig, DiscountCalculator\n# for what should be one function\n```\n\n**Good shape**\n\n```python\ndef calculate_discount(amount: float, percent: float) -> float:\n    return amount * (percent / 100)\n```\n\n**Test**: count types and classes added per actual use\ncase. If types-added exceeds use-cases-served, the\npattern is premature.\n\n## AP-4: Speculative Features\n\n**Maps to**: Principle 2 (Simplicity First)\n\n**Trigger**: \"Save user preferences to database\"\nbecomes a class with optional caching, validation,\nnotification hooks, and merge semantics. None were\nasked for.\n\n**Bad shape**\n\n```python\nclass PreferenceManager:\n    def __init__(self, db, cache=None, validator=None):\n        ...\n    def save(self, user_id, prefs,\n             merge=True, validate=True, notify=False):\n        # 60 lines of optional behavior\n```\n\n**Good shape**\n\n```python\ndef save_preferences(db, user_id: int, preferences: dict):\n    db.execute(\n        \"UPDATE users SET preferences = ? WHERE id = ?\",\n        (json.dumps(preferences), user_id),\n    )\n```\n\n**Test**: list every parameter that was not in the\nrequest. If you cannot point at a sentence in the\nrequest that demanded it, delete the parameter.\n\n## AP-5: Drive-by Refactoring\n\n**Maps to**: Principle 3 (Surgical Changes)\n\n**Trigger**: User reports a single bug. The agent\nfixes the bug, then \"improves\" three other functions,\nadds docstrings, and tightens validation logic that\nnobody asked about.\n\n**Bad shape**: a 90-line diff to fix a 4-line bug,\nwith related but unrequested cleanups across two more\nfiles.\n\n**Good shape**: a 4-line diff that fixes only the\nreported bug. If you noticed unrelated issues, list\nthem in the response and ask before touching them.\n\n**Test**: read the diff line by line. For each\nchanged line, ask \"which sentence in the user's\nrequest demanded this?\" Lines without an answer are\ncandidates for removal from the diff.\n\n## AP-6: Style Drift During Edit\n\n**Maps to**: Principle 3 (Surgical Changes)\n\n**Trigger**: User asks for one logging line in an\nupload function. The agent ships type hints,\ndocstrings, single-quote-to-double-quote conversion,\nand a flattened control flow.\n\n**Bad shape**\n\n```diff\n- def upload_file(file_path, destination):\n+ def upload_file(file_path: str, destination: str) -> bool:\n+     \"\"\"Upload file to destination.\"\"\"\n      try:\n-         with open(file_path, 'rb') as f:\n+         with open(file_path, \"rb\") as f:\n              ...\n```\n\n**Good shape**\n\n```diff\n+ logger = logging.getLogger(__name__)\n+\n  def upload_file(file_path, destination):\n+     logger.info(f'Starting upload: {file_path}')\n      try:\n          with open(file_path, 'rb') as f:\n              ...\n```\n\n**Test**: the diff should not change quote style,\ntype hint presence, docstring presence, or whitespace\npatterns unless the request named them.\n\n## AP-7: Vague Success Criteria\n\n**Maps to**: Principle 4 (Goal-Driven Execution)\n\n**Trigger**: \"Fix the authentication system.\" The\nagent responds \"I will review the code, identify\nissues, make improvements, and test.\" That is a plan\nshape, not a verification plan.\n\n**Bad shape**: a four-step list of verbs (review,\nidentify, improve, test) with no checkable signal per\nstep.\n\n**Good shape**\n\n> \"Fix authentication\" is too broad to verify. Best\n> guess at the actual problem: sessions stay alive\n> after password change. Plan:\n>\n> 1. Write failing test: change password, prior\n>    session is rejected.\n> 2. Implement: invalidate sessions on change.\n> 3. Edge cases: concurrent change, multi-session.\n> 4. Regression: full auth suite green.\n>\n> Is the password-change-stale-session bug the right\n> target?\n\n**Test**: each step in the plan should have a one-\nsentence \"verify by ...\" attached. If a step lacks\none, the step is ill-specified.\n\n## AP-8: Multi-Step Plan Without Verification Gates\n\n**Maps to**: Principle 4 (Goal-Driven Execution)\n\n**Trigger**: \"Add rate limiting.\" The agent ships 300\nlines with Redis backend, configuration system, and\nmonitoring in one commit, with no per-step\nverification.\n\n**Bad shape**: one large commit covering basic limits,\nmiddleware extraction, Redis integration, and\nconfiguration. Nothing is independently shippable or\nrevertible.\n\n**Good shape**\n\n> Plan, each step independently verifiable:\n>\n> 1. In-memory limit on one endpoint. Verify: 11\n>    requests, the 11th gets 429.\n> 2. Extract to middleware, apply broadly. Verify:\n>    /users and /posts both rate-limit; existing tests\n>    pass.\n> 3. Redis backend. Verify: limits persist across\n>    restarts; two instances share counters.\n> 4. Per-endpoint config. Verify: /search 10/min,\n>    /users 100/min, parsed config tested.\n>\n> Start with step 1?\n\n**Test**: each step in the plan should be revertible\non its own. If reverting step 3 breaks step 4, the\nsteps are not independent and the plan needs a redraw.\n\n## How to Use This Module\n\nWhen reviewing your own diff or someone else's, name\nthe rail you see. \"This is AP-5: drive-by\nrefactoring\" travels faster than \"this could be\nsimpler somehow.\" Naming the rail is the first half\nof fixing the rail.\n\nCross-references for the rails:\n\n- AP-3, AP-4 connect to `Skill(imbue:scope-guard)` and\n  `Skill(leyline:additive-bias-defense)`\n- AP-5, AP-6 connect to `Skill(imbue:justify)` and\n  the `bounded-discovery.md` rule\n- AP-7, AP-8 connect to `Skill(imbue:proof-of-work)`\n  and `Skill(superpowers:test-driven-development)`\n\nFile v1.9.14:modules/senior-engineer-test.md\n\n# The Senior Engineer Test\n\nA three-question battery to apply to your own code\nbefore claiming it is done. The questions stand in\nfor the senior engineer who is not in the room.\n\nAdapted from a self-check Karpathy calls out for\nagentic coding: ask whether a senior engineer would\nsay this is overcomplicated. We expand the question\ninto three concrete sub-questions that map to common\nLLM coding failures.\n\n## The Question\n\n> Would a senior engineer who is busy and a little\n> grumpy say this code is overcomplicated?\n\nIf yes, the diff is not ready.\n\n## The Three Sub-Questions\n\n### Q1: Could this be 50% shorter without losing meaning?\n\nMost LLM-written code can be cut by a third to a\nhalf. If the diff is 200 lines, ask: which 100 lines\nexist because the agent felt clever, not because the\nproblem demanded them?\n\nCommon 50% wins:\n\n- Replace abstract base class plus two subclasses\n  with one function plus a parameter.\n- Replace try-except wrapping every call with a\n  single boundary handler.\n- Replace explicit getter and setter with direct\n  attribute access.\n- Replace nested conditionals with a flat early-return\n  pattern.\n\n### Q2: Are abstractions earning their weight?\n\nAn abstraction earns its weight when it is used three\nor more times, or when it isolates a real boundary\n(network, disk, locale). A class with one consumer is\nceremony. A factory with one product is ceremony.\n\nTest: for every type, class, or helper added, count\nthe call sites. One call site means the abstraction\ncosts more than it saves.\n\n### Q3: Could a junior dev follow this in six months?\n\nSix months means: docs may have rotted, original\ncontext is gone, the original author is on another\nteam. The code has to carry its own meaning.\n\nFailure signals:\n\n- Names that mean something only if you remember the\n  ticket\n- Comments that describe what the code does (the code\n  shows that) instead of why\n- Indirection that requires three jumps to find the\n  actual logic\n- Implicit invariants that nothing checks and nothing\n  documents\n\n## The Decision Tree\n\n```\nFor each of Q1, Q2, Q3:\n  - Yes -> next question\n  - No  -> stop and address before shipping\n\nIf three Yes -> ship\nIf any No   -> rework that dimension first\n```\n\nA No answer is not a failure of the agent; it is the\nagent doing its job. Catching the violation before\nthe senior engineer catches it is the entire point.\n\n## Worked Example\n\nDiff under review: a class hierarchy for a single\ndiscount calculation.\n\n- Q1 (50% shorter)? Yes obviously: one function\n  replaces five classes.\n- Q2 (abstractions earning weight)? No: zero\n  additional call sites for the strategy pattern.\n- Q3 (junior in six months)? No: two indirection hops\n  to find the multiplication.\n\nTwo No answers means rework. Replace the hierarchy\nwith the function. Now Q1, Q2, Q3 are all yes.\n\n## When the Test Does Not Apply\n\nThe senior-engineer test assumes the code will be\nread again. For a one-shot data migration that runs\nonce and is deleted, the test is too strict. See\n`tradeoff-acknowledgment.md` for the boundary cases.\n\n## Cross-References\n\n- `Skill(imbue:scope-guard)` formalizes Q2 (does the\n  abstraction earn its weight) into the Worthiness\n  formula.\n- `Skill(conserve:code-quality-principles)` is the\n  KISS / YAGNI / SOLID foundation that Q1 leans on.\n- `Skill(leyline:additive-bias-defense)` is the\n  burden-of-proof contract that backs Q2.\n\nFile v1.9.14:modules/tradeoff-acknowledgment.md\n\n# Tradeoff Acknowledgment: When Not to Apply These\n\nThe four principles bias toward caution. That bias\ncosts speed. For a substantial portion of coding work\nthe cost is worth paying. For a non-trivial minority,\nthe cost is wrong. This module names the boundary\nhonestly.\n\nThe upstream framing puts it as: \"These guidelines\nbias toward caution over speed. For trivial tasks,\nuse judgment.\" That sentence does the same work as\nthis module, just compressed.\n\n## When the Principles Do Not Apply\n\n### Trivial One-Line Fixes\n\nAsking three clarifying questions before fixing a\ntypo in a comment is a parody of caution. For diffs\nunder five lines with obvious intent, ship and move\non. Principle 1 (Think Before Coding) is for\nambiguous requests, not unambiguous ones.\n\n### Exploratory Spikes and Throwaway Scripts\n\nA 50-line script that runs once, produces a CSV, and\ngets deleted does not need the senior-engineer test.\nIt does not need TDD. It does not need careful\nabstraction analysis. The artifact's lifetime caps\nthe time worth investing in its quality.\n\nTest: if the script will run again next week, treat\nit like real code. If you will throw it away in an\nhour, do not over-invest.\n\n### Documentation-Only Changes\n\nStyle drift in docs is often the point. Rewriting a\nparagraph for clarity touches every line by design.\nPrinciple 3 (Surgical Changes) was written for code\ndiffs, where adjacent edits hide intent. Prose is\ndifferent.\n\n### Time-Boxed Prototypes\n\nA \"by Friday or we move on\" prototype is a different\nartifact from a feature. Verifiable success criteria\nfor a prototype look like \"the demo runs end to\nend,\" not \"the test suite is green.\" Calibrate\nambition to the deadline.\n\n### Production Fires\n\nWhen the database is on fire, \"let's write a failing\ntest first\" is the wrong move. Stop the fire, then\nwrite the test that prevents the next fire. The Iron\nLaw assumes a normal-operations context.\n\n## Contrarian Voices Worth Engaging\n\nThree voices push back on rigorous-by-default LLM\ncoding rules. Their critiques sharpen the boundary.\n\n**Simon Willison** (\"Not all AI-assisted programming\nis vibe coding,\" March 2025) defends throwaway\nprototyping as legitimate. His golden rule: do not\ncommit code you cannot explain. That rule is\ncompatible with everything in this skill, but it\nmakes the throwaway-prototype boundary explicit.\n\n**Mastering Product HQ** (\"What Karpathy's CLAUDE.md\nmisses\") argues code simplicity does not equal scope\nsimplicity. A 50-line solution to the wrong problem\nis still waste. The principles help with how to\nbuild; they do not help with what to build. For\n\"what,\" see `Skill(imbue:scope-guard)` and\n`Skill(imbue:feature-review)`.\n\n**NMN.gl** (\"Vibe Coding Considered Harmful,\" March\n2025) warns that vibed black boxes compound. This is\nadjacent support for the principles, not pushback,\nbut it names the real cost of skipping them at scale:\neach black box you accept becomes a future debugging\nexpense.\n\n## The Honest Bottom Line\n\nThese principles solve a specific class of problem:\nLLM agents shipping wrong-shape code on tasks they\ncould have shipped right with five minutes of\nupfront thought. That class is large. It is not\nuniversal.\n\nIf you find yourself about to invoke these\nprinciples on a task that fits in a sticky note,\nstop. The principles are the heavier path. Use the\nheavier path when the cost of getting it wrong is\nlarger than the cost of slowing down. Otherwise, ship\nand move on.\n\n## Cross-References\n\n- `Skill(imbue:scope-guard)` for \"should we build\n  this at all\" (the scope question this module\n  punts on).\n- `Skill(imbue:feature-review)` for prioritization\n  using RICE / WSJF / Kano scoring.\n- `Skill(conserve:decisive-action)` for guidance on\n  when to skip clarification and proceed.\n\nFile v1.9.14:modules/verifiable-goals.md\n\n# Verifiable Goals: A Reformulation Template\n\nVague tasks generate vague work. The fix is a\nmechanical reformulation: rewrite the request as a\ngoal that has an unambiguous \"done\" signal. Then loop\nuntil the signal fires.\n\nThis module makes the reformulation explicit, with a\ntemplate and worked examples.\n\n## The Template\n\n```\nOriginal request: <user's words>\n\nReformulated goal:\n  Success signal: <something a script or test can check>\n  Test that proves the signal fires: <how>\n  Out-of-scope cleanups noticed: <list, do not fix>\n```\n\nThe success signal must be checkable without human\njudgment. \"It feels faster\" is not a signal.\n\"p95 under 200ms on the seed dataset\" is.\n\n## Worked Examples\n\n### Example 1: \"Add validation\"\n\n```\nOriginal: Add validation to the user signup endpoint.\n\nReformulated:\n  Success signal: requests with invalid email,\n    missing username, or password under 8 chars\n    return HTTP 400 with a JSON error.\n  Test: three pytest cases, one per failure mode,\n    asserting status 400 and a specific error key.\n  Out of scope: rate limiting, password complexity\n    rules, captcha. Mention but do not implement.\n```\n\n### Example 2: \"Fix the bug\"\n\n```\nOriginal: Fix the bug where empty emails crash the\n  validator.\n\nReformulated:\n  Success signal: validate_user with email '' or\n    None raises ValueError, not AttributeError or\n    TypeError.\n  Test: test_validate_user_empty_email and\n    test_validate_user_none_email, both asserting\n    ValueError before the fix lands.\n  Out of scope: improving username validation, adding\n    docstrings, refactoring quote style.\n```\n\nThis pattern is the heart of `Skill(imbue:proof-of-work)`\nand the Iron Law: write the failing test first.\n\n### Example 3: \"Refactor X\"\n\n```\nOriginal: Refactor the upload service.\n\nReformulated:\n  Success signal: the existing test suite for\n    upload (12 tests) is green before the refactor,\n    green after, with no test changes.\n  Test: pytest tests/upload/ both before and after\n    the diff, with diff capture.\n  Out of scope: anything that requires changing a\n    test. If a test must change, the request is\n    behavior change, not refactor, and needs a new\n    spec.\n```\n\n### Example 4: \"Make it faster\"\n\n```\nOriginal: Make the search faster.\n\nReformulated:\n  Success signal: median latency on the seed query\n    set drops from current N ms to under M ms (M\n    chosen with the user).\n  Test: a benchmark script that runs 100 queries\n    against the seed dataset and reports median.\n    Captured before and after the change.\n  Out of scope: throughput optimization, perceived\n    speed, frontend caching. These are different\n    \"faster\" axes. Confirm which one before starting.\n```\n\nThis example also illustrates AP-2 (Multiple\nInterpretations): when \"faster\" is ambiguous, name\nthe axis before reformulating.\n\n### Example 5: \"Improve UX\"\n\n```\nOriginal: Improve the checkout UX.\n\nReformulated:\n  Success signal: a test user completes the checkout\n    flow in 4 clicks or fewer (current: 7), with no\n    blocking validation surprises.\n  Test: a Playwright or manual click-through script\n    that records click count and timestamp per step.\n  Out of scope: visual redesign, copy revisions,\n    accessibility audit. Mention but do not bundle.\n```\n\n### Example 6: \"Add rate limiting\"\n\n```\nOriginal: Add rate limiting to the API.\n\nReformulated:\n  Success signal (step 1): the 11th request to\n    /signup in 60 seconds returns 429.\n  Test: a curl loop in CI plus a pytest that\n    simulates 11 sequential calls.\n  Out of scope (this step): Redis backend,\n    per-endpoint configuration, monitoring. Each\n    is a separate reformulation.\n```\n\nThis example illustrates AP-8 (Multi-Step Plan\nWithout Verification Gates): each step gets its own\nreformulation, its own success signal, its own test.\n\n## Why This Works\n\nWhen the success signal is a script or test, three\nthings become true:\n\n1. The agent can loop independently. No need to ask\n   \"is it good now?\" The test answers.\n2. The user can review by running the test, not by\n   reading 300 lines of diff.\n3. The work is self-documenting. The next person can\n   see what \"done\" meant for this task.\n\n## Cross-References\n\n- `Skill(imbue:proof-of-work)` is the contract that\n  enforces this template under the Iron Law.\n- `Skill(superpowers:test-driven-development)` is the\n  RED-GREEN-REFACTOR loop this template feeds.\n- `/spec-kit:speckit-clarify` command helps when the\n  reformulation surfaces an ambiguity that needs the\n  user's input first.\n\nFile v1.9.14:skill-card.md\n\n## Description: <br>\nPre-implementation gate covering think-first, simplicity, surgical edits, and verifiable goals. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[athola](https://clawhub.ai/user/athola) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and coding agents use this skill before, during, or after coding work to surface assumptions, avoid overengineering, keep edits scoped, and define verifiable completion criteria. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill can slow simple coding tasks by encouraging assumptions, scope checks, and verification plans before implementation. <br>\nMitigation: Use the full workflow for non-trivial or ambiguous coding work; abbreviate or skip it for obvious typo fixes, throwaway scripts, documentation-only edits, time-boxed prototypes, and production fires. <br>\nRisk: Referenced companion skills may add separate behavior or review obligations if installed together. <br>\nMitigation: Review the companion skills listed in the ClawHub metadata before enabling them in the same agent workflow. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/athola/skills/nm-imbue-karpathy-principles) <br>\n- [ClawHub publisher profile](https://clawhub.ai/user/athola) <br>\n- [OpenClaw homepage](https://github.com/athola/claude-night-market/tree/master/plugins/imbue) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [guidance, markdown, text] <br>\n**Output Format:** [Markdown and plain-text guidance for planning, review, and verification steps] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Documentation-only skill; no API calls, shell execution, or generated files are required by the skill itself.] <br>\n\n## Skill Version(s): <br>\n1.9.14 (source: ClawHub server release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.9.13: 7 files, 14991 bytes\n\nFiles: modules/anti-patterns.md (8298b), modules/senior-engineer-test.md (3378b), modules/tradeoff-acknowledgment.md (3757b), modules/verifiable-goals.md (4501b), skill-card.md (2176b), SKILL.md (7513b), _meta.json (148b)\n\nFile v1.9.13:SKILL.md\n\n---\nname: karpathy-principles\ndescription: |\n  Pre-implementation gate covering think-first, simplicity, surgical edits, and verifiable goals\nversion: 1.9.8\ntriggers:\n  - karpathy\n  - coding-pitfalls\n  - synthesis\n  - entry-point\n  - discipline\n  - anti-overengineering\n  - TDD\n  - starting implementation to verify the approach\nmetadata: {\"openclaw\": {\"homepage\": \"https://github.com/athola/claude-night-market/tree/master/plugins/imbue\", \"emoji\": \"\\ud83e\\udd9e\", \"requires\": {\"config\": [\"night-market.imbue:scope-guard\", \"night-market.imbue:proof-of-work\", \"night-market.imbue:rigorous-reasoning\", \"night-market.leyline:additive-bias-defense\", \"night-market.conserve:code-quality-principles\"]}}}\nsource: claude-night-market\nsource_plugin: imbue\n---\n\n> **Night Market Skill** — ported from [claude-night-market/imbue](https://github.com/athola/claude-night-market/tree/master/plugins/imbue). For the full experience with agents, hooks, and commands, install the Claude Code plugin.\n\n\n> The models make wrong assumptions on your behalf and\n> just run along with them without checking. They don't\n> manage their confusion, don't seek clarifications,\n> don't surface inconsistencies, don't present\n> tradeoffs, don't push back when they should.\n>\n> -- Andrej Karpathy, on agentic coding failure modes\n\n## What This Is\n\nA four-principle contract for reducing the most common\nLLM coding pitfalls. Compact entry-point. Each\nprinciple has a deeper-dive skill in night-market;\nthis skill is the index, not the encyclopedia.\n\nDerivation: distilled by Forrest Chang\n(forrestchang/andrej-karpathy-skills, MIT) from\nKarpathy's observations. Full attribution in\n`references/source-attribution.md`.\n\n## When to Use\n\n- Before starting any coding task larger than a typo\n- During code review, to name the failure mode you see\n- After writing a diff, to self-audit before claiming\n  done\n- When training a junior engineer to read agent diffs\n\n## When NOT to Use\n\nThese principles bias toward caution over speed. For\ncases listed in `modules/tradeoff-acknowledgment.md`,\nuse judgment: trivial fixes, exploratory spikes,\ndocumentation-only edits, and time-boxed prototypes.\n\n## The Four Principles\n\n### 1. Think Before Coding\n\n**State assumptions. Surface confusion. Match tone to\nevidence.**\n\n- If multiple interpretations of the request exist,\n  list them. Do not silently pick.\n- If a simpler approach exists, name it. Push back\n  when the simpler path is correct.\n- If something is unclear, stop and ask. Hidden\n  assumptions are the cheapest bug to prevent and the\n  most expensive to find later.\n- Make claims no stronger than the evidence supports.\n  Calibrated tone beats confident hand-waving.\n\nDeep dives: `Skill(imbue:rigorous-reasoning)` for the\nsycophancy guard, `Skill(superpowers:brainstorming)`\nfor option generation, `/spec-kit:speckit-clarify`\ncommand for ambiguity drilldown.\n\n### 2. Simplicity First\n\n**Minimum code that solves the problem. Nothing\nspeculative.**\n\n> They really like to overcomplicate code and APIs,\n> bloat abstractions.\n>\n> -- Andrej Karpathy, on the same agentic-coding thread\n\n\n\n- No features beyond what was asked\n- No abstractions for single-use code\n- No flexibility or configurability that wasn't\n  requested\n- No error handling for impossible scenarios\n- If you wrote 200 lines and it could be 50, rewrite\n  it\n\nSelf-check: would a senior engineer say this is\novercomplicated? See `modules/senior-engineer-test.md`.\n\nDeep dives: `Skill(imbue:scope-guard)` for the\nworthiness formula and branch budgets,\n`Skill(leyline:additive-bias-defense)` for burden of\nproof on every addition,\n`Skill(conserve:code-quality-principles)` for the\nKISS / YAGNI / SOLID foundation.\n\n### 3. Surgical Changes\n\n**Touch only what you must. Clean up only your own\nmess.**\n\n- Do not improve adjacent code, comments, or\n  formatting\n- Do not refactor things that aren't broken\n- Match existing style even when you would do it\n  differently\n- If you notice unrelated dead code, mention it; do\n  not delete it\n- When your changes orphan imports or variables,\n  remove the orphans you created. Pre-existing dead\n  code stays unless asked.\n\nThe trace-back test: every changed line should trace\ndirectly to the user's request.\n\nDeep dives: `Skill(imbue:justify)` for additive-bias\naudits on diffs, `Skill(leyline:additive-bias-defense)`\nfor the burden-of-proof contract, the\n`bounded-discovery.md` rule for read-budget caps.\n\n### 4. Goal-Driven Execution\n\n**Define verifiable success criteria. Loop until\nverified.**\n\nTransform vague tasks into checkable goals:\n\n- \"Add validation\" becomes \"tests for invalid inputs\n  pass\"\n- \"Fix the bug\" becomes \"test reproducing the bug,\n  then make it pass\"\n- \"Refactor X\" becomes \"tests pass before and after\"\n- \"Make it faster\" becomes \"benchmark Y under N ms\"\n\nFor multi-step tasks, state a brief plan with\nverification per step. Strong success criteria let\nyou loop independently. Weak criteria require\nconstant clarification.\n\nSee `modules/verifiable-goals.md` for the full\nreformulation template.\n\nDeep dives: `Skill(imbue:proof-of-work)` for the Iron\nLaw (no implementation without a failing test first),\n`Skill(superpowers:test-driven-development)` for the\nRED-GREEN-REFACTOR loop.\n\n## The Karpathy Self-Check\n\nBefore you ship, four questions:\n\n| Principle | Question |\n|-----------|----------|\n| Think Before Coding | Did I list assumptions, or did I guess silently? |\n| Simplicity First | Would a senior engineer call this overcomplicated? |\n| Surgical Changes | Does every changed line trace to the request? |\n| Goal-Driven Execution | Can I prove this is done with a check, not a feeling? |\n\nFour \"yes\" answers means ship. Anything else means\niterate.\n\n## Modules\n\n- `modules/anti-patterns.md` - eight named drift rails\n  with before/after diffs\n- `modules/senior-engineer-test.md` - the\n  three-question self-check battery\n- `modules/verifiable-goals.md` - vague-to-verifiable\n  reformulation template with worked examples\n- `modules/tradeoff-acknowledgment.md` - when the four\n  principles do not apply\n\n## References\n\n- `references/source-attribution.md` - Karpathy\n  primary citation, Forrest Chang derivation, license,\n  adjacent prior art\n\n## Related Skills\n\n- `Skill(imbue:scope-guard)` - worthiness formula and\n  branch budgets\n- `Skill(imbue:proof-of-work)` - Iron Law TDD gate\n- `Skill(imbue:rigorous-reasoning)` - sycophancy and\n  hidden-assumption guard\n- `Skill(imbue:justify)` - additive-bias diff audit\n- `Skill(leyline:additive-bias-defense)` - burden of\n  proof on every addition\n- `Skill(conserve:code-quality-principles)` - KISS,\n  YAGNI, SOLID\n- `Skill(superpowers:test-driven-development)` -\n  RED-GREEN-REFACTOR\n- `Skill(superpowers:brainstorming)` - generate\n  options before committing\n- See `docs/quality-gates.md#skill-level-quality-gate-composition`\n  for the full gate-skill federation graph (this skill\n  is the synthesis hub)\n\n## Required TodoWrite Items\n\nWhen invoked as a pre-flight gate, create:\n\n- `karpathy:assumptions-listed` - principle 1 satisfied\n- `karpathy:simplicity-checked` - principle 2 satisfied\n- `karpathy:trace-back-verified` - principle 3 satisfied\n- `karpathy:success-criteria-defined` - principle 4\n  satisfied\n\n## Exit Criteria\n\n- Each of the four principles has been answered with a\n  concrete artifact (assumption list, scope rationale,\n  diff trace, verification plan).\n- The senior-engineer test was applied at least once.\n- Verifiable success criteria are written down before\n  the implementation begins.\n\nFile v1.9.13:_meta.json\n\n{\n  \"ownerId\": \"kn7d107jg9jv602h9ytsegydq184a42s\",\n  \"slug\": \"nm-imbue-karpathy-principles\",\n  \"version\": \"1.9.13\",\n  \"publishedAt\": 1782577130722\n}\n\nFile v1.9.13:modules/anti-patterns.md\n\n# Anti-Patterns: Eight Named Drift Rails\n\nEach drift rail names a specific way LLM-driven coding\ngoes wrong, with a trigger pattern, a short bad\nexample, a short good example, and a one-sentence\ntest you can apply to your own diff. Naming the rail\nhelps recall.\n\nAdapted from observations distilled by Forrest Chang\nfrom Karpathy's notes; see\n`references/source-attribution.md` for the citation\nchain. Examples are rewritten in our own prose to\nillustrate the failure shape, not lifted verbatim.\n\n## AP-1: Hidden Assumptions\n\n**Maps to**: Principle 1 (Think Before Coding)\n\n**Trigger**: User asks for a feature with multiple\nunstated dimensions (scope, format, volume, privacy).\nThe agent picks one silently and ships.\n\n**Bad shape**\n\n```python\ndef export_users(format='json'):\n    users = User.query.all()\n    if format == 'json':\n        with open('users.json', 'w') as f:\n            json.dump([u.to_dict() for u in users], f)\n```\n\nAssumed: all users (privacy?), file path, fields,\nvolume.\n\n**Good shape**\n\n> Before implementing, I want to confirm:\n>\n> - Scope: all users, or filtered?\n> - Format: download, background job, or API endpoint?\n> - Fields: which? Some are sensitive.\n> - Volume: thousands or millions?\n>\n> Default proposal: a paginated JSON endpoint. Adjust?\n\n**Test**: count silent assumptions in the diff. Three\nor more is a rail violation.\n\n## AP-2: Multiple Interpretations Picked Silently\n\n**Maps to**: Principle 1 (Think Before Coding)\n\n**Trigger**: A request like \"make the search faster\"\nhas at least three valid readings (response time,\nthroughput, perceived speed). The agent picks one and\nimplements 200 lines without flagging the choice.\n\n**Bad shape**\n\n```python\n@lru_cache(maxsize=1000)\nasync def search(query: str) -> List[Result]:\n    # 200 lines of caching, async, indexes, all picked\n    # without confirming what \"faster\" means\n    ...\n```\n\n**Good shape**\n\n> \"Faster\" could mean response time, throughput, or\n> perceived speed. Each is a different fix:\n>\n> 1. Response time: indexes plus query caching (~2h)\n> 2. Throughput: async path plus pooling (~4h)\n> 3. Perceived: progressive load and partial results\n>    (~3h)\n>\n> Current: ~500ms typical. Which axis matters?\n\n**Test**: when the verb in the request is ambiguous\n(faster, better, cleaner, simpler), did the agent name\nthe alternatives or pick one?\n\n## AP-3: Strategy Pattern for One Function\n\n**Maps to**: Principle 2 (Simplicity First)\n\n**Trigger**: User asks for a single function. The\nagent ships an abstract base class, two implementing\nclasses, a config dataclass, and a coordinator class\nfor ten lines of arithmetic.\n\n**Bad shape**\n\n```python\nclass DiscountStrategy(ABC):\n    @abstractmethod\n    def calculate(self, amount: float) -> float: ...\n\nclass PercentageDiscount(DiscountStrategy):\n    def __init__(self, p): self.p = p\n    def calculate(self, a): return a * (self.p / 100)\n\n# Plus FixedDiscount, DiscountConfig, DiscountCalculator\n# for what should be one function\n```\n\n**Good shape**\n\n```python\ndef calculate_discount(amount: float, percent: float) -> float:\n    return amount * (percent / 100)\n```\n\n**Test**: count types and classes added per actual use\ncase. If types-added exceeds use-cases-served, the\npattern is premature.\n\n## AP-4: Speculative Features\n\n**Maps to**: Principle 2 (Simplicity First)\n\n**Trigger**: \"Save user preferences to database\"\nbecomes a class with optional caching, validation,\nnotification hooks, and merge semantics. None were\nasked for.\n\n**Bad shape**\n\n```python\nclass PreferenceManager:\n    def __init__(self, db, cache=None, validator=None):\n        ...\n    def save(self, user_id, prefs,\n             merge=True, validate=True, notify=False):\n        # 60 lines of optional behavior\n```\n\n**Good shape**\n\n```python\ndef save_preferences(db, user_id: int, preferences: dict):\n    db.execute(\n        \"UPDATE users SET preferences = ? WHERE id = ?\",\n        (json.dumps(preferences), user_id),\n    )\n```\n\n**Test**: list every parameter that was not in the\nrequest. If you cannot point at a sentence in the\nrequest that demanded it, delete the parameter.\n\n## AP-5: Drive-by Refactoring\n\n**Maps to**: Principle 3 (Surgical Changes)\n\n**Trigger**: User reports a single bug. The agent\nfixes the bug, then \"improves\" three other functions,\nadds docstrings, and tightens validation logic that\nnobody asked about.\n\n**Bad shape**: a 90-line diff to fix a 4-line bug,\nwith related but unrequested cleanups across two more\nfiles.\n\n**Good shape**: a 4-line diff that fixes only the\nreported bug. If you noticed unrelated issues, list\nthem in the response and ask before touching them.\n\n**Test**: read the diff line by line. For each\nchanged line, ask \"which sentence in the user's\nrequest demanded this?\" Lines without an answer are\ncandidates for removal from the diff.\n\n## AP-6: Style Drift During Edit\n\n**Maps to**: Principle 3 (Surgical Changes)\n\n**Trigger**: User asks for one logging line in an\nupload function. The agent ships type hints,\ndocstrings, single-quote-to-double-quote conversion,\nand a flattened control flow.\n\n**Bad shape**\n\n```diff\n- def upload_file(file_path, destination):\n+ def upload_file(file_path: str, destination: str) -> bool:\n+     \"\"\"Upload file to destination.\"\"\"\n      try:\n-         with open(file_path, 'rb') as f:\n+         with open(file_path, \"rb\") as f:\n              ...\n```\n\n**Good shape**\n\n```diff\n+ logger = logging.getLogger(__name__)\n+\n  def upload_file(file_path, destination):\n+     logger.info(f'Starting upload: {file_path}')\n      try:\n          with open(file_path, 'rb') as f:\n              ...\n```\n\n**Test**: the diff should not change quote style,\ntype hint presence, docstring presence, or whitespace\npatterns unless the request named them.\n\n## AP-7: Vague Success Criteria\n\n**Maps to**: Principle 4 (Goal-Driven Execution)\n\n**Trigger**: \"Fix the authentication system.\" The\nagent responds \"I will review the code, identify\nissues, make improvements, and test.\" That is a plan\nshape, not a verification plan.\n\n**Bad shape**: a four-step list of verbs (review,\nidentify, improve, test) with no checkable signal per\nstep.\n\n**Good shape**\n\n> \"Fix authentication\" is too broad to verify. Best\n> guess at the actual problem: sessions stay alive\n> after password change. Plan:\n>\n> 1. Write failing test: change password, prior\n>    session is rejected.\n> 2. Implement: invalidate sessions on change.\n> 3. Edge cases: concurrent change, multi-session.\n> 4. Regression: full auth suite green.\n>\n> Is the password-change-stale-session bug the right\n> target?\n\n**Test**: each step in the plan should have a one-\nsentence \"verify by ...\" attached. If a step lacks\none, the step is ill-specified.\n\n## AP-8: Multi-Step Plan Without Verification Gates\n\n**Maps to**: Principle 4 (Goal-Driven Execution)\n\n**Trigger**: \"Add rate limiting.\" The agent ships 300\nlines with Redis backend, configuration system, and\nmonitoring in one commit, with no per-step\nverification.\n\n**Bad shape**: one large commit covering basic limits,\nmiddleware extraction, Redis integration, and\nconfiguration. Nothing is independently shippable or\nrevertible.\n\n**Good shape**\n\n> Plan, each step independently verifiable:\n>\n> 1. In-memory limit on one endpoint. Verify: 11\n>    requests, the 11th gets 429.\n> 2. Extract to middleware, apply broadly. Verify:\n>    /users and /posts both rate-limit; existing tests\n>    pass.\n> 3. Redis backend. Verify: limits persist across\n>    restarts; two instances share counters.\n> 4. Per-endpoint config. Verify: /search 10/min,\n>    /users 100/min, parsed config tested.\n>\n> Start with step 1?\n\n**Test**: each step in the plan should be revertible\non its own. If reverting step 3 breaks step 4, the\nsteps are not independent and the plan needs a redraw.\n\n## How to Use This Module\n\nWhen reviewing your own diff or someone else's, name\nthe rail you see. \"This is AP-5: drive-by\nrefactoring\" travels faster than \"this could be\nsimpler somehow.\" Naming the rail is the first half\nof fixing the rail.\n\nCross-references for the rails:\n\n- AP-3, AP-4 connect to `Skill(imbue:scope-guard)` and\n  `Skill(leyline:additive-bias-defense)`\n- AP-5, AP-6 connect to `Skill(imbue:justify)` and\n  the `bounded-discovery.md` rule\n- AP-7, AP-8 connect to `Skill(imbue:proof-of-work)`\n  and `Skill(superpowers:test-driven-development)`\n\nFile v1.9.13:modules/senior-engineer-test.md\n\n# The Senior Engineer Test\n\nA three-question battery to apply to your own code\nbefore claiming it is done. The questions stand in\nfor the senior engineer who is not in the room.\n\nAdapted from a self-check Karpathy calls out for\nagentic coding: ask whether a senior engineer would\nsay this is overcomplicated. We expand the question\ninto three concrete sub-questions that map to common\nLLM coding failures.\n\n## The Question\n\n> Would a senior engineer who is busy and a little\n> grumpy say this code is overcomplicated?\n\nIf yes, the diff is not ready.\n\n## The Three Sub-Questions\n\n### Q1: Could this be 50% shorter without losing meaning?\n\nMost LLM-written code can be cut by a third to a\nhalf. If the diff is 200 lines, ask: which 100 lines\nexist because the agent felt clever, not because the\nproblem demanded them?\n\nCommon 50% wins:\n\n- Replace abstract base class plus two subclasses\n  with one function plus a parameter.\n- Replace try-except wrapping every call with a\n  single boundary handler.\n- Replace explicit getter and setter with direct\n  attribute access.\n- Replace nested conditionals with a flat early-return\n  pattern.\n\n### Q2: Are abstractions earning their weight?\n\nAn abstraction earns its weight when it is used three\nor more times, or when it isolates a real boundary\n(network, disk, locale). A class with one consumer is\nceremony. A factory with one product is ceremony.\n\nTest: for every type, class, or helper added, count\nthe call sites. One call site means the abstraction\ncosts more than it saves.\n\n### Q3: Could a junior dev follow this in six months?\n\nSix months means: docs may have rotted, original\ncontext is gone, the original author is on another\nteam. The code has to carry its own meaning.\n\nFailure signals:\n\n- Names that mean something only if you remember the\n  ticket\n- Comments that describe what the code does (the code\n  shows that) instead of why\n- Indirection that requires three jumps to find the\n  actual logic\n- Implicit invariants that nothing checks and nothing\n  documents\n\n## The Decision Tree\n\n```\nFor each of Q1, Q2, Q3:\n  - Yes -> next question\n  - No  -> stop and address before shipping\n\nIf three Yes -> ship\nIf any No   -> rework that dimension first\n```\n\nA No answer is not a failure of the agent; it is the\nagent doing its job. Catching the violation before\nthe senior engineer catches it is the entire point.\n\n## Worked Example\n\nDiff under review: a class hierarchy for a single\ndiscount calculation.\n\n- Q1 (50% shorter)? Yes obviously: one function\n  replaces five classes.\n- Q2 (abstractions earning weight)? No: zero\n  additional call sites for the strategy pattern.\n- Q3 (junior in six months)? No: two indirection hops\n  to find the multiplication.\n\nTwo No answers means rework. Replace the hierarchy\nwith the function. Now Q1, Q2, Q3 are all yes.\n\n## When the Test Does Not Apply\n\nThe senior-engineer test assumes the code will be\nread again. For a one-shot data migration that runs\nonce and is deleted, the test is too strict. See\n`tradeoff-acknowledgment.md` for the boundary cases.\n\n## Cross-References\n\n- `Skill(imbue:scope-guard)` formalizes Q2 (does the\n  abstraction earn its weight) into the Worthiness\n  formula.\n- `Skill(conserve:code-quality-principles)` is the\n  KISS / YAGNI / SOLID foundation that Q1 leans on.\n- `Skill(leyline:additive-bias-defense)` is the\n  burden-of-proof contract that backs Q2.\n\nFile v1.9.13:modules/tradeoff-acknowledgment.md\n\n# Tradeoff Acknowledgment: When Not to Apply These\n\nThe four principles bias toward caution. That bias\ncosts speed. For a substantial portion of coding work\nthe cost is worth paying. For a non-trivial minority,\nthe cost is wrong. This module names the boundary\nhonestly.\n\nThe upstream framing puts it as: \"These guidelines\nbias toward caution over speed. For trivial tasks,\nuse judgment.\" That sentence does the same work as\nthis module, just compressed.\n\n## When the Principles Do Not Apply\n\n### Trivial One-Line Fixes\n\nAsking three clarifying questions before fixing a\ntypo in a comment is a parody of caution. For diffs\nunder five lines with obvious intent, ship and move\non. Principle 1 (Think Before Coding) is for\nambiguous requests, not unambiguous ones.\n\n### Exploratory Spikes and Throwaway Scripts\n\nA 50-line script that runs once, produces a CSV, and\ngets deleted does not need the senior-engineer test.\nIt does not need TDD. It does not need careful\nabstraction analysis. The artifact's lifetime caps\nthe time worth investing in i\n\nArchive v1.9.12: 7 files, 14902 bytes\n\nFiles: modules/anti-patterns.md (8298b), modules/senior-engineer-test.md (3378b), modules/tradeoff-acknowledgment.md (3757b), modules/verifiable-goals.md (4501b), skill-card.md (2011b), SKILL.md (7513b), _meta.json (148b)\n\nArchive v1.0.0: 7 files, 15027 bytes\n\nFiles: modules/anti-patterns.md (8298b), modules/senior-engineer-test.md (3378b), modules/tradeoff-acknowledgment.md (3757b), modules/verifiable-goals.md (4501b), skill-card.md (2294b), SKILL.md (7513b), _meta.json (147b)","readmeExcerpt":"Skill: karpathy-principles Owner: athola Summary: Pre-implementation gate covering think-first, simplicity, surgical edits, and verifiable goals Tags: latest:1.9.19 Version history: v1.9.19 | 2026-08-26T13:12:45.526Z | user Release v1.9.19 v1.9.17 | 2026-07-30T05:34:19.068Z | user Release v1.9.17 v1.9.16 | 2026-07-14T19:51:02.878Z | user Release v1.9.16 v1.9.14 | 2026-06-30T18:00:06.428Z | user Release v1.9.14 v1.9.1","codeSnippets":[],"executableExamples":[{"language":"python","snippet":"def export_users(format='json'):\n    users = User.query.all()\n    if format == 'json':\n        with open('users.json', 'w') as f:\n            json.dump([u.to_dict() for u in users], f)"},{"language":"python","snippet":"@lru_cache(maxsize=1000)\nasync def search(query: str) -> List[Result]:\n    # 200 lines of caching, async, indexes, all picked\n    # without confirming what \"faster\" means\n    ..."},{"language":"python","snippet":"class DiscountStrategy(ABC):\n    @abstractmethod\n    def calculate(self, amount: float) -> float: ...\n\nclass PercentageDiscount(DiscountStrategy):\n    def __init__(self, p): self.p = p\n    def calculate(self, a): return a * (self.p / 100)\n\n# Plus FixedDiscount, DiscountConfig, DiscountCalculator\n# for what should be one function"},{"language":"python","snippet":"def calculate_discount(amount: float, percent: float) -> float:\n    return amount * (percent / 100)"},{"language":"python","snippet":"class PreferenceManager:\n    def __init__(self, db, cache=None, validator=None):\n        ...\n    def save(self, user_id, prefs,\n             merge=True, validate=True, notify=False):\n        # 60 lines of optional behavior"},{"language":"python","snippet":"def save_preferences(db, user_id: int, preferences: dict):\n    db.execute(\n        \"UPDATE users SET preferences = ? WHERE id = ?\",\n        (json.dumps(preferences), user_id),\n    )"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: karpathy-principles\ndescription: |\n  Pre-implementation gate covering think-first, simplicity, surgical edits, and verifiable goals\nversion: 1.9.8\ntriggers:\n  - karpathy\n  - coding-pitfalls\n  - synthesis\n  - entry-point\n  - discipline\n  - anti-overengineering\n  - TDD\n  - starting implementation to verify the approach\nmetadata: {\"openclaw\": {\"homepage\": \"https://github.com/athola/claude-night-market/tree/master/plugins/imbue\", \"emoji\": \"\\ud83e\\udd9e\", \"requires\": {\"config\": [\"night-market.imbue:scope-guard\", \"night-market.imbue:proof-of-work\", \"night-market.imbue:rigorous-reasoning\", \"night-market.leyline:additive-bias-defense\", \"night-market.conserve:code-quality-principles\"]}}}\nsource: claude-night-market\nsource_plugin: imbue\n---\n\n> **Night Market Skill** — ported from [claude-night-market/imbue](https://github.com/athola/claude-night-market/tree/master/plugins/imbue). For the full experience with agents, hooks, and commands, install the Claude Code plugin.\n\n\n> The models make wrong assumptions on your behalf and\n> just run along with them without checking. They don't\n> manage their confusion, don't seek clarifications,\n> don't surface inconsistencies, don't present\n> tradeoffs, don't push back when they should.\n>\n> -- Andrej Karpathy, on agentic coding failure modes\n\n## What This Is\n\nA four-principle contract for reducing the most common\nLLM coding pitfalls. Compact entry-point. Each\nprinciple has a deeper-dive skill in night-market;\nthis skill is the index, not the encyclopedia.\n\nDerivation: distilled by Forrest Chang\n(forrestchang/andrej-karpathy-skills, MIT) from\nKarpathy's observations. Full attribution in\n`references/source-attribution.md`.\n\n## When to Use\n\n- Before starting any coding task larger than a typo\n- During code review, to name the failure mode you see\n- After writing a diff, to self-audit before claiming\n  done\n- When training a junior engineer to read agent diffs\n\n## When NOT to Use\n\nThese principles bias toward caution over speed. For\ncases listed in `modules/tradeoff-acknowledgment.md`,\nuse judgment: trivial fixes, exploratory spikes,\ndocumentation-only edits, and time-boxed prototypes.\n\n## The Four Principles\n\n### 1. Think Before Coding\n\n**State assumptions. Surface confusion. Match tone to\nevidence.**\n\n- If multiple interpretations of the request exist,\n  list them. Do not silently pick.\n- If a simpler approach exists, name it. Push back\n  when the simpler path is correct.\n- If something is unclear, stop and ask. Hidden\n  assumptions are the cheapest bug to prevent and the\n  most expensive to find later.\n- Make claims no stronger than the evidence supports.\n  Calibrated tone beats confident hand-waving.\n\nDeep dives: `Skill(imbue:rigorous-reasoning)` for the\nsycophancy guard, `Skill(superpowers:brainstorming)`\nfor option generation, `/spec-kit:speckit-clarify`\ncommand for ambiguity drilldown.\n\n### 2. Simplicity First\n\n**Minimum code that solves the problem. Nothing\nspeculative.**\n\n> They really like to overcomplic"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7d107jg9jv602h9ytsegydq184a42s\",\n  \"slug\": \"nm-imbue-karpathy-principles\",\n  \"version\": \"1.9.19\",\n  \"publishedAt\": 1787749965526\n}"},{"path":"modules/anti-patterns.md","content":"# Anti-Patterns: Eight Named Drift Rails\n\nEach drift rail names a specific way LLM-driven coding\ngoes wrong, with a trigger pattern, a short bad\nexample, a short good example, and a one-sentence\ntest you can apply to your own diff. Naming the rail\nhelps recall.\n\nAdapted from observations distilled by Forrest Chang\nfrom Karpathy's notes; see\n`references/source-attribution.md` for the citation\nchain. Examples are rewritten in our own prose to\nillustrate the failure shape, not lifted verbatim.\n\n## AP-1: Hidden Assumptions\n\n**Maps to**: Principle 1 (Think Before Coding)\n\n**Trigger**: User asks for a feature with multiple\nunstated dimensions (scope, format, volume, privacy).\nThe agent picks one silently and ships.\n\n**Bad shape**\n\n```python\ndef export_users(format='json'):\n    users = User.query.all()\n    if format == 'json':\n        with open('users.json', 'w') as f:\n            json.dump([u.to_dict() for u in users], f)\n```\n\nAssumed: all users (privacy?), file path, fields,\nvolume.\n\n**Good shape**\n\n> Before implementing, I want to confirm:\n>\n> - Scope: all users, or filtered?\n> - Format: download, background job, or API endpoint?\n> - Fields: which? Some are sensitive.\n> - Volume: thousands or millions?\n>\n> Default proposal: a paginated JSON endpoint. Adjust?\n\n**Test**: count silent assumptions in the diff. Three\nor more is a rail violation.\n\n## AP-2: Multiple Interpretations Picked Silently\n\n**Maps to**: Principle 1 (Think Before Coding)\n\n**Trigger**: A request like \"make the search faster\"\nhas at least three valid readings (response time,\nthroughput, perceived speed). The agent picks one and\nimplements 200 lines without flagging the choice.\n\n**Bad shape**\n\n```python\n@lru_cache(maxsize=1000)\nasync def search(query: str) -> List[Result]:\n    # 200 lines of caching, async, indexes, all picked\n    # without confirming what \"faster\" means\n    ...\n```\n\n**Good shape**\n\n> \"Faster\" could mean response time, throughput, or\n> perceived speed. Each is a different fix:\n>\n> 1. Response time: indexes plus query caching (~2h)\n> 2. Throughput: async path plus pooling (~4h)\n> 3. Perceived: progressive load and partial results\n>    (~3h)\n>\n> Current: ~500ms typical. Which axis matters?\n\n**Test**: when the verb in the request is ambiguous\n(faster, better, cleaner, simpler), did the agent name\nthe alternatives or pick one?\n\n## AP-3: Strategy Pattern for One Function\n\n**Maps to**: Principle 2 (Simplicity First)\n\n**Trigger**: User asks for a single function. The\nagent ships an abstract base class, two implementing\nclasses, a config dataclass, and a coordinator class\nfor ten lines of arithmetic.\n\n**Bad shape**\n\n```python\nclass DiscountStrategy(ABC):\n    @abstractmethod\n    def calculate(self, amount: float) -> float: ...\n\nclass PercentageDiscount(DiscountStrategy):\n    def __init__(self, p): self.p = p\n    def calculate(self, a): return a * (self.p / 100)\n\n# Plus FixedDiscount, DiscountConfig, DiscountCalculator\n# for what should be one function\n```\n\n**Good shape**\n\n```pyt"},{"path":"modules/senior-engineer-test.md","content":"# The Senior Engineer Test\n\nA three-question battery to apply to your own code\nbefore claiming it is done. The questions stand in\nfor the senior engineer who is not in the room.\n\nAdapted from a self-check Karpathy calls out for\nagentic coding: ask whether a senior engineer would\nsay this is overcomplicated. We expand the question\ninto three concrete sub-questions that map to common\nLLM coding failures.\n\n## The Question\n\n> Would a senior engineer who is busy and a little\n> grumpy say this code is overcomplicated?\n\nIf yes, the diff is not ready.\n\n## The Three Sub-Questions\n\n### Q1: Could this be 50% shorter without losing meaning?\n\nMost LLM-written code can be cut by a third to a\nhalf. If the diff is 200 lines, ask: which 100 lines\nexist because the agent felt clever, not because the\nproblem demanded them?\n\nCommon 50% wins:\n\n- Replace abstract base class plus two subclasses\n  with one function plus a parameter.\n- Replace try-except wrapping every call with a\n  single boundary handler.\n- Replace explicit getter and setter with direct\n  attribute access.\n- Replace nested conditionals with a flat early-return\n  pattern.\n\n### Q2: Are abstractions earning their weight?\n\nAn abstraction earns its weight when it is used three\nor more times, or when it isolates a real boundary\n(network, disk, locale). A class with one consumer is\nceremony. A factory with one product is ceremony.\n\nTest: for every type, class, or helper added, count\nthe call sites. One call site means the abstraction\ncosts more than it saves.\n\n### Q3: Could a junior dev follow this in six months?\n\nSix months means: docs may have rotted, original\ncontext is gone, the original author is on another\nteam. The code has to carry its own meaning.\n\nFailure signals:\n\n- Names that mean something only if you remember the\n  ticket\n- Comments that describe what the code does (the code\n  shows that) instead of why\n- Indirection that requires three jumps to find the\n  actual logic\n- Implicit invariants that nothing checks and nothing\n  documents\n\n## The Decision Tree\n\n```\nFor each of Q1, Q2, Q3:\n  - Yes -> next question\n  - No  -> stop and address before shipping\n\nIf three Yes -> ship\nIf any No   -> rework that dimension first\n```\n\nA No answer is not a failure of the agent; it is the\nagent doing its job. Catching the violation before\nthe senior engineer catches it is the entire point.\n\n## Worked Example\n\nDiff under review: a class hierarchy for a single\ndiscount calculation.\n\n- Q1 (50% shorter)? Yes obviously: one function\n  replaces five classes.\n- Q2 (abstractions earning weight)? No: zero\n  additional call sites for the strategy pattern.\n- Q3 (junior in six months)? No: two indirection hops\n  to find the multiplication.\n\nTwo No answers means rework. Replace the hierarchy\nwith the function. Now Q1, Q2, Q3 are all yes.\n\n## When the Test Does Not Apply\n\nThe senior-engineer test assumes the code will be\nread again. For a one-shot data migration that runs\nonce and is deleted, the test is too strict. See\n`trad"},{"path":"modules/tradeoff-acknowledgment.md","content":"# Tradeoff Acknowledgment: When Not to Apply These\n\nThe four principles bias toward caution. That bias\ncosts speed. For a substantial portion of coding work\nthe cost is worth paying. For a non-trivial minority,\nthe cost is wrong. This module names the boundary\nhonestly.\n\nThe upstream framing puts it as: \"These guidelines\nbias toward caution over speed. For trivial tasks,\nuse judgment.\" That sentence does the same work as\nthis module, just compressed.\n\n## When the Principles Do Not Apply\n\n### Trivial One-Line Fixes\n\nAsking three clarifying questions before fixing a\ntypo in a comment is a parody of caution. For diffs\nunder five lines with obvious intent, ship and move\non. Principle 1 (Think Before Coding) is for\nambiguous requests, not unambiguous ones.\n\n### Exploratory Spikes and Throwaway Scripts\n\nA 50-line script that runs once, produces a CSV, and\ngets deleted does not need the senior-engineer test.\nIt does not need TDD. It does not need careful\nabstraction analysis. The artifact's lifetime caps\nthe time worth investing in its quality.\n\nTest: if the script will run again next week, treat\nit like real code. If you will throw it away in an\nhour, do not over-invest.\n\n### Documentation-Only Changes\n\nStyle drift in docs is often the point. Rewriting a\nparagraph for clarity touches every line by design.\nPrinciple 3 (Surgical Changes) was written for code\ndiffs, where adjacent edits hide intent. Prose is\ndifferent.\n\n### Time-Boxed Prototypes\n\nA \"by Friday or we move on\" prototype is a different\nartifact from a feature. Verifiable success criteria\nfor a prototype look like \"the demo runs end to\nend,\" not \"the test suite is green.\" Calibrate\nambition to the deadline.\n\n### Production Fires\n\nWhen the database is on fire, \"let's write a failing\ntest first\" is the wrong move. Stop the fire, then\nwrite the test that prevents the next fire. The Iron\nLaw assumes a normal-operations context.\n\n## Contrarian Voices Worth Engaging\n\nThree voices push back on rigorous-by-default LLM\ncoding rules. Their critiques sharpen the boundary.\n\n**Simon Willison** (\"Not all AI-assisted programming\nis vibe coding,\" March 2025) defends throwaway\nprototyping as legitimate. His golden rule: do not\ncommit code you cannot explain. That rule is\ncompatible with everything in this skill, but it\nmakes the throwaway-prototype boundary explicit.\n\n**Mastering Product HQ** (\"What Karpathy's CLAUDE.md\nmisses\") argues code simplicity does not equal scope\nsimplicity. A 50-line solution to the wrong problem\nis still waste. The principles help with how to\nbuild; they do not help with what to build. For\n\"what,\" see `Skill(imbue:scope-guard)` and\n`Skill(imbue:feature-review)`.\n\n**NMN.gl** (\"Vibe Coding Considered Harmful,\" March\n2025) warns that vibed black boxes compound. This is\nadjacent support for the principles, not pushback,\nbut it names the real cost of skipping them at scale:\neach black box you accept becomes a future debugging\nexpense.\n\n## The Honest Bottom Line\n\nThese principles solve a "}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Pre-implementation gate covering think-first, simplicity, surgical edits, and verifiable goals Skill: karpathy-principles Owner: athola Summary: Pre-implementation gate covering think-first, simplicity, surgical edits, and verifiable goals Tags: latest:1.9.19 Version history: v1.9.19 | 2026-08-26T13:12:45.526Z | user Release v1.9.19 v1.9.17 | 2026-07-30T05:34:19.068Z | user Release v1.9.17 v1.9.16 | 2026-07-14T19:51:02.878Z | user Release v1.9.16 v1.9.14 | 2026-06-30T18:00:06.428Z | user Release v1.9.14 v1.9.1","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1822,"uniquenessScore":50,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T17:54:19.397Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T17:54:19.397Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T21:00:25.862Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}