{"id":"3ac9092b-2b9a-4c33-85e3-583dfc22c740","entityType":"agent","slug":"clawhub-x-rayluan-openclaw-self-improvement","name":"OpenClaw Self-Improvement","canonicalUrl":"https://www.xpersona.co/agent/clawhub-x-rayluan-openclaw-self-improvement","canonicalPath":"/agent/clawhub-x-rayluan-openclaw-self-improvement","generatedAt":"2026-10-09T16:29:32.105Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-09T08:11:38.169Z","emptyReason":null},"description":"A reusable operator-guided workflow improvement skill for OpenClaw and ClawLite that turns repeated failures into logged learnings, binary eval loops, SOPs,... Skill: OpenClaw Self-Improvement Owner: x-rayluan Summary: A reusable operator-guided workflow improvement skill for OpenClaw and ClawLite that turns repeated failures into logged learnings, binary eval loops, SOPs,... Tags: latest:0.2.11 Version history: v0.2.11 | 2026-05-01T08:13:57.502Z | user Add scorecard repair loop with recovery ticket generation. v0.2.10 | 2026-05-01T08:07:07.960Z | user Add harness observabi","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 3.4K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s17c0n0ce78mre7mrytgzehcwn83hkp1:openclaw-self-improvement","sourceUrl":"https://clawhub.ai/x-rayluan/openclaw-self-improvement","homepage":"https://clawhub.ai/x-rayluan/skills/openclaw-self-improvement","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/x-rayluan/openclaw-self-improvement","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/x-rayluan/skills/openclaw-self-improvement","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":64,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"A reusable operator-guided workflow improvement skill for OpenClaw and ClawLite that turns repeated failures into logged learnings, binary eval loops, SOPs,... "},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T08:11:38.169Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T08:11:38.169Z","emptyReason":null},"stars":null,"forks":null,"downloads":3419,"packageName":null,"latestVersion":"0.2.11","tractionLabel":"3.4K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T08:11:38.169Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T08:11:38.169Z","lastCrawledAt":"2026-10-09T08:11:38.169Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T08:11:38.169Z","lastVerifiedAt":null,"highlights":[{"version":"0.2.11","createdAt":"2026-05-01T08:13:57.502Z","changelog":"Add scorecard repair loop with recovery ticket generation.","fileCount":29,"zipByteSize":35330},{"version":"0.2.10","createdAt":"2026-05-01T08:07:07.960Z","changelog":"Add harness observability helpers and daily agent scorecards.","fileCount":26,"zipByteSize":30499},{"version":"0.2.9","createdAt":"2026-04-15T07:11:14.329Z","changelog":"Align metadata and write-boundary declarations.","fileCount":22,"zipByteSize":23467},{"version":"0.2.8","createdAt":"2026-04-15T07:04:20.417Z","changelog":"Harden promotion paths and align docs.","fileCount":21,"zipByteSize":22781},{"version":"0.2.7","createdAt":"2026-04-14T06:48:15.839Z","changelog":"Update listing descriptions for SEO and GEO clarity.","fileCount":20,"zipByteSize":22060},{"version":"0.2.6","createdAt":"2026-03-30T06:27:30.428Z","changelog":"Sync SKILL.md summary with short multilingual description","fileCount":15,"zipByteSize":20083},{"version":"0.2.5","createdAt":"2026-03-30T06:23:04.102Z","changelog":"Shorten package description to avoid UI truncation while keeping multilingual cues","fileCount":15,"zipByteSize":20437},{"version":"0.2.4","createdAt":"2026-03-30T06:15:06.565Z","changelog":"Update package description with inline multilingual text for ClawHub listing","fileCount":15,"zipByteSize":20671}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17c0n0ce78mre7mrytgzehcwn83hkp1:openclaw-self-improvement","setupComplexity":"low","setupSteps":["Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-x-rayluan-openclaw-self-improvement/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-x-rayluan-openclaw-self-improvement/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-x-rayluan-openclaw-self-improvement/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-x-rayluan-openclaw-self-improvement/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-x-rayluan-openclaw-self-improvement/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-x-rayluan-openclaw-self-improvement/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-09T16:29:32.101Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-x-rayluan-openclaw-self-improvement/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-x-rayluan-openclaw-self-improvement/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-x-rayluan-openclaw-self-improvement/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-x-rayluan-openclaw-self-improvement/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-09T08:11:38.169Z","emptyReason":null},"readme":"Skill: OpenClaw Self-Improvement\n\nOwner: x-rayluan\n\nSummary: A reusable operator-guided workflow improvement skill for OpenClaw and ClawLite that turns repeated failures into logged learnings, binary eval loops, SOPs,...\n\nTags: latest:0.2.11\n\nVersion history:\n\nv0.2.11 | 2026-05-01T08:13:57.502Z | user\n\nAdd scorecard repair loop with recovery ticket generation.\n\nv0.2.10 | 2026-05-01T08:07:07.960Z | user\n\nAdd harness observability helpers and daily agent scorecards.\n\nv0.2.9 | 2026-04-15T07:11:14.329Z | user\n\nAlign metadata and write-boundary declarations.\n\nv0.2.8 | 2026-04-15T07:04:20.417Z | user\n\nHarden promotion paths and align docs.\n\nv0.2.7 | 2026-04-14T06:48:15.839Z | user\n\nUpdate listing descriptions for SEO and GEO clarity.\n\nv0.2.6 | 2026-03-30T06:27:30.428Z | user\n\nSync SKILL.md summary with short multilingual description\n\nv0.2.5 | 2026-03-30T06:23:04.102Z | user\n\nShorten package description to avoid UI truncation while keeping multilingual cues\n\nv0.2.4 | 2026-03-30T06:15:06.565Z | user\n\nUpdate package description with inline multilingual text for ClawHub listing\n\nv0.2.3 | 2026-03-30T06:11:10.491Z | user\n\nFix SKILL description formatting so multilingual text is rendered inline on directory listing\n\nv0.2.2 | 2026-03-30T05:58:25.143Z | user\n\nAdd multilingual SKILL description (EN/ZH/JA/KO/ES) for openclaw-self-improvement intro\n\nv0.2.1 | 2026-03-29T14:12:08.945Z | user\n\nSafer packaging: workspace-local default exports, explicit local-only safety notes, removed embedded .git metadata.\n\nv0.2.0 | 2026-03-27T13:46:02.088Z | user\n\nAdd eval-loop support, decision rules, practical examples, experiment summary helper, and self-improvement daily routine integration.\n\nv0.1.0 | 2026-03-14T18:48:52.348Z | user\n\nInitial self-improvement skill with learnings logs, promotion flow, and Obsidian support.\n\nArchive index:\n\nArchive v0.2.11: 29 files, 35330 bytes\n\nFiles: package.json (1049b), README.md (11969b), references/decision-rules.md (3334b), references/eval-loop.md (3305b), references/examples.md (4367b), references/promotion-guide.md (1175b), references/schema.md (1846b), releases/v0.2.0-clawhub-listing.md (1678b), releases/v0.2.0-github-release.md (3744b), releases/v0.2.1-github-release.md (119b), releases/v0.2.10-github-release.md (829b), releases/v0.2.11-github-release.md (830b), releases/v0.2.2-github-release.md (119b), releases/v0.2.3-github-release.md (127b), releases/v0.2.4-github-release.md (147b), releases/v0.2.7-github-release.md (147b), releases/v0.2.8-github-release.md (221b), releases/v0.2.9-github-release.md (204b), scripts/analyze-openclaw-failures.mjs (2558b), scripts/daily-agent-scorecard.mjs (2270b), scripts/experiment-summary.mjs (1749b), scripts/harness-improvement-lib.mjs (13597b), scripts/log-experiment.mjs (1699b), scripts/log-learning.mjs (1683b), scripts/promote-learning.mjs (1263b), skill-card.md (2324b), SKILL.md (8398b), tests/harness-repair-loop.test.mjs (2540b), _meta.json (145b)\n\nFile v0.2.11:SKILL.md\n\n---\nname: openclaw-self-improvement\ndescription: A reusable operator-guided workflow improvement skill for OpenClaw and ClawLite that turns repeated failures into logged learnings, binary eval loops, SOPs, checklists, and proof-based operational improvements.\nmetadata:\n  {\n    \"openclaw\":\n      {\n        \"requires\": { \"bins\": [\"node\"] },\n        \"writes\": [\".learnings/\", \"memory/harness-backlog-latest.md\", \"mission-control/data/delivery-receipts/agent-scorecard-YYYY-MM-DD.md\", \"AGENTS.md\", \"TOOLS.md\", \"SOUL.md\"],\n        \"env\": [\"WORKSPACE\", \"OBSIDIAN_LEARNINGS_DIR\"],\n        \"network\": false,\n        \"notes\": \"Local-file workflow only. Promotion writes should be reviewed and can be previewed with --dry-run.\"\n      }\n  }\n---\n\n# OpenClaw / ClawLite Self-Improvement\n\nUse this skill to turn mistakes, corrections, blockers, and better approaches into durable operating knowledge.\n\n## What problem this solves\nAI ops often repeat the same failures because mistakes stay in chat history instead of becoming system rules. This skill creates a lightweight improvement loop:\n- log failures and learnings\n- separate errors from feature requests\n- run small eval-driven experiments on repeated failures\n- classify harness/runtime failures instead of blaming vague “model issues”\n- generate daily agent scorecards from real evidence chains\n- promote important patterns into AGENTS.md / TOOLS.md / SOUL.md\n- write operator notes into Obsidian vault\n- support stricter acceptance via Karen / Mission Control\n\n## When to use\nUse this skill when the user asks:\n- \"make the agent improve itself\"\n- \"capture learnings\"\n- \"log mistakes so we do not repeat them\"\n- \"record blockers / corrections / feature gaps\"\n- \"build a self-improving OpenClaw workflow\"\n- \"operationalize lessons learned\"\n- \"test whether this new rule actually helps\"\n- \"run an eval loop on this workflow/skill/SOP\"\n- \"should we keep this new guardrail or discard it\"\n- \"why did the agents fail today\"\n- \"why is daily marketing not closing automatically\"\n- \"classify OpenClaw harness failures\"\n- \"generate agent delivery scorecard\"\n\n## Files this skill uses\n- `.learnings/LEARNINGS.md`\n- `.learnings/ERRORS.md`\n- `.learnings/FEATURE_REQUESTS.md`\n- `.learnings/EXPERIMENTS.md`\n- `memory/harness-backlog-latest.md`\n- `mission-control/data/delivery-receipts/agent-scorecard-YYYY-MM-DD.md`\n- Optional export under `.learnings/exports/obsidian/` by default, or `OBSIDIAN_LEARNINGS_DIR` if explicitly configured\n\n## Safety boundaries\n- Local-file workflow only, no network I/O\n- Promotion can append to `AGENTS.md`, `TOOLS.md`, or `SOUL.md`\n- Always review promotion targets first, or run `scripts/promote-learning.mjs ... --dry-run`\n- `OBSIDIAN_LEARNINGS_DIR` should only point at a path you intend to modify\n\n## Command examples\n```bash\nnode {baseDir}/scripts/log-learning.mjs learning \"Summary\" \"Details\" \"Suggested action\"\nnode {baseDir}/scripts/log-learning.mjs error \"Summary\" \"Error details\" \"Suggested fix\"\nnode {baseDir}/scripts/log-learning.mjs feature \"Capability name\" \"User context\" \"Suggested implementation\"\nnode {baseDir}/scripts/log-learning.mjs experiment \"Target problem\" \"Baseline failure\" \"Single mutation to test\"\nnode {baseDir}/scripts/log-experiment.mjs \"Target problem\" \"Baseline failure\" \"Single mutation\" \"eval1|eval2|eval3\" \"Result summary\" \"testing\"\nnode {baseDir}/scripts/promote-learning.mjs workflow \"Rule text\"\nnode {baseDir}/scripts/analyze-openclaw-failures.mjs --output /Users/m1/.openclaw/workspace/memory/harness-backlog-latest.md\nnode {baseDir}/scripts/daily-agent-scorecard.mjs --output /Users/m1/.openclaw/workspace/mission-control/data/delivery-receipts/agent-scorecard-$(date +%F).md\nnode {baseDir}/scripts/daily-agent-scorecard.mjs --repair --output /Users/m1/.openclaw/workspace/mission-control/data/delivery-receipts/agent-scorecard-$(date +%F).md\n```\n\n## Categories\n### learning\nUse for:\n- user corrections\n- better recurring workflows\n- tool gotchas\n- operational lessons\n\n### error\nUse for:\n- command failures\n- integration failures\n- runtime blockers\n- broken release / deploy behavior\n\n### feature\nUse for:\n- missing capability requests\n- operator workflow gaps\n- recurring requests that deserve a build item\n\n### experiment\nUse for:\n- repeated failures that need a tested guardrail\n- checklist/SOP/schema changes that should be validated before broad promotion\n- keep/discard decisions on new operating rules\n- binary eval loops for skills, workflows, receipts, summaries, or deploy closeout rules\n\n### harness\nUse for:\n- gateway, channel, provider, tool, session, or platform failures\n- repeated \"agent did not respond / did not finish / forgot identity\" incidents\n- daily workflow failures where Mission Control says one thing but proof chains say another\n- scorecards that compare agent delivery against real receipts, URLs, and closeout evidence\n\nDefault failure taxonomy:\n- `NetworkPolicyBlocked` - provider/tool blocked by local or external network policy\n- `GatewayUnavailable` - gateway process, port, websocket, or reachability failure\n- `SessionContextRot` - stale session, stale skill snapshot, identity drift, or outdated config context\n- `SkillMissing` - expected skill absent from installed path or session snapshot\n- `ToolInvalidArguments` - malformed tool/edit call or bad argument shape\n- `ProviderError` - provider/model/API failure not caused by network policy\n- `ExternalPlatformBlocked` - X/LinkedIn/Facebook/Feishu/etc. platform/API/login/visibility blocker\n- `HumanApprovalRequired` - real approval boundary for external, destructive, production, money, or ambiguous action\n\nHarness workflow:\n1. Scan logs and receipts with `scripts/analyze-openclaw-failures.mjs`.\n2. Generate same-day agent scorecard with `scripts/daily-agent-scorecard.mjs`.\n3. Run `scripts/daily-agent-scorecard.mjs --repair` to create/update recovery tickets for failed, blocked, or pending lanes.\n4. Convert repeated classes into an `error`, `experiment`, or promoted rule.\n5. Do not call a workflow closed until the scorecard has proof links or explicit blocker evidence.\n\nRepair loop rules:\n- Every failed/blocked/pending lane should have a `failureClass`, `repairState`, `nextAction`, `repeatCount7d`, and evidence.\n- `ProofMissing`, `UpstreamMissing`, and `HumanApprovalRequired` must not be blindly retried.\n- Repeated `agent + lane + failureClass` failures within 7 days should become `EXPERIMENT_REQUIRED`.\n- Recovery tickets should be written under `mission-control/data/recovery-tickets-v3/YYYY-MM-DD/`.\n\n## Promotion targets\n- `AGENTS.md` → workflow / delegation / execution rules\n- `TOOLS.md` → tool gotchas, secrets locations, environment routing rules\n- `SOUL.md` → behavior / communication / non-negotiable principles\n- Obsidian vault → reusable operator log and content proof asset\n\n## Karen / Mission Control compatibility\nThis skill is designed to work with stricter ops governance:\n- Karen can reference learnings when repeated failures happen\n- Mission Control can treat promoted learnings as new operating rules\n- recurring blockers can be elevated from chat into tracked operational knowledge\n- experiments can test whether a new summary contract, receipt rule, or deploy closeout guardrail actually reduced the failure pattern\n\n## Eval loop rule\nWhen a repeated failure is turning into a new rule/SOP/checklist, do not only log it.\nAlso:\n1. define 3-5 binary evals\n2. record the baseline failure state\n3. change one thing at a time\n4. re-check the same evals\n5. classify the change as keep / discard / partial_keep\n\nUse `{baseDir}/references/eval-loop.md` for the experiment format and examples.\n\n## Output goal\nA good use of this skill should produce one of:\n- a durable learning entry\n- a durable error entry\n- a durable feature request entry\n- a durable experiment entry with binary evals\n- a promoted rule in AGENTS.md / TOOLS.md / SOUL.md\n- an Obsidian vault operations note\n\n## Important limits\n- Logging is not the same as fixing.\n- Do not treat a learning entry as closure for a broken deliverable.\n- Use this skill to reduce repeated mistakes, not to excuse them.\n\n## References\n- `{baseDir}/references/schema.md`\n- `{baseDir}/references/promotion-guide.md`\n- `{baseDir}/references/eval-loop.md`\n- `{baseDir}/references/examples.md`\n- `{baseDir}/references/decision-rules.md`\n\n- `{baseDir}/references/eval-loop.md`\n- `{baseDir}/references/examples.md`\n\nFile v0.2.11:README.md\n\n# OpenClaw Self-Improvement\n\n**OpenClaw Self-Improvement** is a reusable agent skill for turning repeated AI-agent mistakes into durable operational improvements, measurable guardrails, and inspectable workflow upgrades.\n\n## Landing-page summary\n\nMost AI agents do not really improve. They repeat mistakes, hide partial failures behind optimistic language, and leave lessons trapped in chat history.\n\nOpenClaw Self-Improvement gives you a practical operating loop for **self-improving AI agents**:\n- capture repeated failures\n- test one guardrail at a time\n- verify whether it actually reduces failure\n- promote proven fixes into SOPs, checklists, policies, and workflow rules\n\nIf you want **AI agents that get more reliable over time**, **multi-agent workflows that stop repeating the same mistakes**, or **proof-based QA for agent operations**, this skill is built for that exact use case.\n\n## Multilingual summary\n\n**中文：** 这是一个面向 AI agent 自我改进的实战型 skill，用来减少重复犯错、建立 guardrails、验证修复是否真的有效，并把经验沉淀成 SOP、检查清单和可复用规则。\n\n**日本語：** これは自己改善する AI エージェント向けの実践的な skill です。繰り返し発生する失敗を減らし、ガードレールを検証し、改善を SOP・チェックリスト・再利用可能な運用ルールへ昇格させます。\n\n**한국어：** 이 스킬은 스스로 개선하는 AI 에이전트를 위한 운영형 skill입니다. 반복 실수를 줄이고, 가드레일이 실제로 효과가 있는지 검증하며, 개선 사항을 SOP·체크리스트·재사용 가능한 규칙으로 승격합니다.\n\n**Español：** Esta skill está diseñada para agentes de IA que deben mejorar con el tiempo. Ayuda a reducir errores repetidos, validar guardrails operativos y convertir mejoras en SOP, checklists y reglas reutilizables.\n\nIf you want a practical way to deploy OpenClaw with cheaper tokens, BYOK flexibility, and operator control, see **[ClawLite](https://clawlite.ai)**.\n\n## TL;DR\n\nIf you are looking for a practical system for **self-improving AI agents**, **AI workflow optimization**, **multi-agent failure prevention**, **binary eval loops**, or **agent operations QA**, this skill is designed for that exact job.\n\nIt helps OpenClaw / ClawLite operators and agent teams:\n- log recurring failures\n- separate one-off errors from reusable lessons\n- run lightweight **binary eval loops** on new guardrails\n- classify changes as **keep**, **partial_keep**, or **discard**\n- promote proven fixes into SOPs, checklists, workflow rules, and operating policy\n\nIf you care about reducing fake-complete states, tightening QA truth, improving deploy closeout, and making agent learning inspectable, this skill is built for that job.\n\n---\n\n## Why this skill exists\n\nMany AI systems say they \"learn,\" but most only store lessons in chat history or loose notes.\n\nThat is not enough.\n\nOperationally, repeated failures tend to come back in the same forms:\n- delivery gets described as complete before proof exists\n- receipts are missing or too thin\n- back-end fixes never reach the operator-facing surface\n- code-ready states get confused with production-ready states\n- teams add new rules without checking whether those rules actually reduce failure\n\nOpenClaw Self-Improvement gives you a lightweight operating loop for fixing that.\n\n---\n\n## What problem it solves\n\nThis skill helps with:\n- **self-improving AI workflows**\n- **AI operations learning loops**\n- **binary evals for agent guardrails**\n- **Mission Control truth-state improvement**\n- **deploy closeout verification**\n- **receipt / proof completeness**\n- **repeated failure prevention in multi-agent systems**\n- **AI agent reliability engineering**\n- **agent QA systems for OpenClaw, ClawLite, and similar stacks**\n\nIt is especially useful in OpenClaw-style environments where multiple agents, tools, SOPs, and truth surfaces interact.\n\n## Who this is for\n\nThis skill is useful for:\n- OpenClaw operators\n- ClawLite growth / marketing / QA lanes\n- AI agent builders who need durable postmortems instead of vague “reflection”\n- teams running multi-agent workflows with receipts, truth surfaces, and closeout gates\n- anyone trying to reduce repeated AI-agent mistakes in production-like operations\n\n---\n\n## What’s new in v0.2.0\n\n### New capabilities\n- **Experiment mode** for repeated failures\n- **Binary eval loops** for testing whether a new guardrail or SOP actually helps\n- **Keep / partial_keep / discard** decision model\n- **Practical examples** for:\n  - Mission Control summary quality\n  - deploy closeout / production verification\n- **Experiment summary helper** for surfacing unresolved follow-up debt\n- **Decision rules** for when to log, experiment, or promote\n- **Daily routine integration** with heartbeat, Karen QA, AGENTS, and ops policy\n\n### Why it matters\nThis release makes self-improvement more operational.\n\nInstead of stopping at “lesson learned,” you can now:\n1. capture the repeated problem\n2. define a baseline\n3. test one change at a time\n4. evaluate with binary checks\n5. keep, discard, or mark `partial_keep`\n\nThat makes the improvement loop much more auditable and much less hand-wavy.\n\n---\n\n## Core workflow\n\nOpenClaw Self-Improvement now supports a practical loop:\n\n1. **Capture** a learning, error, feature request, or experiment\n2. **Store** it in structured local files\n3. **Experiment** when a repeated failure needs a tested guardrail\n4. **Evaluate** the change with binary checks\n5. **Promote** proven fixes into durable rules, SOPs, or policies\n6. **Track follow-up debt** when a fix is only partial\n\n---\n\n## Best use cases\n\nUse this skill when you want to:\n- capture lessons so agents stop repeating the same mistake\n- log recurring operational errors\n- track feature gaps revealed by repeated work\n- test whether a new workflow rule really improves outcomes\n- run a lightweight eval loop on a skill, SOP, checklist, schema, or handoff rule\n- decide whether a new guardrail should be kept, discarded, or promoted\n- build a self-improving OpenClaw or ClawLite operating loop\n\nTypical targets include:\n- Mission Control summary quality\n- deploy closeout gates\n- receipt requirements\n- QA wording rules\n- truth-surface rendering checks\n- handoff contracts between agents\n\n---\n\n## Files it manages\n\n- `.learnings/LEARNINGS.md`\n- `.learnings/ERRORS.md`\n- `.learnings/FEATURE_REQUESTS.md`\n- `.learnings/EXPERIMENTS.md`\n- Optional Obsidian export directory via `OBSIDIAN_LEARNINGS_DIR`\n- Default local export fallback: `.learnings/exports/obsidian/`\n- Safe by default: no hard-coded external vault path, and `scripts/promote-learning.mjs` prints the resolved write target before writing\n\n---\n\n## Install\n\n```bash\nnpm install\n```\n\n---\n\n## Usage\n\n### Log a learning\n\n```bash\nnode scripts/log-learning.mjs learning \"Summary\" \"Details\" \"Suggested action\"\n```\n\n### Log an error\n\n```bash\nnode scripts/log-learning.mjs error \"Summary\" \"Error details\" \"Suggested fix\"\n```\n\n### Log a feature request\n\n```bash\nnode scripts/log-learning.mjs feature \"Capability name\" \"User context\" \"Suggested implementation\"\n```\n\n### Log a tested experiment\n\n```bash\nnode scripts/log-experiment.mjs \"Target problem\" \"Baseline failure\" \"Single mutation\" \"eval1|eval2|eval3\" \"Result summary\" \"testing\"\n```\n\n### Promote a rule\n\n```bash\nnode scripts/promote-learning.mjs workflow \"Rule text\"\nnode scripts/promote-learning.mjs obsidian \"Reusable learning\" --dry-run\n```\n\n### Summarize experiment outcomes\n\n```bash\nnode scripts/experiment-summary.mjs\n```\n\n---\n\n## Decision model\n\nThis skill uses three levels of action:\n\n### 1. Log only\nUse when:\n- the issue happened once\n- root cause is still unclear\n- there is not enough evidence yet to make a rule\n\n### 2. Experiment\nUse when:\n- the issue repeated 2+ times\n- a new guardrail / SOP / checklist / schema change is being proposed\n- you can define 3–5 binary evals\n\n### 3. Promote\nUse when:\n- the rule is clearly right and low-risk\n- the issue is severe enough that waiting would be irresponsible\n- the rule is about ownership, truth, or a non-negotiable operating principle\n\n---\n\n## Practical examples\n\nThe skill now includes concrete examples for:\n- **Mission Control summary link-complete gates**\n- **ClawLite deploy closeout gates**\n\nThese examples show how to:\n- define the repeated failure\n- capture the baseline\n- propose one mutation\n- evaluate with binary checks\n- classify the outcome as `keep`, `partial_keep`, or `discard`\n\n---\n\n## Promotion targets\n\nPromote proven improvements into:\n- `AGENTS.md` — workflow / delegation / execution rules\n- `TOOLS.md` — tool gotchas and environment routing rules\n- `SOUL.md` — behavior / communication / non-negotiable principles\n- `docs/ops/*.md` — SOPs, policy, and operating contracts\n- Obsidian vault — reusable operator notes and operational memory\n\n---\n\n## Important limits\n\n- Logging is **not** the same as fixing.\n- A learning entry does **not** close a broken deliverable.\n- A back-end-only improvement is not complete if the visible operator-facing surface is still stale.\n- `partial_keep` should be treated as **active follow-up debt**, not as closure.\n\n---\n\n## Repository contents\n\n- `SKILL.md` — agent-facing routing and usage guidance\n- `scripts/log-learning.mjs` — append a learning / error / feature request / experiment\n- `scripts/log-experiment.mjs` — append a structured experiment with binary evals\n- `scripts/experiment-summary.mjs` — summarize keep / partial_keep / discard outcomes and flag follow-up debt\n- `scripts/promote-learning.mjs` — promote a lesson into durable operating rules with explicit path echo and optional `--dry-run`\n- `scripts/analyze-openclaw-failures.mjs` — classify OpenClaw gateway, provider, session, tool, and platform failures\n- `scripts/daily-agent-scorecard.mjs` — generate a same-day agent delivery scorecard from Mission Control and session evidence\n- `scripts/daily-agent-scorecard.mjs --repair` — create deterministic recovery tickets for failed, blocked, and pending lanes\n- `references/schema.md` — data structure guidance\n- `references/promotion-guide.md` — what to promote and where\n- `references/eval-loop.md` — how to run lightweight binary-eval improvement loops\n- `references/examples.md` — practical examples for summary gates and deploy closeout gates\n- `references/decision-rules.md` — when to log only, run an experiment, or promote immediately\n\n---\n\n## SEO / GEO positioning\n\nThis skill is intentionally legible to both search engines and AI answer engines because it is built around concrete, reusable operational concepts rather than vague “AI reflection” language.\n\n### Primary search themes\n- self-improving AI workflows\n- self-improving agent systems\n- AI agent reliability engineering\n- binary eval loops for agent guardrails\n- repeated failure prevention in multi-agent systems\n- deploy closeout verification\n- proof-based QA for AI operations\n- operational learning loops for OpenClaw / ClawLite style stacks\n\n### Why that matters\nThese phrases map to real operator intent:\n- “how do I stop my AI agents from repeating mistakes?”\n- “how do I build a self-improving agent workflow?”\n- “how do I test whether a new guardrail actually works?”\n- “how do I make AI operations auditable?”\n\nThat makes the skill easier to explain, cite, retrieve, and reuse than generic memory or reflection systems.\n\n---\n\n## Bottom line\n\nIf you want OpenClaw to improve over time instead of repeating the same mistakes across sessions, this repo gives you:\n- an operational memory loop\n- a lightweight eval loop for testing whether a new guardrail actually helps\n- a durable promotion path from mistake -> experiment -> policy -> reusable system asset\n- a decision framework for when to log, experiment, or promote\n- a way to keep unresolved partial improvements visible until they are actually closed\n\nFile v0.2.11:_meta.json\n\n{\n  \"ownerId\": \"kn7d88952ey3hbm158x7ejqs1d81zmyq\",\n  \"slug\": \"openclaw-self-improvement\",\n  \"version\": \"0.2.11\",\n  \"publishedAt\": 1777623237502\n}\n\nFile v0.2.11:references/decision-rules.md\n\n# Decision Rules for Self-Improvement\n\nUse this reference to decide whether a new issue should become:\n- a simple learning entry\n- an experiment with binary evals\n- a promoted operating rule\n\n---\n\n## Option 1 — Log only\n\nUse **log only** when:\n- the issue happened once\n- root cause is still unclear\n- there is not enough evidence yet to turn it into a rule\n- the lesson is useful, but not broadly reusable yet\n\nTypical output:\n- `learning`\n- `error`\n- `feature`\n\nExamples:\n- one-off API outage\n- first-time tool glitch with unclear cause\n- user preference that does not affect system-wide ops\n\n---\n\n## Option 2 — Run an experiment\n\nUse an **experiment** when:\n- the same failure happened 2+ times\n- a new guardrail/SOP/checklist/schema change is being proposed\n- you can define 3-5 binary evals\n- you want evidence that the change helped before promoting it broadly\n\nTypical output:\n- `experiment`\n- baseline + mutation + binary evals + keep/discard decision\n\nExamples:\n- Mission Control summaries repeatedly missing links/details\n- deploy closeout repeatedly confusing code-ready with live\n- repeated missing receipts or incomplete proof bundles\n- repeated front-end / operator-surface mismatch after backend fixes\n\n---\n\n## Option 3 — Promote immediately\n\nUse **promote immediately** when:\n- the rule is already obviously correct and low-risk\n- the issue is severe enough that waiting would be irresponsible\n- the required change is a principle or ownership rule, not an uncertain optimization\n- operator review already confirms the new rule should become standard\n\nTypical output:\n- promotion into `AGENTS.md`, `TOOLS.md`, `SOUL.md`, or `docs/ops/*.md`\n\nExamples:\n- deployment owner must be explicit\n- code-ready is not the same as live\n- missing receipt cannot be treated as delivered\n- summary without proof links is not operator-complete\n\n---\n\n## Promote after experiment\n\nUse **experiment first, then promote** when:\n- the rule sounds plausible but may add friction\n- you are not sure whether the added checklist/schema field actually reduces errors\n- the change could create process overhead without improving truth quality\n\nExamples:\n- adding new summary schema fields\n- adding new receipt requirements\n- adding extra verification steps to handoff or QA lanes\n\n---\n\n## Anti-patterns\n\nDo **not** run an experiment when:\n- there is no clear repeated failure\n- the evals would be vague or subjective\n- the issue is really just missing execution, not missing learning\n- the fix requires immediate owner action, not more analysis\n\nDo **not** promote immediately when:\n- the rule is still based on one anecdote\n- the change is likely to create bureaucracy without proof of benefit\n- the actual failure surface is still unclear\n\n---\n\n## Quick decision tree\n\n1. Did this happen only once?\n- Yes → log only\n- No → continue\n\n2. Is the new rule obviously necessary and low-risk?\n- Yes → promote immediately\n- No → continue\n\n3. Can you define 3-5 binary evals for the proposed change?\n- Yes → run an experiment\n- No → log only until the failure is clearer\n\n4. Did the experiment materially improve the failure pattern?\n- Yes → keep and consider promotion\n- No → discard or partial_keep\n\n---\n\n## One-line heuristic\n\n**Single incident = log. Repeated pattern = experiment. Clear principle/ownership rule = promote.**\n\nFile v0.2.11:references/eval-loop.md\n\n# Eval Loop for Self-Improvement\n\nUse this reference when a repeated failure should become a tested operational improvement instead of only a logged lesson.\n\n## Goal\n\nDo not only ask \"what did we learn?\"\nAlso ask:\n- what is the current baseline?\n- what exact guardrail or rule changed?\n- how will we measure whether it helped?\n- should we keep or discard the change?\n\n## Use this loop for\n- repeated Mission Control wording failures\n- missing receipts / missing proof chains\n- deploy closeout failures\n- stale operator-facing surfaces\n- repeated handoff mistakes between agents\n- recurring SOP/checklist changes\n\n## 1. Define the target\n\nState one concrete thing you want to improve.\n\nExamples:\n- Hunter summary should always include concrete links and details\n- ClawLite deploy closeout should never stop at code-ready status\n- Mission Control front-end should render source links from structured fields\n\n## 2. Write 3-5 binary evals\n\nEach eval must be yes/no.\n\nExamples for summary quality:\n- Does the summary include at least one artifact path or URL?\n- Does the summary include evidence links when external proof matters?\n- Does the summary include a detail block describing what actually changed?\n- Does the summary include the next handoff or recovery action?\n- Does the operator-facing surface actually render these fields?\n\nExamples for deploy closeout:\n- Is the deployed commit hash recorded?\n- Is a deployment ref/URL recorded?\n- Was the production page or sitemap actually verified?\n- Was a structured receipt written?\n- Is the final state classified with the correct deploy-state vocabulary?\n\n## 3. Capture baseline\n\nBefore changing the rule/SOP/skill/checklist:\n- record the current failure pattern\n- record which evals currently fail\n- treat this as the baseline state\n\n## 4. Change only one thing\n\nGood changes:\n- one wording rule\n- one new checklist item\n- one schema field\n- one render mapping\n- one validation step\n\nBad changes:\n- rewriting everything at once\n- adding five new rules at once\n- changing wording and schema and code together unless absolutely required\n\n## 5. Re-check and classify\n\nAfter the single change:\n- run the same evals again\n- note which checks improved\n- decide:\n  - KEEP\n  - DISCARD\n  - PARTIAL_KEEP\n\n## 6. Promotion rule\n\nOnly promote broadly reusable changes after they pass the eval loop or after operator review confirms the change materially reduced the failure.\n\n## Suggested experiment entry format\n\n```md\n## [EXP-YYYYMMDD-XXX] experiment\n\n**Logged**: ISO-8601 timestamp\n**Priority**: medium | high | critical\n**Status**: baseline | testing | keep | discard | partial_keep\n**Area**: workflow | tools | product | growth | security | infra | ops\n\n### Target\nWhat repeated problem is being improved\n\n### Baseline\nWhat was failing before the change\n\n### Mutation\nThe single change introduced\n\n### Binary Evals\n- [ ] Eval 1\n- [ ] Eval 2\n- [ ] Eval 3\n\n### Result\nWhat improved / did not improve\n\n### Keep or Discard\nkeep | discard | partial_keep\n\n### Metadata\n- Source: review | postmortem | user_feedback | qa\n- Related Files:\n- Tags:\n```\n\n## Important limit\n\nA logged experiment is not the same as a finished fix. If the production surface or operator-visible truth is still wrong, the experiment remains incomplete even if the local change looks promising.\n\nFile v0.2.11:references/examples.md\n\n# Practical Examples for Self-Improvement Eval Loops\n\nUse these examples when you want to turn a repeated operational mistake into a tested guardrail.\n\n---\n\n## Example 1 — Mission Control summary link-complete gate\n\n### Repeated failure\nMission Control agent summaries were updated at the JSON layer but still too thin for operator use. They lacked concrete links, detail blocks, and next handoff context. In some cases the front-end also failed to render the newly added structured fields.\n\n### Target\nMake Mission Control summaries decision-ready instead of status-only.\n\n### Baseline\nBefore the change:\n- summary text could say \"DELIVERED\" without concrete artifact links\n- evidence existed in underlying files but not in summary fields\n- operator-facing UI could still show \"No source links in current summary\"\n\n### Single mutation example\nAdd a summary contract requiring:\n- at least one artifact path or URL\n- evidence links when external proof matters\n- a detail block\n- next handoff / recovery action\n\n### Binary evals\n- [ ] Does the summary include at least one artifact path or URL?\n- [ ] Does the summary include evidence links when external proof matters?\n- [ ] Does the summary include a detail block explaining what changed?\n- [ ] Does the summary include the next handoff or recovery action?\n- [ ] Does the operator-facing surface actually render these fields?\n\n### Result classification guidance\n- **KEEP** if the summary and rendered surface both become decision-ready.\n- **DISCARD** if the JSON changed but the operator-facing surface still hides the proof.\n- **PARTIAL_KEEP** if the summary structure improved but the rendering layer still needs a separate fix.\n\n### Example command\n```bash\nnode scripts/log-experiment.mjs \\\n  \"Mission Control summaries should be link-complete and decision-ready\" \\\n  \"DELIVERED summaries lacked links/details and the UI still showed no source links\" \\\n  \"Added artifactLinks/evidenceLinks/details/nextHandoff contract to summary schema\" \\\n  \"summary has artifact path|summary has evidence links|summary has detail block|summary has next handoff|UI renders structured fields\" \\\n  \"Summary JSON improved, but front-end rendering still needed a separate mapping fix\" \\\n  \"partial_keep\"\n```\n\n---\n\n## Example 2 — ClawLite deploy closeout gate\n\n### Repeated failure\nCode changes were being treated as effectively live before production verification. Sitemap and page changes could be committed and locally built, while production remained stale or blocked.\n\n### Target\nEnsure web changes are never called live without production verification.\n\n### Baseline\nBefore the change:\n- code-ready and build-ready states were often described as if they were live\n- no explicit deploy owner existed\n- production sitemap verification was not consistently required\n\n### Single mutation example\nAdd a deploy closeout SOP with Peter as owner and require deploy-state vocabulary plus production verification.\n\n### Binary evals\n- [ ] Is the deployed commit hash recorded?\n- [ ] Is a deployment ref or URL recorded?\n- [ ] Was the production page or sitemap actually verified?\n- [ ] Was a structured deployment receipt written?\n- [ ] Is the final state classified as CODE_READY_NOT_DEPLOYED / DEPLOY_PENDING / DEPLOYED_NOT_VERIFIED / LIVE_VERIFIED / BLOCKED_DEPLOY instead of vague done/not-done wording?\n\n### Result classification guidance\n- **KEEP** if the team stops calling code-ready changes \"live\" without production proof.\n- **DISCARD** if the new SOP adds wording but nobody records deploy refs or production verification.\n- **PARTIAL_KEEP** if the receipt/state vocabulary lands but enforcement is still inconsistent.\n\n### Example command\n```bash\nnode scripts/log-experiment.mjs \\\n  \"ClawLite web changes should require production deploy closeout before being called live\" \\\n  \"Code/build truth was repeatedly confused with production truth\" \\\n  \"Added Peter-owned deploy SOP and deploy-state vocabulary\" \\\n  \"commit hash recorded|deployment ref recorded|production sitemap verified|deployment receipt written|state uses deploy vocabulary\" \\\n  \"Deploy blocker was surfaced truthfully as BLOCKED_DEPLOY instead of being misreported as live\" \\\n  \"keep\"\n```\n\n---\n\n## Rule of thumb\n\nIf the change improves only the hidden back-end layer but leaves the operator-facing truth surface stale, the experiment is not a full keep yet.\n\nFile v0.2.11:references/promotion-guide.md\n\n# Promotion Guide\n\nUse promotion only when a learning is broadly reusable.\n\n## Promote to AGENTS.md\nWhen the learning changes execution workflow.\nExamples:\n- deploy ownership rules\n- acceptance ownership rules\n- escalation timing\n\n## Promote to TOOLS.md\nWhen the learning is an environment/tool routing rule.\nExamples:\n- use Tavily before Brave\n- key locations in Keychain\n- browser session attach rules\n\n## Promote to SOUL.md\nWhen the learning is a behavior/principle rule.\nExamples:\n- do not let no-assignment closeout replace required deliverables\n- do not treat shallow checks as full acceptance\n\n## Promote to Obsidian\nWhen the learning should become reusable operator material, marketing proof, or an operations note outside transient chat.\n\nBy default, Obsidian-style exports go to the local safe fallback:\n- `.learnings/exports/obsidian/`\n\nIf you want a real vault destination, set `OBSIDIAN_LEARNINGS_DIR` explicitly before running the promotion script.\nAlways confirm the printed target path first, or use `--dry-run`.\n\nExample:\n- `node scripts/promote-learning.mjs obsidian \"Reusable learning\" --dry-run`\n- then rerun without `--dry-run` after confirming the path\n\nFile v0.2.11:references/schema.md\n\n# Learning Schema\n\n## Files\n- `.learnings/LEARNINGS.md`\n- `.learnings/ERRORS.md`\n- `.learnings/FEATURE_REQUESTS.md`\n\n## Learning entry\n```md\n## [LRN-YYYYMMDD-XXX] category\n\n**Logged**: ISO-8601 timestamp\n**Priority**: low | medium | high | critical\n**Status**: pending\n**Area**: workflow | tools | product | growth | security | infra\n\n### Summary\nOne-line learning\n\n### Details\nWhat happened and what is now understood\n\n### Suggested Action\nSpecific next action\n\n### Metadata\n- Source: user_feedback | error | review | postmortem\n- Related Files: path/to/file\n- Tags: tag1, tag2\n```\n\n## Error entry\n```md\n## [ERR-YYYYMMDD-XXX] name\n\n**Logged**: ISO-8601 timestamp\n**Priority**: high\n**Status**: pending\n**Area**: infra | product | growth | security | ops\n\n### Summary\nWhat failed\n\n### Error\nActual error or concise failure output\n\n### Suggested Fix\nLikely fix or next step\n\n### Metadata\n- Reproducible: yes | no | unknown\n- Related Files: path/to/file\n```\n\n## Feature request entry\n```md\n## [FEAT-YYYYMMDD-XXX] capability\n\n**Logged**: ISO-8601 timestamp\n**Priority**: medium\n**Status**: pending\n**Area**: product | ops | growth | security\n\n### Requested Capability\nWhat is missing\n\n### User Context\nWhy it matters\n\n### Suggested Implementation\nMinimal implementation direction\n```\n\n## Experiment entry\n```md\n## [EXP-YYYYMMDD-XXX] experiment\n\n**Logged**: ISO-8601 timestamp\n**Priority**: medium | high | critical\n**Status**: baseline | testing | keep | discard | partial_keep\n**Area**: workflow | tools | product | growth | security | infra | ops\n\n### Target\nWhat repeated problem is being improved\n\n### Baseline\nWhat was failing before the change\n\n### Mutation\nThe single change introduced\n\n### Binary Evals\n- [ ] Eval 1\n- [ ] Eval 2\n- [ ] Eval 3\n\n### Result\nWhat improved / did not improve\n\n### Keep or Discard\nkeep | discard | partial_keep\n```\n\nFile v0.2.11:releases/v0.2.0-clawhub-listing.md\n\n# ClawHub Listing Copy — OpenClaw Self-Improvement v0.2.0\n\n## Short description\nTurn repeated AI-agent mistakes into durable operational improvements with learning capture, binary eval loops, keep/discard decisions, and promotion into SOPs, checklists, and workflow rules.\n\n## Long description\n**OpenClaw Self-Improvement** is a skill for OpenClaw / ClawLite operators and multi-agent teams who want to stop repeating the same operational mistakes.\n\nIt helps agents and operators:\n- log learnings, errors, feature gaps, and experiments\n- run lightweight **binary eval loops** on repeated failures\n- test whether a new guardrail, SOP, checklist, or schema change actually helps\n- classify changes as **keep**, **partial_keep**, or **discard**\n- promote proven improvements into `AGENTS.md`, `TOOLS.md`, `SOUL.md`, and `docs/ops/*.md`\n\nThis makes it especially useful for:\n- self-improving AI workflows\n- Mission Control truth-state quality\n- deploy closeout verification\n- receipt / proof completeness\n- repeated failure prevention in multi-agent systems\n\n### Best keywords / search phrases\n- self-improving AI workflow\n- binary evals for agent operations\n- OpenClaw self-improvement skill\n- repeated failure prevention for AI agents\n- deploy closeout verification\n- Mission Control QA improvement\n- durable operational learning for multi-agent systems\n\n## What’s new in v0.2.0\n- Experiment mode for repeated failures\n- Binary eval loops\n- Keep / partial_keep / discard decisions\n- Practical examples for summary quality and deploy closeout\n- Experiment summary helper\n- Decision rules for log vs experiment vs promote\n- Daily routine integration with heartbeat and Karen QA\n\nFile v0.2.11:releases/v0.2.0-github-release.md\n\n# OpenClaw Self-Improvement v0.2.0\n\n**OpenClaw Self-Improvement v0.2.0** turns self-improvement from a vague idea into an operational loop.\n\nThis release is for OpenClaw / ClawLite operators and multi-agent teams that want to reduce repeated failures in a durable, inspectable way.\n\n## Highlights\n\n### New in v0.2.0\n- **Experiment mode** for repeated failures\n- **Binary eval loops** for testing new guardrails, SOPs, and workflow changes\n- **Keep / partial_keep / discard** decision model\n- **Practical examples** for:\n  - Mission Control summary quality\n  - deploy closeout / production verification\n- **Experiment summary helper** for surfacing unresolved follow-up debt\n- **Decision rules** for when to log, experiment, or promote\n- **Daily routine integration** with heartbeat, Karen QA, AGENTS, and self-improving ops policy\n\n## Why this release matters\n\nMany agent systems can log lessons.\nFar fewer can test whether a new rule, checklist, or guardrail actually reduced the repeated failure.\n\nv0.2.0 adds a lightweight eval-driven improvement loop:\n1. capture the repeated problem\n2. define a baseline\n3. change one thing at a time\n4. evaluate with binary checks\n5. classify the result as `keep`, `partial_keep`, or `discard`\n6. promote proven improvements into durable operating rules\n\nThis makes self-improvement more operational and less hand-wavy.\n\n## What problems this skill is built for\n\nOpenClaw Self-Improvement is especially useful when your agent workflows keep repeating problems like:\n- fake-complete states\n- missing receipts or weak proof bundles\n- back-end fixes that never reach operator-facing truth surfaces\n- code-ready or build-ready states being confused with production-ready states\n- new rules being added without evidence that they actually improve outcomes\n\n## Best-fit use cases\n\n- self-improving AI workflows\n- binary evals for agent operations\n- Mission Control truth-state improvement\n- deploy closeout verification\n- repeated failure prevention in OpenClaw or ClawLite-style systems\n- durable learning loops for multi-agent teams\n\n## Practical examples included\n\nThis release includes concrete examples for:\n- **Mission Control summary link-complete gates**\n- **ClawLite deploy closeout gates**\n\nThese examples show how to move from repeated failure → baseline → mutation → binary eval → keep/discard decision.\n\n## New files / capabilities\n\n### Scripts\n- `scripts/log-experiment.mjs`\n- `scripts/experiment-summary.mjs`\n\n### References\n- `references/eval-loop.md`\n- `references/examples.md`\n- `references/decision-rules.md`\n\n### Ops integration\n- Karen QA now treats `partial_keep` as active follow-up debt\n- heartbeat can check unresolved experiment follow-up items\n- daily routine + policy + AGENTS entry updated to make self-improvement part of normal operations\n\n## Why `partial_keep` matters\n\nOne of the most important additions in this release is the idea that a fix can be **directionally correct but still incomplete**.\n\nExample:\n- summary JSON becomes better\n- but the front-end still fails to render the new fields\n\nThat should not be treated as “done.”\nIt should be tracked as `partial_keep` until the paired fix lands.\n\n## Upgrade summary\n\nIf you already used earlier versions of this skill, v0.2.0 adds:\n- a stronger experiment layer\n- clearer decision rules\n- better examples\n- a runtime-friendly summary tool\n- tighter integration with OpenClaw-style QA and heartbeat workflows\n\n## Bottom line\n\n**OpenClaw Self-Improvement v0.2.0 is for teams that want self-improvement to mean more than writing down lessons.**\n\nIt helps you test whether a new workflow rule actually reduces repeated failure, keep what works, and keep unfinished follow-up debt visible until it is truly closed.\n\nFile v0.2.11:releases/v0.2.1-github-release.md\n\n# v0.2.1\n\n- improve README positioning for SEO / GEO\n- add clearer landing-page summary for GitHub and ClawHub readers\n\nFile v0.2.11:releases/v0.2.10-github-release.md\n\n# v0.2.10 — Harness observability and daily scorecards\n\nAdds OpenClaw harness observability helpers for classifying repeated runtime failures and generating daily agent delivery scorecards.\n\nChanges:\n- Added harness failure taxonomy for gateway, provider, session, skill, tool, platform, and approval blockers.\n- Added `scripts/analyze-openclaw-failures.mjs`.\n- Added `scripts/daily-agent-scorecard.mjs`.\n- Added bin aliases `openclaw-analyze-failures` and `openclaw-agent-scorecard`.\n- Updated SKILL.md so repeated agent delivery failures can become logged errors, experiments, or promoted guardrails.\n\nVerification:\n- `node --check scripts/harness-improvement-lib.mjs`\n- `node --check scripts/analyze-openclaw-failures.mjs`\n- `node --check scripts/daily-agent-scorecard.mjs`\n- `node scripts/daily-agent-scorecard.mjs --json`\n\nArchive v0.2.10: 26 files, 30499 bytes\n\nFiles: package.json (1049b), README.md (11841b), references/decision-rules.md (3334b), references/eval-loop.md (3305b), references/examples.md (4367b), references/promotion-guide.md (1175b), references/schema.md (1846b), releases/v0.2.0-clawhub-listing.md (1678b), releases/v0.2.0-github-release.md (3744b), releases/v0.2.1-github-release.md (119b), releases/v0.2.10-github-release.md (829b), releases/v0.2.2-github-release.md (119b), releases/v0.2.3-github-release.md (127b), releases/v0.2.4-github-release.md (147b), releases/v0.2.7-github-release.md (147b), releases/v0.2.8-github-release.md (221b), releases/v0.2.9-github-release.md (204b), scripts/analyze-openclaw-failures.mjs (2558b), scripts/daily-agent-scorecard.mjs (1909b), scripts/experiment-summary.mjs (1749b), scripts/harness-improvement-lib.mjs (8386b), scripts/log-experiment.mjs (1699b), scripts/log-learning.mjs (1683b), scripts/promote-learning.mjs (1263b), SKILL.md (7666b), _meta.json (145b)\n\nFile v0.2.10:SKILL.md\n\n---\nname: openclaw-self-improvement\ndescription: A reusable operator-guided workflow improvement skill for OpenClaw and ClawLite that turns repeated failures into logged learnings, binary eval loops, SOPs, checklists, and proof-based operational improvements.\nmetadata:\n  {\n    \"openclaw\":\n      {\n        \"requires\": { \"bins\": [\"node\"] },\n        \"writes\": [\".learnings/\", \"memory/harness-backlog-latest.md\", \"mission-control/data/delivery-receipts/agent-scorecard-YYYY-MM-DD.md\", \"AGENTS.md\", \"TOOLS.md\", \"SOUL.md\"],\n        \"env\": [\"WORKSPACE\", \"OBSIDIAN_LEARNINGS_DIR\"],\n        \"network\": false,\n        \"notes\": \"Local-file workflow only. Promotion writes should be reviewed and can be previewed with --dry-run.\"\n      }\n  }\n---\n\n# OpenClaw / ClawLite Self-Improvement\n\nUse this skill to turn mistakes, corrections, blockers, and better approaches into durable operating knowledge.\n\n## What problem this solves\nAI ops often repeat the same failures because mistakes stay in chat history instead of becoming system rules. This skill creates a lightweight improvement loop:\n- log failures and learnings\n- separate errors from feature requests\n- run small eval-driven experiments on repeated failures\n- classify harness/runtime failures instead of blaming vague “model issues”\n- generate daily agent scorecards from real evidence chains\n- promote important patterns into AGENTS.md / TOOLS.md / SOUL.md\n- write operator notes into Obsidian vault\n- support stricter acceptance via Karen / Mission Control\n\n## When to use\nUse this skill when the user asks:\n- \"make the agent improve itself\"\n- \"capture learnings\"\n- \"log mistakes so we do not repeat them\"\n- \"record blockers / corrections / feature gaps\"\n- \"build a self-improving OpenClaw workflow\"\n- \"operationalize lessons learned\"\n- \"test whether this new rule actually helps\"\n- \"run an eval loop on this workflow/skill/SOP\"\n- \"should we keep this new guardrail or discard it\"\n- \"why did the agents fail today\"\n- \"why is daily marketing not closing automatically\"\n- \"classify OpenClaw harness failures\"\n- \"generate agent delivery scorecard\"\n\n## Files this skill uses\n- `.learnings/LEARNINGS.md`\n- `.learnings/ERRORS.md`\n- `.learnings/FEATURE_REQUESTS.md`\n- `.learnings/EXPERIMENTS.md`\n- `memory/harness-backlog-latest.md`\n- `mission-control/data/delivery-receipts/agent-scorecard-YYYY-MM-DD.md`\n- Optional export under `.learnings/exports/obsidian/` by default, or `OBSIDIAN_LEARNINGS_DIR` if explicitly configured\n\n## Safety boundaries\n- Local-file workflow only, no network I/O\n- Promotion can append to `AGENTS.md`, `TOOLS.md`, or `SOUL.md`\n- Always review promotion targets first, or run `scripts/promote-learning.mjs ... --dry-run`\n- `OBSIDIAN_LEARNINGS_DIR` should only point at a path you intend to modify\n\n## Command examples\n```bash\nnode {baseDir}/scripts/log-learning.mjs learning \"Summary\" \"Details\" \"Suggested action\"\nnode {baseDir}/scripts/log-learning.mjs error \"Summary\" \"Error details\" \"Suggested fix\"\nnode {baseDir}/scripts/log-learning.mjs feature \"Capability name\" \"User context\" \"Suggested implementation\"\nnode {baseDir}/scripts/log-learning.mjs experiment \"Target problem\" \"Baseline failure\" \"Single mutation to test\"\nnode {baseDir}/scripts/log-experiment.mjs \"Target problem\" \"Baseline failure\" \"Single mutation\" \"eval1|eval2|eval3\" \"Result summary\" \"testing\"\nnode {baseDir}/scripts/promote-learning.mjs workflow \"Rule text\"\nnode {baseDir}/scripts/analyze-openclaw-failures.mjs --output /Users/m1/.openclaw/workspace/memory/harness-backlog-latest.md\nnode {baseDir}/scripts/daily-agent-scorecard.mjs --output /Users/m1/.openclaw/workspace/mission-control/data/delivery-receipts/agent-scorecard-$(date +%F).md\n```\n\n## Categories\n### learning\nUse for:\n- user corrections\n- better recurring workflows\n- tool gotchas\n- operational lessons\n\n### error\nUse for:\n- command failures\n- integration failures\n- runtime blockers\n- broken release / deploy behavior\n\n### feature\nUse for:\n- missing capability requests\n- operator workflow gaps\n- recurring requests that deserve a build item\n\n### experiment\nUse for:\n- repeated failures that need a tested guardrail\n- checklist/SOP/schema changes that should be validated before broad promotion\n- keep/discard decisions on new operating rules\n- binary eval loops for skills, workflows, receipts, summaries, or deploy closeout rules\n\n### harness\nUse for:\n- gateway, channel, provider, tool, session, or platform failures\n- repeated \"agent did not respond / did not finish / forgot identity\" incidents\n- daily workflow failures where Mission Control says one thing but proof chains say another\n- scorecards that compare agent delivery against real receipts, URLs, and closeout evidence\n\nDefault failure taxonomy:\n- `NetworkPolicyBlocked` - provider/tool blocked by local or external network policy\n- `GatewayUnavailable` - gateway process, port, websocket, or reachability failure\n- `SessionContextRot` - stale session, stale skill snapshot, identity drift, or outdated config context\n- `SkillMissing` - expected skill absent from installed path or session snapshot\n- `ToolInvalidArguments` - malformed tool/edit call or bad argument shape\n- `ProviderError` - provider/model/API failure not caused by network policy\n- `ExternalPlatformBlocked` - X/LinkedIn/Facebook/Feishu/etc. platform/API/login/visibility blocker\n- `HumanApprovalRequired` - real approval boundary for external, destructive, production, money, or ambiguous action\n\nHarness workflow:\n1. Scan logs and receipts with `scripts/analyze-openclaw-failures.mjs`.\n2. Generate same-day agent scorecard with `scripts/daily-agent-scorecard.mjs`.\n3. Convert repeated classes into an `error`, `experiment`, or promoted rule.\n4. Do not call a workflow closed until the scorecard has proof links or explicit blocker evidence.\n\n## Promotion targets\n- `AGENTS.md` → workflow / delegation / execution rules\n- `TOOLS.md` → tool gotchas, secrets locations, environment routing rules\n- `SOUL.md` → behavior / communication / non-negotiable principles\n- Obsidian vault → reusable operator log and content proof asset\n\n## Karen / Mission Control compatibility\nThis skill is designed to work with stricter ops governance:\n- Karen can reference learnings when repeated failures happen\n- Mission Control can treat promoted learnings as new operating rules\n- recurring blockers can be elevated from chat into tracked operational knowledge\n- experiments can test whether a new summary contract, receipt rule, or deploy closeout guardrail actually reduced the failure pattern\n\n## Eval loop rule\nWhen a repeated failure is turning into a new rule/SOP/checklist, do not only log it.\nAlso:\n1. define 3-5 binary evals\n2. record the baseline failure state\n3. change one thing at a time\n4. re-check the same evals\n5. classify the change as keep / discard / partial_keep\n\nUse `{baseDir}/references/eval-loop.md` for the experiment format and examples.\n\n## Output goal\nA good use of this skill should produce one of:\n- a durable learning entry\n- a durable error entry\n- a durable feature request entry\n- a durable experiment entry with binary evals\n- a promoted rule in AGENTS.md / TOOLS.md / SOUL.md\n- an Obsidian vault operations note\n\n## Important limits\n- Logging is not the same as fixing.\n- Do not treat a learning entry as closure for a broken deliverable.\n- Use this skill to reduce repeated mistakes, not to excuse them.\n\n## References\n- `{baseDir}/references/schema.md`\n- `{baseDir}/references/promotion-guide.md`\n- `{baseDir}/references/eval-loop.md`\n- `{baseDir}/references/examples.md`\n- `{baseDir}/references/decision-rules.md`\n\n- `{baseDir}/references/eval-loop.md`\n- `{baseDir}/references/examples.md`\n\nFile v0.2.10:README.md\n\n# OpenClaw Self-Improvement\n\n**OpenClaw Self-Improvement** is a reusable agent skill for turning repeated AI-agent mistakes into durable operational improvements, measurable guardrails, and inspectable workflow upgrades.\n\n## Landing-page summary\n\nMost AI agents do not really improve. They repeat mistakes, hide partial failures behind optimistic language, and leave lessons trapped in chat history.\n\nOpenClaw Self-Improvement gives you a practical operating loop for **self-improving AI agents**:\n- capture repeated failures\n- test one guardrail at a time\n- verify whether it actually reduces failure\n- promote proven fixes into SOPs, checklists, policies, and workflow rules\n\nIf you want **AI agents that get more reliable over time**, **multi-agent workflows that stop repeating the same mistakes**, or **proof-based QA for agent operations**, this skill is built for that exact use case.\n\n## Multilingual summary\n\n**中文：** 这是一个面向 AI agent 自我改进的实战型 skill，用来减少重复犯错、建立 guardrails、验证修复是否真的有效，并把经验沉淀成 SOP、检查清单和可复用规则。\n\n**日本語：** これは自己改善する AI エージェント向けの実践的な skill です。繰り返し発生する失敗を減らし、ガードレールを検証し、改善を SOP・チェックリスト・再利用可能な運用ルールへ昇格させます。\n\n**한국어：** 이 스킬은 스스로 개선하는 AI 에이전트를 위한 운영형 skill입니다. 반복 실수를 줄이고, 가드레일이 실제로 효과가 있는지 검증하며, 개선 사항을 SOP·체크리스트·재사용 가능한 규칙으로 승격합니다.\n\n**Español：** Esta skill está diseñada para agentes de IA que deben mejorar con el tiempo. Ayuda a reducir errores repetidos, validar guardrails operativos y convertir mejoras en SOP, checklists y reglas reutilizables.\n\nIf you want a practical way to deploy OpenClaw with cheaper tokens, BYOK flexibility, and operator control, see **[ClawLite](https://clawlite.ai)**.\n\n## TL;DR\n\nIf you are looking for a practical system for **self-improving AI agents**, **AI workflow optimization**, **multi-agent failure prevention**, **binary eval loops**, or **agent operations QA**, this skill is designed for that exact job.\n\nIt helps OpenClaw / ClawLite operators and agent teams:\n- log recurring failures\n- separate one-off errors from reusable lessons\n- run lightweight **binary eval loops** on new guardrails\n- classify changes as **keep**, **partial_keep**, or **discard**\n- promote proven fixes into SOPs, checklists, workflow rules, and operating policy\n\nIf you care about reducing fake-complete states, tightening QA truth, improving deploy closeout, and making agent learning inspectable, this skill is built for that job.\n\n---\n\n## Why this skill exists\n\nMany AI systems say they \"learn,\" but most only store lessons in chat history or loose notes.\n\nThat is not enough.\n\nOperationally, repeated failures tend to come back in the same forms:\n- delivery gets described as complete before proof exists\n- receipts are missing or too thin\n- back-end fixes never reach the operator-facing surface\n- code-ready states get confused with production-ready states\n- teams add new rules without checking whether those rules actually reduce failure\n\nOpenClaw Self-Improvement gives you a lightweight operating loop for fixing that.\n\n---\n\n## What problem it solves\n\nThis skill helps with:\n- **self-improving AI workflows**\n- **AI operations learning loops**\n- **binary evals for agent guardrails**\n- **Mission Control truth-state improvement**\n- **deploy closeout verification**\n- **receipt / proof completeness**\n- **repeated failure prevention in multi-agent systems**\n- **AI agent reliability engineering**\n- **agent QA systems for OpenClaw, ClawLite, and similar stacks**\n\nIt is especially useful in OpenClaw-style environments where multiple agents, tools, SOPs, and truth surfaces interact.\n\n## Who this is for\n\nThis skill is useful for:\n- OpenClaw operators\n- ClawLite growth / marketing / QA lanes\n- AI agent builders who need durable postmortems instead of vague “reflection”\n- teams running multi-agent workflows with receipts, truth surfaces, and closeout gates\n- anyone trying to reduce repeated AI-agent mistakes in production-like operations\n\n---\n\n## What’s new in v0.2.0\n\n### New capabilities\n- **Experiment mode** for repeated failures\n- **Binary eval loops** for testing whether a new guardrail or SOP actually helps\n- **Keep / partial_keep / discard** decision model\n- **Practical examples** for:\n  - Mission Control summary quality\n  - deploy closeout / production verification\n- **Experiment summary helper** for surfacing unresolved follow-up debt\n- **Decision rules** for when to log, experiment, or promote\n- **Daily routine integration** with heartbeat, Karen QA, AGENTS, and ops policy\n\n### Why it matters\nThis release makes self-improvement more operational.\n\nInstead of stopping at “lesson learned,” you can now:\n1. capture the repeated problem\n2. define a baseline\n3. test one change at a time\n4. evaluate with binary checks\n5. keep, discard, or mark `partial_keep`\n\nThat makes the improvement loop much more auditable and much less hand-wavy.\n\n---\n\n## Core workflow\n\nOpenClaw Self-Improvement now supports a practical loop:\n\n1. **Capture** a learning, error, feature request, or experiment\n2. **Store** it in structured local files\n3. **Experiment** when a repeated failure needs a tested guardrail\n4. **Evaluate** the change with binary checks\n5. **Promote** proven fixes into durable rules, SOPs, or policies\n6. **Track follow-up debt** when a fix is only partial\n\n---\n\n## Best use cases\n\nUse this skill when you want to:\n- capture lessons so agents stop repeating the same mistake\n- log recurring operational errors\n- track feature gaps revealed by repeated work\n- test whether a new workflow rule really improves outcomes\n- run a lightweight eval loop on a skill, SOP, checklist, schema, or handoff rule\n- decide whether a new guardrail should be kept, discarded, or promoted\n- build a self-improving OpenClaw or ClawLite operating loop\n\nTypical targets include:\n- Mission Control summary quality\n- deploy closeout gates\n- receipt requirements\n- QA wording rules\n- truth-surface rendering checks\n- handoff contracts between agents\n\n---\n\n## Files it manages\n\n- `.learnings/LEARNINGS.md`\n- `.learnings/ERRORS.md`\n- `.learnings/FEATURE_REQUESTS.md`\n- `.learnings/EXPERIMENTS.md`\n- Optional Obsidian export directory via `OBSIDIAN_LEARNINGS_DIR`\n- Default local export fallback: `.learnings/exports/obsidian/`\n- Safe by default: no hard-coded external vault path, and `scripts/promote-learning.mjs` prints the resolved write target before writing\n\n---\n\n## Install\n\n```bash\nnpm install\n```\n\n---\n\n## Usage\n\n### Log a learning\n\n```bash\nnode scripts/log-learning.mjs learning \"Summary\" \"Details\" \"Suggested action\"\n```\n\n### Log an error\n\n```bash\nnode scripts/log-learning.mjs error \"Summary\" \"Error details\" \"Suggested fix\"\n```\n\n### Log a feature request\n\n```bash\nnode scripts/log-learning.mjs feature \"Capability name\" \"User context\" \"Suggested implementation\"\n```\n\n### Log a tested experiment\n\n```bash\nnode scripts/log-experiment.mjs \"Target problem\" \"Baseline failure\" \"Single mutation\" \"eval1|eval2|eval3\" \"Result summary\" \"testing\"\n```\n\n### Promote a rule\n\n```bash\nnode scripts/promote-learning.mjs workflow \"Rule text\"\nnode scripts/promote-learning.mjs obsidian \"Reusable learning\" --dry-run\n```\n\n### Summarize experiment outcomes\n\n```bash\nnode scripts/experiment-summary.mjs\n```\n\n---\n\n## Decision model\n\nThis skill uses three levels of action:\n\n### 1. Log only\nUse when:\n- the issue happened once\n- root cause is still unclear\n- there is not enough evidence yet to make a rule\n\n### 2. Experiment\nUse when:\n- the issue repeated 2+ times\n- a new guardrail / SOP / checklist / schema change is being proposed\n- you can define 3–5 binary evals\n\n### 3. Promote\nUse when:\n- the rule is clearly right and low-risk\n- the issue is severe enough that waiting would be irresponsible\n- the rule is about ownership, truth, or a non-negotiable operating principle\n\n---\n\n## Practical examples\n\nThe skill now includes concrete examples for:\n- **Mission Control summary link-complete gates**\n- **ClawLite deploy closeout gates**\n\nThese examples show how to:\n- define the repeated failure\n- capture the baseline\n- propose one mutation\n- evaluate with binary checks\n- classify the outcome as `keep`, `partial_keep`, or `discard`\n\n---\n\n## Promotion targets\n\nPromote proven improvements into:\n- `AGENTS.md` — workflow / delegation / execution rules\n- `TOOLS.md` — tool gotchas and environment routing rules\n- `SOUL.md` — behavior / communication / non-negotiable principles\n- `docs/ops/*.md` — SOPs, policy, and operating contracts\n- Obsidian vault — reusable operator notes and operational memory\n\n---\n\n## Important limits\n\n- Logging is **not** the same as fixing.\n- A learning entry does **not** close a broken deliverable.\n- A back-end-only improvement is not complete if the visible operator-facing surface is still stale.\n- `partial_keep` should be treated as **active follow-up debt**, not as closure.\n\n---\n\n## Repository contents\n\n- `SKILL.md` — agent-facing routing and usage guidance\n- `scripts/log-learning.mjs` — append a learning / error / feature request / experiment\n- `scripts/log-experiment.mjs` — append a structured experiment with binary evals\n- `scripts/experiment-summary.mjs` — summarize keep / partial_keep / discard outcomes and flag follow-up debt\n- `scripts/promote-learning.mjs` — promote a lesson into durable operating rules with explicit path echo and optional `--dry-run`\n- `scripts/analyze-openclaw-failures.mjs` — classify OpenClaw gateway, provider, session, tool, and platform failures\n- `scripts/daily-agent-scorecard.mjs` — generate a same-day agent delivery scorecard from Mission Control and session evidence\n- `references/schema.md` — data structure guidance\n- `references/promotion-guide.md` — what to promote and where\n- `references/eval-loop.md` — how to run lightweight binary-eval improvement loops\n- `references/examples.md` — practical examples for summary gates and deploy closeout gates\n- `references/decision-rules.md` — when to log only, run an experiment, or promote immediately\n\n---\n\n## SEO / GEO positioning\n\nThis skill is intentionally legible to both search engines and AI answer engines because it is built around concrete, reusable operational concepts rather than vague “AI reflection” language.\n\n### Primary search themes\n- self-improving AI workflows\n- self-improving agent systems\n- AI agent reliability engineering\n- binary eval loops for agent guardrails\n- repeated failure prevention in multi-agent systems\n- deploy closeout verification\n- proof-based QA for AI operations\n- operational learning loops for OpenClaw / ClawLite style stacks\n\n### Why that matters\nThese phrases map to real operator intent:\n- “how do I stop my AI agents from repeating mistakes?”\n- “how do I build a self-improving agent workflow?”\n- “how do I test whether a new guardrail actually works?”\n- “how do I make AI operations auditable?”\n\nThat makes the skill easier to explain, cite, retrieve, and reuse than generic memory or reflection systems.\n\n---\n\n## Bottom line\n\nIf you want OpenClaw to improve over time instead of repeating the same mistakes across sessions, this repo gives you:\n- an operational memory loop\n- a lightweight eval loop for testing whether a new guardrail actually helps\n- a durable promotion path from mistake -> experiment -> policy -> reusable system asset\n- a decision framework for when to log, experiment, or promote\n- a way to keep unresolved partial improvements visible until they are actually closed\n\nFile v0.2.10:_meta.json\n\n{\n  \"ownerId\": \"kn7d88952ey3hbm158x7ejqs1d81zmyq\",\n  \"slug\": \"openclaw-self-improvement\",\n  \"version\": \"0.2.10\",\n  \"publishedAt\": 1777622827960\n}\n\nFile v0.2.10:references/decision-rules.md\n\n# Decision Rules for Self-Improvement\n\nUse this reference to decide whether a new issue should become:\n- a simple learning entry\n- an experiment with binary evals\n- a promoted operating rule\n\n---\n\n## Option 1 — Log only\n\nUse **log only** when:\n- the issue happened once\n- root cause is still unclear\n- there is not enough evidence yet to turn it into a rule\n- the lesson is useful, but not broadly reusable yet\n\nTypical output:\n- `learning`\n- `error`\n- `feature`\n\nExamples:\n- one-off API outage\n- first-time tool glitch with unclear cause\n- user preference that does not affect system-wide ops\n\n---\n\n## Option 2 — Run an experiment\n\nUse an **experiment** when:\n- the same failure happened 2+ times\n- a new guardrail/SOP/checklist/schema change is being proposed\n- you can define 3-5 binary evals\n- you want evidence that the change helped before promoting it broadly\n\nTypical output:\n- `experiment`\n- baseline + mutation + binary evals + keep/discard decision\n\nExamples:\n- Mission Control summaries repeatedly missing links/details\n- deploy closeout repeatedly confusing code-ready with live\n- repeated missing receipts or incomplete proof bundles\n- repeated front-end / operator-surface mismatch after backend fixes\n\n---\n\n## Option 3 — Promote immediately\n\nUse **promote immediately** when:\n- the rule is already obviously correct and low-risk\n- the issue is severe enough that waiting would be irresponsible\n- the required change is a principle or ownership rule, not an uncertain optimization\n- operator review already confirms the new rule should become standard\n\nTypical output:\n- promotion into `AGENTS.md`, `TOOLS.md`, `SOUL.md`, or `docs/ops/*.md`\n\nExamples:\n- deployment owner must be explicit\n- code-ready is not the same as live\n- missing receipt cannot be treated as delivered\n- summary without proof links is not operator-complete\n\n---\n\n## Promote after experiment\n\nUse **experiment first, then promote** when:\n- the rule sounds plausible but may add friction\n- you are not sure whether the added checklist/schema field actually reduces errors\n- the change could create process overhead without improving truth quality\n\nExamples:\n- adding new summary schema fields\n- adding new receipt requirements\n- adding extra verification steps to handoff or QA lanes\n\n---\n\n## Anti-patterns\n\nDo **not** run an experiment when:\n- there is no clear repeated failure\n- the evals would be vague or subjective\n- the issue is really just missing execution, not missing learning\n- the fix requires immediate owner action, not more analysis\n\nDo **not** promote immediately when:\n- the rule is still based on one anecdote\n- the change is likely to create bureaucracy without proof of benefit\n- the actual failure surface is still unclear\n\n---\n\n## Quick decision tree\n\n1. Did this happen only once?\n- Yes → log only\n- No → continue\n\n2. Is the new rule obviously necessary and low-risk?\n- Yes → promote immediately\n- No → continue\n\n3. Can you define 3-5 binary evals for the proposed change?\n- Yes → run an experiment\n- No → log only until the failure is clearer\n\n4. Did the experiment materially improve the failure pattern?\n- Yes → keep and consider promotion\n- No → discard or partial_keep\n\n---\n\n## One-line heuristic\n\n**Single incident = log. Repeated pattern = experiment. Clear principle/ownership rule = promote.**\n\nFile v0.2.10:references/eval-loop.md\n\n# Eval Loop for Self-Improvement\n\nUse this reference when a repeated failure should become a tested operational improvement instead of only a logged lesson.\n\n## Goal\n\nDo not only ask \"what did we learn?\"\nAlso ask:\n- what is the current baseline?\n- what exact guardrail or rule changed?\n- how will we measure whether it helped?\n- should we keep or discard the change?\n\n## Use this loop for\n- repeated Mission Control wording failures\n- missing receipts / missing proof chains\n- deploy closeout failures\n- stale operator-facing surfaces\n- repeated handoff mistakes between agents\n- recurring SOP/checklist changes\n\n## 1. Define the target\n\nState one concrete thing you want to improve.\n\nExamples:\n- Hunter summary should always include concrete links and details\n- ClawLite deploy closeout should never stop at code-ready status\n- Mission Control front-end should render source links from structured fields\n\n## 2. Write 3-5 binary evals\n\nEach eval must be yes/no.\n\nExamples for summary quality:\n- Does the summary include at least one artifact path or URL?\n- Does the summary include evidence links when external proof matters?\n- Does the summary include a detail block describing what actually changed?\n- Does the summary include the next handoff or recovery action?\n- Does the operator-facing surface actually render these fields?\n\nExamples for deploy closeout:\n- Is the deployed commit hash recorded?\n- Is a deployment ref/URL recorded?\n- Was the production page or sitemap actually verified?\n- Was a structured receipt written?\n- Is the final state classified with the correct deploy-state vocabulary?\n\n## 3. Capture baseline\n\nBefore changing the rule/SOP/skill/checklist:\n- record the current failure pattern\n- record which evals currently fail\n- treat this as the baseline state\n\n## 4. Change only one thing\n\nGood changes:\n- one wording rule\n- one new checklist item\n- one schema field\n- one render mapping\n- one validation step\n\nBad changes:\n- rewriting everything at once\n- adding five new rules at once\n- changing wording and schema and code together unless absolutely required\n\n## 5. Re-check and classify\n\nAfter the single change:\n- run the same evals again\n- note which checks improved\n- decide:\n  - KEEP\n  - DISCARD\n  - PARTIAL_KEEP\n\n## 6. Promotion rule\n\nOnly promote broadly reusable changes after they pass the eval loop or after operator review confirms the change materially reduced the failure.\n\n## Suggested experiment entry format\n\n```md\n## [EXP-YYYYMMDD-XXX] experiment\n\n**Logged**: ISO-8601 timestamp\n**Priority**: medium | high | critical\n**Status**: baseline | testing | keep | discard | partial_keep\n**Area**: workflow | tools | product | growth | security | infra | ops\n\n### Target\nWhat repeated problem is being improved\n\n### Baseline\nWhat was failing before the change\n\n### Mutation\nThe single change introduced\n\n### Binary Evals\n- [ ] Eval 1\n- [ ] Eval 2\n- [ ] Eval 3\n\n### Result\nWhat improved / did not improve\n\n### Keep or Discard\nkeep | discard | partial_keep\n\n### Metadata\n- Source: review | postmortem | user_feedback | qa\n- Related Files:\n- Tags:\n```\n\n## Important limit\n\nA logged experiment is not the same as a finished fix. If the production surface or operator-visible truth is still wrong, the experiment remains incomplete even if the local change looks promising.\n\nFile v0.2.10:references/examples.md\n\n# Practical Examples for Self-Improvement Eval Loops\n\nUse these examples when you want to turn a repeated operational mistake into a tested guardrail.\n\n---\n\n## Example 1 — Mission Control summary link-complete gate\n\n### Repeated failure\nMission Control agent summaries were updated at the JSON layer but still too thin for operator use. They lacked concrete links, detail blocks, and next handoff context. In some cases the front-end also failed to render the newly added structured fields.\n\n### Target\nMake Mission Control summaries decision-ready instead of status-only.\n\n### Baseline\nBefore the change:\n- summary text could say \"DELIVERED\" without concrete artifact links\n- evidence existed in underlying files but not in summary fields\n- operator-facing UI could still show \"No source links in current summary\"\n\n### Single mutation example\nAdd a summary contract requiring:\n- at least one artifact path or URL\n- evidence links when external proof matters\n- a detail block\n- next handoff / recovery action\n\n### Binary evals\n- [ ] Does the summary include at least one artifact path or URL?\n- [ ] Does the summary include evidence links when external proof matters?\n- [ ] Does the summary include a detail block explaining what changed?\n- [ ] Does the summary include the next handoff or recovery action?\n- [ ] Does the operator-facing surface actually render these fields?\n\n### Result classification guidance\n- **KEEP** if the summary and rendered surface both become decision-ready.\n- **DISCARD** if the JSON changed but the operator-facing surface still hides the proof.\n- **PARTIAL_KEEP** if the summary structure improved but the rendering layer still needs a separate fix.\n\n### Example command\n```bash\nnode scripts/log-experiment.mjs \\\n  \"Mission Control summaries should be link-complete and decision-ready\" \\\n  \"DELIVERED summaries lacked links/details and the UI still showed no source links\" \\\n  \"Added artifactLinks/evidenceLinks/details/nextHandoff contract to summary schema\" \\\n  \"summary has artifact path|summary has evidence links|summary has detail block|summary has next handoff|UI renders structured fields\" \\\n  \"Summary JSON improved, but front-end rendering still needed a separate mapping fix\" \\\n  \"partial_keep\"\n```\n\n---\n\n## Example 2 — ClawLite deploy closeout gate\n\n### Repeated failure\nCode changes were being treated as effectively live before production verification. Sitemap and page changes could be committed and locally built, while production remained stale or blocked.\n\n### Target\nEnsure web changes are never called live without production verification.\n\n### Baseline\nBefore the change:\n- code-ready and build-ready states were often described as if they were live\n- no explicit deploy owner existed\n- production sitemap verification was not consistently required\n\n### Single mutation example\nAdd a deploy closeout SOP with Peter as owner and require deploy-state vocabulary plus production verification.\n\n### Binary evals\n- [ ] Is the deployed commit hash recorded?\n- [ ] Is a deployment ref or URL recorded?\n- [ ] Was the production page or sitemap actually verified?\n- [ ] Was a structured deployment receipt written?\n- [ ] Is the final state classified as CODE_READY_NOT_DEPLOYED / DEPLOY_PENDING / DEPLOYED_NOT_VERIFIED / LIVE_VERIFIED / BLOCKED_DEPLOY instead of vague done/not-done wording?\n\n### Result classification guidance\n- **KEEP** if the team stops calling code-ready changes \"live\" without production proof.\n- **DISCARD** if the new SOP adds wording but nobody records deploy refs or production verification.\n- **PARTIAL_KEEP** if the receipt/state vocabulary lands but enforcement is still inconsistent.\n\n### Example command\n```bash\nnode scripts/log-experiment.mjs \\\n  \"ClawLite web changes should require production deploy closeout before being called live\" \\\n  \"Code/build truth was repeatedly confused with production truth\" \\\n  \"Added Peter-owned deploy SOP and deploy-state vocabulary\" \\\n  \"commit hash recorded|deployment ref recorded|production sitemap verified|deployment receipt written|state uses deploy vocabulary\" \\\n  \"Deploy blocker was surfaced truthfully as BLOCKED_DEPLOY instead of being misreported as live\" \\\n  \"keep\"\n```\n\n---\n\n## Rule of thumb\n\nIf the change improves only the hidden back-end layer but leaves the operator-facing truth surface stale, the experiment is not a full keep yet.\n\nFile v0.2.10:references/promotion-guide.md\n\n# Promotion Guide\n\nUse promotion only when a learning is broadly reusable.\n\n## Promote to AGENTS.md\nWhen the learning changes execution workflow.\nExamples:\n- deploy ownership rules\n- acceptance ownership rules\n- escalation timing\n\n## Promote to TOOLS.md\nWhen the learning is an environment/tool routing rule.\nExamples:\n- use Tavily before Brave\n- key locations in Keychain\n- browser session attach rules\n\n## Promote to SOUL.md\nWhen the learning is a behavior/principle rule.\nExamples:\n- do not let no-assignment closeout replace required deliverables\n- do not treat shallow checks as full acceptance\n\n## Promote to Obsidian\nWhen the learning should become reusable operator material, marketing proof, or an operations note outside transient chat.\n\nBy default, Obsidian-style exports go to the local safe fallback:\n- `.learnings/exports/obsidian/`\n\nIf you want a real vault destination, set `OBSIDIAN_LEARNINGS_DIR` explicitly before running the promotion script.\nAlways confirm the printed target path first, or use `--dry-run`.\n\nExample:\n- `node scripts/promote-learning.mjs obsidian \"Reusable learning\" --dry-run`\n- then rerun without `--dry-run` after confirming the path\n\nFile v0.2.10:references/schema.md\n\n# Learning Schema\n\n## Files\n- `.learnings/LEARNINGS.md`\n- `.learnings/ERRORS.md`\n- `.learnings/FEATURE_REQUESTS.md`\n\n## Learning entry\n```md\n## [LRN-YYYYMMDD-XXX] category\n\n**Logged**: ISO-8601 timestamp\n**Priority**: low | medium | high | critical\n**Status**: pending\n**Area**: workflow | tools | product | growth | security | infra\n\n### Summary\nOne-line learning\n\n### Details\nWhat happened and what is now understood\n\n### Suggested Action\nSpecific next action\n\n### Metadata\n- Source: user_feedback | error | review | postmortem\n- Related Files: path/to/file\n- Tags: tag1, tag2\n```\n\n## Error entry\n```md\n## [ERR-YYYYMMDD-XXX] name\n\n**Logged**: ISO-8601 timestamp\n**Priority**: high\n**Status**: pending\n**Area**: infra | product | growth | security | ops\n\n### Summary\nWhat failed\n\n### Error\nActual error or concise failure output\n\n### Suggested Fix\nLikely fix or next step\n\n### Metadata\n- Reproducible: yes | no | unknown\n- Related Files: path/to/file\n```\n\n## Feature request entry\n```md\n## [FEAT-YYYYMMDD-XXX] capability\n\n**Logged**: ISO-8601 timestamp\n**Priority**: medium\n**Status**: pending\n**Area**: product | ops | growth | security\n\n### Requested Capability\nWhat is missing\n\n### User Context\nWhy it matters\n\n### Suggested Implementation\nMinimal implementation direction\n```\n\n## Experiment entry\n```md\n## [EXP-YYYYMMDD-XXX] experiment\n\n**Logged**: ISO-8601 timestamp\n**Priority**: medium | high | critical\n**Status**: baseline | testing | keep | discard | partial_keep\n**Area**: workflow | tools | product | growth | security | infra | ops\n\n### Target\nWhat repeated problem is being improved\n\n### Baseline\nWhat was failing before the change\n\n### Mutation\nThe single change introduced\n\n### Binary Evals\n- [ ] Eval 1\n- [ ] Eval 2\n- [ ] Eval 3\n\n### Result\nWhat improved / did not improve\n\n### Keep or Discard\nkeep | discard | partial_keep\n```\n\nFile v0.2.10:releases/v0.2.0-clawhub-listing.md\n\n# ClawHub Listing Copy — OpenClaw Self-Improvement v0.2.0\n\n## Short description\nTurn repeated AI-agent mistakes into durable operational improvements with learning capture, binary eval loops, keep/discard decisions, and promotion into SOPs, checklists, and workflow rules.\n\n## Long description\n**OpenClaw Self-Improvement** is a skill for OpenClaw / ClawLite operators and multi-agent teams who want to stop repeating the same operational mistakes.\n\nIt helps agents and operators:\n- log learnings, errors, feature gaps, and experiments\n- run lightweight **binary eval loops** on repeated failures\n- test whether a new guardrail, SOP, checklist, or schema change actually helps\n- classify changes as **keep**, **partial_keep**, or **discard**\n- promote proven improvements into `AGENTS.md`, `TOOLS.md`, `SOUL.md`, and `docs/ops/*.md`\n\nThis makes it especially useful for:\n- self-improving AI workflows\n- Mission Control truth-state quality\n- deploy closeout verification\n- receipt / proof completeness\n- repeated failure prevention in multi-agent systems\n\n### Best keywords / search phrases\n- self-improving AI workflow\n- binary evals for agent operations\n- OpenClaw self-improvement skill\n- repeated failure prevention for AI agents\n- deploy closeout verification\n- Mission Control QA improvement\n- durable operational learning for multi-agent systems\n\n## What’s new in v0.2.0\n- Experiment mode for repeated failures\n- Binary eval loops\n- Keep / partial_keep / discard decisions\n- Practical examples for summary quality and deploy closeout\n- Experiment summary helper\n- Decision rules for log vs experiment vs promote\n- Daily routine integration with heartbeat and Karen QA\n\nFile v0.2.10:releases/v0.2.0-github-release.md\n\n# OpenClaw Self-Improvement v0.2.0\n\n**OpenClaw Self-Improvement v0.2.0** turns self-improvement from a vague idea into an operational loop.\n\nThis release is for OpenClaw / ClawLite operators and multi-agent teams that want to reduce repeated failures in a durable, inspectable way.\n\n## Highlights\n\n### New in v0.2.0\n- **Experiment mode** for repeated failures\n- **Binary eval loops** for testing new guardrails, SOPs, and workflow changes\n- **Keep / partial_keep / discard** decision model\n- **Practical examples** for:\n  - Mission Control summary quality\n  - deploy closeout / production verification\n- **Experiment summary helper** for surfacing unresolved follow-up debt\n- **Decision rules** for when to log, experiment, or promote\n- **Daily routine integration** with heartbeat, Karen QA, AGENTS, and self-improving ops policy\n\n## Why this release matters\n\nMany agent systems can log lessons.\nFar fewer can test whether a new rule, checklist, or guardrail actually reduced the repeated failure.\n\nv0.2.0 adds a lightweight eval-driven improvement loop:\n1. capture the repeated problem\n2. define a baseline\n3. change one thing at a time\n4. evaluate with binary checks\n5. classify the result as `keep`, `partial_keep`, or `discard`\n6. promote proven improvements into durable operating rules\n\nThis makes self-improvement more operational and less hand-wavy.\n\n## What problems this skill is built for\n\nOpenClaw Self-Improvement is especially useful when your agent workflows keep repeating problems like:\n- fake-complete states\n- missing receipts or weak proof bundles\n- back-end fixes that never reach operator-facing truth surfaces\n- code-ready or build-ready states being confused with production-ready states\n- new rules being added without evidence that they actually improve outcomes\n\n## Best-fit use cases\n\n- self-improving AI workflows\n- binary evals for agent operations\n- Mission Control truth-state improvement\n- deploy closeout verification\n- repeated failure prevention in OpenClaw or ClawLite-style systems\n- durable learning loops for multi-agent teams\n\n## Practical examples included\n\nThis release includes concrete examples for:\n- **Mission Control summary link-complete gates**\n- **ClawLite deploy closeout gates**\n\nThese examples show how to move from repeated failure → baseline → mutation → binary eval → keep/discard decision.\n\n## New files / capabilities\n\n### Scripts\n- `scripts/log-experiment.mjs`\n- `scripts/experiment-summary.mjs`\n\n### References\n- `references/eval-loop.md`\n- `references/examples.md`\n- `references/decision-rules.md`\n\n### Ops integration\n- Karen QA now treats `partial_keep` as active follow-up debt\n- heartbeat can check unresolved experiment follow-up items\n- daily routine + policy + AGENTS entry updated to make self-improvement part of normal operations\n\n## Why `partial_keep` matters\n\nOne of the most important additions in this release is the idea that a fix can be **directionally correct but still incomplete**.\n\nExample:\n- summary JSON becomes better\n- but the front-end still fails to render the new fields\n\nThat should not be treated as “done.”\nIt should be tracked as `partial_keep` until the paired fix lands.\n\n## Upgrade summary\n\nIf you already used earlier versions of this skill, v0.2.0 adds:\n- a stronger experiment layer\n- clearer decision rules\n- better examples\n- a runtime-friendly summary tool\n- tighter integration with OpenClaw-style QA and heartbeat workflows\n\n## Bottom line\n\n**OpenClaw Self-Improvement v0.2.0 is for teams that want self-improvement to mean more than writing down lessons.**\n\nIt helps you test whether a new workflow rule actually reduces repeated failure, keep what works, and keep unfinished follow-up debt visible until it is truly closed.\n\nFile v0.2.10:releases/v0.2.1-github-release.md\n\n# v0.2.1\n\n- improve README positioning for SEO / GEO\n- add clearer landing-page summary for GitHub and ClawHub readers\n\nFile v0.2.10:releases/v0.2.10-github-release.md\n\n# v0.2.10 — Harness observability and daily scorecards\n\nAdds OpenClaw harness observability helpers for classifying repeated runtime failures and generating daily agent delivery scorecards.\n\nChanges:\n- Added harness failure taxonomy for gateway, provider, session, skill, tool, platform, and approval blockers.\n- Added `scripts/analyze-openclaw-failures.mjs`.\n- Added `scripts/daily-agent-scorecard.mjs`.\n- Added bin aliases `openclaw-analyze-failures` and `openclaw-agent-scorecard`.\n- Updated SKILL.md so repeated agent delivery failures can become logged errors, experiments, or promoted guardrails.\n\nVerification:\n- `node --check scripts/harness-improvement-lib.mjs`\n- `node --check scripts/analyze-openclaw-failures.mjs`\n- `node --check scripts/daily-agent-scorecard.mjs`\n- `node scripts/daily-agent-scorecard.mjs --json`\n\nArchive v0.2.9: 22 files, 23467 bytes\n\nFiles: package.json (901b), README.md (11770b), references/decision-rules.md (3334b), references/eval-loop.md (3305b), references/examples.md (4367b), references/promotion-guide.md (1175b), references/schema.md (1846b), releases/v0.2.0-clawhub-listing.md (1678b), releases/v0.2.0-github-release.md (3744b), releases/v0.2.1-github-release.md (119b), releases/v0.2.2-github-release.md (119b), releases/v0.2.3-github-release.md (127b), releases/v0.2.4-github-release.md (147b), releases/v0.2.7-github-release.md (147b), releases/v0.2.8-github-release.md (221b), releases/v0.2.9-github-release.md (204b), scripts/experiment-summary.mjs (1749b), scripts/log-experiment.mjs (1699b), scripts/log-learning.mjs (1683b), scripts/promote-learning.mjs (1263b), SKILL.md (5416b), _meta.json (144b)\n\nFile v0.2.9:SKILL.md\n\n---\nname: openclaw-self-improvement\ndescription: A reusable operator-guided workflow improvement skill for OpenClaw and ClawLite that turns repeated failures into logged learnings, binary eval loops, SOPs, checklists, and proof-based operational improvements.\nmetadata:\n  {\n    \"openclaw\":\n      {\n        \"requires\": { \"bins\": [\"node\"] },\n        \"writes\": [\".learnings/\", \"AGENTS.md\", \"TOOLS.md\", \"SOUL.md\"],\n        \"env\": [\"WORKSPACE\", \"OBSIDIAN_LEARNINGS_DIR\"],\n        \"network\": false,\n        \"notes\": \"Local-file workflow only. Promotion writes should be reviewed and can be previewed with --dry-run.\"\n      }\n  }\n---\n\n# OpenClaw / ClawLite Self-Improvement\n\nUse this skill to turn mistakes, corrections, blockers, and better approaches into durable operating knowledge.\n\n## What problem this solves\nAI ops often repeat the same failures because mistakes stay in chat history instead of becoming system rules. This skill creates a lightweight improvement loop:\n- log failures and learnings\n- separate errors from feature requests\n- run small eval-driven experiments on repeated failures\n- promote important patterns into AGENTS.md / TOOLS.md / SOUL.md\n- write operator notes into Obsidian vault\n- support stricter acceptance via Karen / Mission Control\n\n## When to use\nUse this skill when the user asks:\n- \"make the agent improve itself\"\n- \"capture learnings\"\n- \"log mistakes so we do not repeat them\"\n- \"record blockers / corrections / feature gaps\"\n- \"build a self-improving OpenClaw workflow\"\n- \"operationalize lessons learned\"\n- \"test whether this new rule actually helps\"\n- \"run an eval loop on this workflow/skill/SOP\"\n- \"should we keep this new guardrail or discard it\"\n\n## Files this skill uses\n- `.learnings/LEARNINGS.md`\n- `.learnings/ERRORS.md`\n- `.learnings/FEATURE_REQUESTS.md`\n- `.learnings/EXPERIMENTS.md`\n- Optional export under `.learnings/exports/obsidian/` by default, or `OBSIDIAN_LEARNINGS_DIR` if explicitly configured\n\n## Safety boundaries\n- Local-file workflow only, no network I/O\n- Promotion can append to `AGENTS.md`, `TOOLS.md`, or `SOUL.md`\n- Always review promotion targets first, or run `scripts/promote-learning.mjs ... --dry-run`\n- `OBSIDIAN_LEARNINGS_DIR` should only point at a path you intend to modify\n\n## Command examples\n```bash\nnode {baseDir}/scripts/log-learning.mjs learning \"Summary\" \"Details\" \"Suggested action\"\nnode {baseDir}/scripts/log-learning.mjs error \"Summary\" \"Error details\" \"Suggested fix\"\nnode {baseDir}/scripts/log-learning.mjs feature \"Capability name\" \"User context\" \"Suggested implementation\"\nnode {baseDir}/scripts/log-learning.mjs experiment \"Target problem\" \"Baseline failure\" \"Single mutation to test\"\nnode {baseDir}/scripts/log-experiment.mjs \"Target problem\" \"Baseline failure\" \"Single mutation\" \"eval1|eval2|eval3\" \"Result summary\" \"testing\"\nnode {baseDir}/scripts/promote-learning.mjs workflow \"Rule text\"\n```\n\n## Categories\n### learning\nUse for:\n- user corrections\n- better recurring workflows\n- tool gotchas\n- operational lessons\n\n### error\nUse for:\n- command failures\n- integration failures\n- runtime blockers\n- broken release / deploy behavior\n\n### feature\nUse for:\n- missing capability requests\n- operator workflow gaps\n- recurring requests that deserve a build item\n\n### experiment\nUse for:\n- repeated failures that need a tested guardrail\n- checklist/SOP/schema changes that should be validated before broad promotion\n- keep/discard decisions on new operating rules\n- binary eval loops for skills, workflows, receipts, summaries, or deploy closeout rules\n\n## Promotion targets\n- `AGENTS.md` → workflow / delegation / execution rules\n- `TOOLS.md` → tool gotchas, secrets locations, environment routing rules\n- `SOUL.md` → behavior / communication / non-negotiable principles\n- Obsidian vault → reusable operator log and content proof asset\n\n## Karen / Mission Control compatibility\nThis skill is designed to work with stricter ops governance:\n- Karen can reference learnings when repeated failures happen\n- Mission Control can treat promoted learnings as new operating rules\n- recurring blockers can be elevated from chat into tracked operational knowledge\n- experiments can test whether a new summary contract, receipt rule, or deploy closeout guardrail actually reduced the failure pattern\n\n## Eval loop rule\nWhen a repeated failure is turning into a new rule/SOP/checklist, do not only log it.\nAlso:\n1. define 3-5 binary evals\n2. record the baseline failure state\n3. change one thing at a time\n4. re-check the same evals\n5. classify the change as keep / discard / partial_keep\n\nUse `{baseDir}/references/eval-loop.md` for the experiment format and examples.\n\n## Output goal\nA good use of this skill should produce one of:\n- a durable learning entry\n- a durable error entry\n- a durable feature request entry\n- a durable experiment entry with binary evals\n- a promoted rule in AGENTS.md / TOOLS.md / SOUL.md\n- an Obsidian vault operations note\n\n## Important limits\n- Logging is not the same as fixing.\n- Do not treat a learning entry as closure for a broken deliverable.\n- Use this skill to reduce repeated mistakes, not to excuse them.\n\n## References\n- `{baseDir}/references/schema.md`\n- `{baseDir}/references/promotion-guide.md`\n- `{baseDir}/references/eval-loop.md`\n- `{baseDir}/references/examples.md`\n- `{baseDir}/references/decision-rules.md`\n\n- `{baseDir}/references/eval-loop.md`\n- `{baseDir}/references/examples.md`\n\nFile v0.2.9:README.md\n\n# OpenClaw Self-Improvement\n\n**OpenClaw Self-Improvement** is a reusable agent skill for turning repeated AI-agent mistakes into durable operational improvements, measurable guardrails, and inspectable workflow upgrades.\n\n## Landing-page summary\n\nMost AI agents do not really improve. They repeat mistakes, hide partial failures behind optimistic language, and leave lessons trapped in chat history.\n\nOpenClaw Self-Improvement gives you a practical operating loop for **self-improving AI agents**:\n- capture repeated failures\n- test one guardrail at a time\n- verify whether it actually reduces failure\n- promote proven fixes into SOPs, checklists, policies, and workflow rules\n\nIf you want **AI agents that get more reliable over time**, **multi-agent workflows that stop repeating the same mistakes**, or **proof-based QA for agent operations**, this skill is built for that exact use case.\n\n## Multilingual summary\n\n**中文：** 这是一个面向 AI agent 自我改进的实战型 skill，用来减少重复犯错、建立 guardrails、验证修复是否真的有效，并把经验沉淀成 SOP、检查清单和可复用规则。\n\n**日本語：** これは自己改善する AI エージェント向けの実践的な skill です。繰り返し発生する失敗を減らし、ガードレールを検証し、改善を SOP・チェックリスト・再利用可能な運用ルールへ昇格させます。\n\n**한국어：** 이 스킬은 스스로 개선하는 AI 에이전트를 위한 운영형 skill입니다. 반복 실수를 줄이고, 가드레일이 실제로 효과가 있는지 검증하며, 개선 사항을 SOP·체크리스트·재사용 가능한 규칙으로 승격합니다.\n\n**Español：** Esta skill está diseñada para agentes de IA que deben mejorar con el tiempo. Ayuda a reducir errores repetidos, validar guardrails operativos y convertir mejoras en SOP, checklists y reglas reutilizables.\n\nIf you want a practical way to deploy OpenClaw with cheaper tokens, BYOK flexibility, and operator control, see **[ClawLite](https://clawlite.ai)**.\n\n## TL;DR\n\nIf you are looking for a practical system for **self-improving AI agents**, **AI workflow optimization**, **multi-agent failure prevention**, **binary eval loops**, or **agent operations QA**, this skill is designed for that exact job.\n\nIt helps OpenClaw / ClawLite operators and agent teams:\n- log recurring failures\n- separate one-off errors from reusable lessons\n- run lightweight **binary eval loops** on new guardrails\n- classify changes as **keep**, **partial_keep**, or **discard**\n- promote proven fixes into SOPs, checklists, workflow rules, and operating policy\n\nIf you care about reducing fake-complete states, tightening QA truth, improving deploy closeout, and making agent learning inspectable, this skill is built for that job.\n\n---\n\n## Why this skill exists\n\nMany AI systems say they \"learn,\" but most only store lessons in chat history or loose notes.\n\nThat is not enough.\n\nOperationally, repeated failures tend to come back in the same forms:\n- delivery gets described as complete before proof exists\n- receipts are missing or too thin\n- back-end fixes never reach the operator-facing surface\n- code-ready states get confused with production-ready states\n- teams add new rules without checking whether those rules actually reduce failure\n\nOpenClaw Self-Improvement gives you a lightweight operating loop for fixing that.\n\n---\n\n## What problem it solves\n\nThis skill helps with:\n- **self-improving AI workflows**\n- **AI operations learning loops**\n- **binary evals for agent guardrails**\n- **Mission Control truth-state improvement**\n- **deploy closeout verification**\n- **receipt / proof completeness**\n- **repeated failure prevention in multi-agent systems**\n- **AI agent reliability engineering**\n- **agent QA systems for OpenClaw, ClawLite, and similar stacks**\n\nIt is especially useful in OpenClaw-style environments where multiple agents, tools, SOPs, and truth surfaces interact.\n\n## Who this is for\n\nThis skill is useful for:\n- OpenClaw operators\n- ClawLite growth / marketing / QA lanes\n- AI agent builders who need durable postmortems instead of vague “reflection”\n- teams running multi-agent workflows with receipts, truth surfaces, and closeout gates\n- anyone trying to reduce repeated AI-agent mistakes in production-like operations\n\n---\n\n## What’s new in v0.2.0\n\n### New capabilities\n- **Experiment mode** for repeated failures\n- **Binary eval loops** for testing whether a new guardrail or SOP actually helps\n- **Keep / partial_keep / discard** decision model\n- **Practical examples** for:\n  - Mission Control summary quality\n  - deploy closeout / production verification\n- **Experiment summary helper** for surfacing unresolved follow-up debt\n- **Decision rules** for when to log, experiment, or promote\n- **Daily routine integration** with heartbeat, Karen QA, AGENTS, and ops policy\n\n### Why it matters\nThis release makes self-improvement more operational.\n\nInstead of stopping at “lesson learned,” you can now:\n1. capture the repeated problem\n2. define a baseline\n3. test one change at a time\n4. evaluate with binary checks\n5. keep, discard, or mark `partial_keep`\n\nThat makes the improvement loop much more auditable and much less hand-wavy.\n\n---\n\n## Core workflow\n\nOpenClaw Self-Improvement now supports a practical loop:\n\n1. **Capture** a learning, error, feature request, or experiment\n2. **Store** it in structured local files\n3. **Experiment** when a repeated failure needs a tested guardrail\n4. **Evaluate** the change with binary checks\n5. **Promote** proven fixes into durable rules, SOPs, or policies\n6. **Track follow-up debt** when a fix is only partial\n\n---\n\n## Best use cases\n\nUse this skill when you want to:\n- capture lessons so agents stop repeating the same mistake\n- log recurring operational errors\n- track feature gaps revealed by repeated work\n- test whether a new workflow rule really improves outcomes\n- run a lightweight eval loop on a skill, SOP, checklist, schema, or handoff rule\n- decide whether a new guardrail should be kept, discarded, or promoted\n- build a self-improving OpenClaw or ClawLite operating loop\n\nTypical targets include:\n- Mission Control summary quality\n- deploy closeout gates\n- receipt requirements\n- QA wording rules\n- truth-surface rendering checks\n- handoff contracts between agents\n\n---\n\n## Files it manages\n\n- `.learnings/LEARNINGS.md`\n- `.learnings/ERRORS.md`\n- `.learnings/FEATURE_REQUESTS.md`\n- `.learnings/EXPERIMENTS.md`\n- Optional Obsidian export directory via `OBSIDIAN_LEARNINGS_DIR`\n- Default local export fallback: `.learnings/exports/obsidian/`\n- Safe by default: no hard-coded external vault path, and `scripts/promote-learning.mjs` prints the resolved write target before writing\n\n---\n\n## Install\n\n```bash\nnpm install\n```\n\n---\n\n## Usage\n\n### Log a learning\n\n```bash\nnode scripts/log-learning.mjs learning \"Summary\" \"Details\" \"Suggested action\"\n```\n\n### Log an error\n\n```bash\nnode scripts/log-learning.mjs error \"Summary\" \"Error details\" \"Suggested fix\"\n```\n\n### Log a feature request\n\n```bash\nnode scripts/log-learning.mjs feature \"Capability name\" \"User context\" \"Suggested implementation\"\n```\n\n### Log a tested experiment\n\n```bash\nnode scripts/log-experiment.mjs \"Target problem\" \"Baseline failure\" \"Single mutation\" \"eval1|eval2|eval3\" \"Result summary\" \"testing\"\n```\n\n### Promote a rule\n\n```bash\nnode scripts/promote-learning.mjs workflow \"Rule text\"\nnode scripts/promote-learning.mjs obsidian \"Reusable learning\" --dry-run\n```\n\n### Summarize experiment outcomes\n\n```bash\nnode scripts/experiment-summary.mjs\n```\n\n---\n\n## Decision model\n\nThis skill uses three levels of action:\n\n### 1. Log only\nUse when:\n- the issue happened once\n- root cause is still unclear\n- there is not enough evidence yet to make a rule\n\n### 2. Experiment\nUse when:\n- the issue repeated 2+ times\n- a new guardrail / SOP / checklist / schema change is being proposed\n- you can define 3–5 binary evals\n\n### 3. Promote\nUse when:\n- the rule is clearly right and low-risk\n- the issue is severe enough that waiting would be irresponsible\n- the rule is about ownership, truth, or a non-negotiable operating principle\n\n---\n\n## Practical examples\n\nThe skill now includes concrete examples for:\n- **Mission Control summary link-complete gates**\n- **ClawLite deploy closeout gates**\n\nThese examples show how to:\n- define the repeated failure\n- capture the baseline\n- propose one mutation\n- evaluate with binary checks\n- classify the outcome as `keep`, `partial_keep`, or `discard`\n\n---\n\n## Promotion targets\n\nPromote proven improvements into:\n- `AGENTS.md` — workflow / delegation / execution rules\n- `TOOLS.md` — tool gotchas and environment routing rules\n- `SOUL.md` — behavior / communication / non-negotiable principles\n- `docs/ops/*.md` — SOPs, policy, and operating contracts\n- Obsidian vault — reusable operator notes and operational memory\n\n---\n\n## Important limits\n\n- Logging is **not** the same as fixing.\n- A learning entry does **not** close a broken deliverable.\n- A back-end-only improvement is not complete if the visible operator-facing surface is still stale.\n- `partial_keep` should be treated as **active follow-up debt**, not as closure.\n\n---\n\n## Repository contents\n\n- `SKILL.md` — agent-facing routing and usage guidance\n- `scripts/log-learning.mjs` — append a learning / error / feature request / experiment\n- `scripts/log-experiment.mjs` — append a structured experiment with binary evals\n- `scripts/experiment-summary.mjs` — summarize keep / partial_keep / discard outcomes and flag follow-up debt\n- `scripts/promote-learning.mjs` — promote a lesson into durable operating rules with explicit path echo and optional `--dry-run`\n- `references/schema.md` — data structure guidance\n- `references/promotion-guide.md` — what to promote and where\n- `references/eval-loop.md` — how to run lightweight binary-eval improvement loops\n- `references/examples.md` — practical examples for summary gates and deploy closeout gates\n- `references/decision-rules.md` — when to log only, run an experiment, or promote immediately\n\n---\n\n## SEO / GEO positioning\n\nThis skill is intentionally legible to both search engines and AI answer engines because it is built around concrete, reusable operational concepts rather than vague “AI reflection” language.\n\n### Primary search themes\n- self-improving AI workflows\n- self-improving agent systems\n- AI agent reliability engineering\n- binary eval loops for agent guardrails\n- repeated failure prevention in multi-agent systems\n- deploy closeout verification\n- proof-based QA for AI operations\n- operational learning loops for OpenClaw / ClawLite style stacks\n\n### Why that matters\nThese phrases map to real operator intent:\n- “how do I stop my AI agents from repeating mistakes?”\n- “how do I build a self-improving agent workflow?”\n- “how do I test whether a new guardrail actually works?”\n- “how do I make AI operations auditable?”\n\nThat makes the skill easier to explain, cite, retrieve, and reuse than generic memory or reflection systems.\n\n---\n\n## Bottom line\n\nIf you want OpenClaw to improve over time instead of repeating the same mistakes across sessions, this repo gives you:\n- an operational memory loop\n- a lightweight eval loop for testing whether a new guardrail actually helps\n- a durable promotion path from mistake -> experiment -> policy -> reusable system asset\n- a decision framework for when to log, experiment, or promote\n- a way to keep unresolved partial improvements visible until they are actually closed\new guardrail actually helps\n- a decision framework for when to log, experiment, or promote\n- a way to keep unresolved partial improvements visible until they are actually closed\n\nFile v0.2.9:_meta.json\n\n{\n  \"ownerId\": \"kn7d88952ey3hbm158x7ejqs1d81zmyq\",\n  \"slug\": \"openclaw-self-improvement\",\n  \"version\": \"0.2.9\",\n  \"publishedAt\": 1776237074329\n}\n\nFile v0.2.9:references/decision-rules.md\n\n# Decision Rules for Self-Improvement\n\nUse this reference to decide whether a new issue should become:\n- a simple learning entry\n- an experiment with binary evals\n- a promoted operating rule\n\n---\n\n## Option 1 — Log only\n\nUse **log only** when:\n- the issue happened once\n- root cause is still unclear\n- there is not enough evidence yet to turn it into a rule\n- the lesson is useful, but not broadly reusable yet\n\nTypical output:\n- `learning`\n- `error`\n- `feature`\n\nExamples:\n- one-off API outage\n- first-time tool glitch with unclear cause\n- user preference that does not affect system-wide ops\n\n---\n\n## Option 2 — Run an experiment\n\nUse an **experiment** when:\n- the same failure happened 2+ times\n- a new guardrail/SOP/checklist/schema change is being proposed\n- you can define 3-5 binary evals\n- you want evidence that the change helped before promoting it broadly\n\nTypical output:\n- `experiment`\n- baseline + mutation + binary evals + keep/discard decision\n\nExamples:\n- Mission Control summaries repeatedly missing links/details\n- deploy closeout repeatedly confusing code-ready with live\n- repeated missing receipts or incomplete proof bundles\n- repeated front-end / operator-surface mismatch after backend fixes\n\n---\n\n## Option 3 — Promote immediately\n\nUse **promote immediately** when:\n- the rule is already obviously correct and low-risk\n- the issue is severe enough that waiting would be irresponsible\n- the required change is a principle or ownership rule, not an uncertain optimization\n- operator review already confirms the new rule should become standard\n\nTypical output:\n- promotion into `AGENTS.md`, `TOOLS.md`, `SOUL.md`, or `docs/ops/*.md`\n\nExamples:\n- deployment owner must be explicit\n- code-ready is not the same as live\n- missing receipt cannot be treated as delivered\n- summary without proof links is not operator-complete\n\n---\n\n## Promote after experiment\n\nUse **experiment first, then promote** when:\n- the rule sounds plausible but may add friction\n- you are not sure whether the added checklist/schema field actually reduces errors\n- the change could create process overhead without improving truth quality\n\nExamples:\n- adding new summary schema fields\n- adding new receipt requirements\n- adding extra verification steps to handoff or QA lanes\n\n---\n\n## Anti-patterns\n\nDo **not** run an experiment when:\n- there is no clear repeated failure\n- the evals would be vague or subjective\n- the issue is really just missing execution, not missing learning\n- the fix requires immediate owner action, not more analysis\n\nDo **not** promote immediately when:\n- the rule is still based on one anecdote\n- the change is likely to create bureaucracy without proof of benefit\n- the actual failure surface is still unclear\n\n---\n\n## Quick decision tree\n\n1. Did this happen only once?\n- Yes → log only\n- No → continue\n\n2. Is the new rule obviously necessary and low-risk?\n- Yes → promote immediately\n- No → continue\n\n3. Can you define 3-5 binary evals for the proposed change?\n- Yes → run an experiment\n- No → log only until the failure is clearer\n\n4. Did the experiment materially improve the failure pattern?\n- Yes → keep and consider promotion\n- No → discard or partial_keep\n\n---\n\n## One-line heuristic\n\n**Single incident = log. Repeated pattern = experiment. Clear principle/ownership rule = promote.**\n\nFile v0.2.9:references/eval-loop.md\n\n# Eval Loop for Self-Improvement\n\nUse this reference when a repeated failure should become a tested operational improvement instead of only a logged lesson.\n\n## Goal\n\nDo not only ask \"what did we learn?\"\nAlso ask:\n- what is the current baseline?\n- what exact guardrail or rule changed?\n- how will we measure whether it helped?\n- should we keep or discard the change?\n\n## Use this loop for\n- repeated Mission Control wording failures\n- missing receipts / missing proof chains\n- deploy closeout failures\n- stale operator-facing surfaces\n- repeated handoff mistakes between agents\n- recurring SOP/checklist changes\n\n## 1. Define the target\n\nState one concrete thing you want to improve.\n\nExamples:\n- Hunter summary should always include concrete links and details\n- ClawLite deploy closeout should never stop at code-ready status\n- Mission Control front-end should render source links from structured fields\n\n## 2. Write 3-5 binary evals\n\nEach eval must be yes/no.\n\nExamples for summary quality:\n- Does the summary include at least one artifact path or URL?\n- Does the summary include evidence links when external proof matters?\n- Does the summary include a detail block describing what actually changed?\n- Does the summary include the next handoff or recovery action?\n- Does the operator-facing surface actually render these fields?\n\nExamples for deploy closeout:\n- Is the deployed commit hash recorded?\n- Is a deployment ref/URL recorded?\n- Was the production page or sitemap actually verified?\n- Was a structured receipt written?\n- Is the final state classified with the correct deploy-state vocabulary?\n\n## 3. Capture baseline\n\nBefore changing the rule/SOP/skill/checklist:\n- record the current failure pattern\n- record which evals currently fail\n- treat this as the baseline state\n\n## 4. Change only one thing\n\nGood changes:\n- one wording rule\n- one new checklist item\n- one schema field\n- one render mapping\n- one validation step\n\nBad changes:\n- rewriting everything at once\n- adding five new rules at once\n- changing wording and schema and code together unless absolutely required\n\n## 5. Re-check and classify\n\nAfter the single change:\n- run the same evals again\n- note which checks improved\n- decide:\n  - KEEP\n  - DISCARD\n  - PARTIAL_KEEP\n\n## 6. Promotion rule\n\nOnly promote broadly reusable changes after they pass the eval loop or after operator review confirms the change materially reduced the failure.\n\n## Suggested experiment entry format\n\n```md\n## [EXP-YYYYMMDD-XXX] experiment\n\n**Logged**: ISO-8601 timestamp\n**Priority**: medium | high | critical\n**Status**: baseline | testing | keep | discard | partial_keep\n**Area**: workflow | tools | product | growth | security | infra | ops\n\n### Target\nWhat repeated problem is being improved\n\n### Baseline\nWhat was failing before the change\n\n### Mutation\nThe single change introduced\n\n### Binary Evals\n- [ ] Eval 1\n- [ ] Eval 2\n- [ ] Eval 3\n\n### Result\nWhat improved / did not improve\n\n### Keep or Discard\nkeep | discard | partial_keep\n\n### Metadata\n- Source: review | postmortem | user_feedback | qa\n- Related Files:\n- Tags:\n```\n\n## Important limit\n\nA logged experiment is not the same as a finished fix. If the production surface or operator-visible truth is still wrong, the experiment remains incomplete even if the local change looks promising.\n\nFile v0.2.9:references/examples.md\n\n# Practical Examples for Self-Improvement Eval Loops\n\nUse these examples when you want to turn a repeated operational mistake into a tested guardrail.\n\n---\n\n## Example 1 — Mission Control summary link-complete gate\n\n### Repeated failure\nMission Control agent summaries were updated at the JSON layer but still too thin for operator use. They lacked concrete links, detail blocks, and next handoff context. In some cases the front-end also failed to render the newly added structured fields.\n\n### Target\nMake Mission Control summaries decision-ready instead of status-only.\n\n### Baseline\nBefore the change:\n- summary text could say \"DELIVERED\" without concrete artifact links\n- evidence existed in underlying files but not in summary fields\n- operator-facing UI could still show \"No source links in current summary\"\n\n### Single mutation example\nAdd a summary contract requiring:\n- at least one artifact path or URL\n- evidence links when external proof matters\n- a detail block\n- next handoff / recovery action\n\n### Binary evals\n- [ ] Does the summary include at least one artifact path or URL?\n- [ ] Does the summary include evidence links when external proof matters?\n- [ ] Does the summary include a detail block explaining what changed?\n- [ ] Does the summary include the next handoff or recovery action?\n- [ ] Does the operator-facing surface actually render these fields?\n\n### Result classification guidance\n- **KEEP** if the summary and rendered surface both become decision-ready.\n- **DISCARD** if the JSON changed but the operator-facing surface still hides the proof.\n- **PARTIAL_KEEP** if the summary structure improved but the rendering layer still needs a separate fix.\n\n### Example command\n```bash\nnode scripts/log-experiment.mjs \\\n  \"Mission Control summaries should be link-complete and decision-ready\" \\\n  \"DELIVERED summaries lacked links/details and the UI still showed no source links\" \\\n  \"Added artifactLinks/evidenceLinks/details/nextHandoff contract to summary schema\" \\\n  \"summary has artifact path|summary has evidence links|summary has detail block|summary has next handoff|UI renders structured fields\" \\\n  \"Summary JSON improved, but front-end rendering still needed a separate mapping fix\" \\\n  \"partial_keep\"\n```\n\n---\n\n## Example 2 — ClawLite deploy closeout gate\n\n### Repeated failure\nCode changes were being treated as effectively live before production verification. Sitemap and page changes could be committed and locally built, while production remained stale or blocked.\n\n### Target\nEnsure web changes are never called live without production verification.\n\n### Baseline\nBefore the change:\n- code-ready and build-ready states were often described as if they were live\n- no explicit deploy owner existed\n- production sitemap verification was not consistently required\n\n### Single mutation example\nAdd a deploy closeout SOP with Peter as owner and require deploy-state vocabulary plus production verification.\n\n### Binary evals\n- [ ] Is the deployed commit hash recorded?\n- [ ] Is a deployment ref or URL recorded?\n- [ ] Was the production page or sitemap actually verified?\n- [ ] Was a structured deployment receipt written?\n- [ ] Is the final state classified as CODE_READY_NOT_DEPLOYED / DEPLOY_PENDING / DEPLOYED_NOT_VERIFIED / LIVE_VERIFIED / BLOCKED_DEPLOY instead of vague done/not-done wording?\n\n### Result classification guidance\n- **KEEP** if the team stops calling code-ready changes \"live\" without production proof.\n- **DISCARD** if the new SOP adds wording but nobody records deploy refs or production verification.\n- **PARTIAL_KEEP** if the receipt/state vocabulary lands but enforcement is still inconsistent.\n\n### Example command\n```bash\nnode scripts/log-experiment.mjs \\\n  \"ClawLite web changes should require production deploy closeout before being called live\" \\\n  \"Code/build truth was repeatedly confused with production truth\" \\\n  \"Added Peter-owned deploy SOP and deploy-state vocabulary\" \\\n  \"commit hash recorded|deployment ref recorded|production sitemap verified|deployment receipt written|state uses deploy vocabulary\" \\\n  \"Deploy blocker was surfaced truthfully as BLOCKED_DEPLOY instead of being misreported as live\" \\\n  \"keep\"\n```\n\n---\n\n## Rule of thumb\n\nIf the change improves only the hidden back-end layer but leaves the operator-facing truth surface stale, the experiment is not a full keep yet.\n\nFile v0.2.9:references/promotion-guide.md\n\n# Promotion Guide\n\nUse promotion only when a learning is broadly reusable.\n\n## Promote to AGENTS.md\nWhen the learning changes execution workflow.\nExamples:\n- deploy ownership rules\n- acceptance ownership rules\n- escalation timing\n\n## Promote to TOOLS.md\nWhen the learning is an environment/tool routing rule.\nExamples:\n- use Tavily before Brave\n- key locations in Keychain\n- browser session attach rules\n\n## Promote to SOUL.md\nWhen the learning is a behavior/principle rule.\nExamples:\n- do not let no-assignment closeout replace required deliverables\n- do not treat shallow checks as full acceptance\n\n## Promote to Obsidian\nWhen the learning should become reusable operator material, marketing proof, or an operations note outside transient chat.\n\nBy default, Obsidian-style exports go to the local safe fallback:\n- `.learnings/exports/obsidian/`\n\nIf you want a real vault destination, set `OBSIDIAN_LEARNINGS_DIR` explicitly before running the promotion script.\nAlways confirm the printed target path first, or use `--dry-run`.\n\nExample:\n- `node scripts/promote-learning.mjs obsidian \"Reusable learning\" --dry-run`\n- then rerun without `--dry-run` after confirming the path\n\nFile v0.2.9:references/schema.md\n\n# Learning Schema\n\n## Files\n- `.learnings/LEARNINGS.md`\n- `.learnings/ERRORS.md`\n- `.learnings/FEATURE_REQUESTS.md`\n\n## Learning entry\n```md\n## [LRN-YYYYMMDD-XXX] category\n\n**Logged**: ISO-8601 timestamp\n**Priority**: low | medium | high | critical\n**Status**: pending\n**Area**: workflow | tools | product | growth | security | infra\n\n### Summary\nOne-line learning\n\n### Details\nWhat happened and what is now understood\n\n### Suggested Action\nSpecific next action\n\n### Metadata\n- Source: user_feedback | error | review | postmortem\n- Related Files: path/to/file\n- Tags: tag1, tag2\n```\n\n## Error entry\n```md\n## [ERR-YYYYMMDD-XXX] name\n\n**Logged**: ISO-8601 timestamp\n**Priority**: high\n**Status**: pending\n**Area**: infra | product | growth | security | ops\n\n### Summary\nWhat failed\n\n### Error\nActual error or concise failure output\n\n### Suggested Fix\nLikely fix or next step\n\n### Metadata\n- Reproducible: yes | no | unknown\n- Related Files: path/to/file\n```\n\n## Feature request entry\n```md\n## [FEAT-YYYYMMDD-XXX] capability\n\n**Logged**: ISO-8601 timestamp\n**Priority**: medium\n**Status**: pending\n**Area**: product | ops | growth | security\n\n### Requested Capability\nWhat is missing\n\n### User Context\nWhy it matters\n\n### Suggested Implementation\nMinimal implementation direction\n```\n\n## Experiment entry\n```md\n## [EXP-YYYYMMDD-XXX] experiment\n\n**Logged**: ISO-8601 timestamp\n**Priority**: medium | high | critical\n**Status**: baseline | testing | keep | discard | partial_keep\n**Area**: workflow | tools | product | growth | security | infra | ops\n\n### Target\nWhat repeated problem is being improved\n\n### Baseline\nWhat was failing before the change\n\n### Mutation\nThe single change introduced\n\n### Binary Evals\n- [ ] Eval 1\n- [ ] Eval 2\n- [ ] Eval 3\n\n### Result\nWhat improved / did not improve\n\n### Keep or Discard\nkeep | discard | partial_keep\n```\n\nFile v0.2.9:releases/v0.2.0-clawhub-listing.md\n\n# ClawHub Listing Copy — OpenClaw Self-Improvement v0.2.0\n\n## Short description\nTurn repeated AI-agent mistakes into durable operational improvements with learning capture, binary eval loops, keep/discard decisions, and promotion into SOPs, checklists, and workflow rules.\n\n## Long description\n**OpenClaw Self-Improvement** is a skill for OpenClaw / ClawLite operators and multi-agent teams who want to stop repeating the same operational mistakes.\n\nIt helps agents and operators:\n- log learnings, errors, feature gaps, and experiments\n- run lightweight **binary eval loops** on repeated failures\n- test whether a new guardrail, SOP, checklist, or schema change actually helps\n- classify changes as **keep**, **partial_keep**, or **discard**\n- promote proven improvements into `AGENTS.md`, `TOOLS.md`, `SOUL.md`, and `docs/ops/*.md`\n\nThis makes it especially useful for:\n- self-improving AI workflows\n- Mission Control truth-state quality\n- deploy closeout verification\n- receipt / proof completeness\n- repeated failure prevention in multi-agent systems\n\n### Best keywords / search phrases\n- self-improving AI workflow\n- binary evals for agent operations\n- OpenClaw self-improvement skill\n- repeated failure prevention for AI agents\n- deploy closeout verification\n- Mission Control QA improvement\n- durable operational learning for multi-agent systems\n\n## What’s new in v0.2.0\n- Experiment mode for repeated failures\n- Binary eval loops\n- Keep / partial_keep / discard decisions\n- Practical examples for summary quality and deploy closeout\n- Experiment summary helper\n- Decision rules for log vs experiment vs promote\n- Daily routine integration with heartbeat and Karen QA\n\nFile v0.2.9:releases/v0.2.0-github-release.md\n\n# OpenClaw Self-Improvement v0.2.0\n\n**OpenClaw Self-Improvement v0.2.0** turns self-improvement from a vague idea into an operational loop.\n\nThis release is for OpenClaw / ClawLite operators and multi-agent teams that want to reduce repeated failures in a durable, inspectable way.\n\n## Highlights\n\n### New in v0.2.0\n- **Experiment mode** for repeated failures\n- **Binary eval loops** for testing new guardrails, SOPs, and workflow changes\n- **Keep / partial_keep / discard** decision model\n- **Practical examples** for:\n  - Mission Control summary quality\n  - deploy closeout / production verification\n- **Experiment summary helper** for surfacing unresolved follow-up debt\n- **Decision rules** for when to log, experiment, or promote\n- **Daily routine integration** with heartbeat, Karen QA, AGENTS, and self-improving ops policy\n\n## Why this release matters\n\nMany agent systems can log lessons.\nFar fewer can test whether a new rule, checklist, or guardrail actually reduced the repeated failure.\n\nv0.2.0 adds a lightweight eval-driven improvement loop:\n1. capture the repeated problem\n2. define a baseline\n3. change one thing at a time\n4. evaluate with binary checks\n5. classify the result as `keep`, `partial_keep`, or `discard`\n6. promote proven improvements into durable operating rules\n\nThis makes self-improvement more operational and less hand-wavy.\n\n## What problems this skill is built for\n\nOpenClaw Self-Improvement is especially useful when your agent workflows keep repeating problems like:\n- fake-complete states\n- missing receipts or weak proof bundles\n- back-end fixes that never reach operator-facing truth surfaces\n- code-ready or build-ready states being confused with production-ready states\n- new rules being added without evidence that they actually improve outcomes\n\n## Best-fit use cases\n\n- self-improving AI workflows\n- binary evals for agent operations\n- Mission Control truth-state improvement\n- deploy closeout verification\n- repeated failure prevention in OpenClaw or ClawLite-style systems\n- durable learning loops for multi-agent teams\n\n## Practical examples included\n\nThis release includes concrete examples for:\n- **Mission Control summary link-complete gates**\n- **ClawLite deploy closeout gates**\n\nThese examples show how to move from repeated failure → baseline → mutation → binary eval → keep/discard decision.\n\n## New files / capabilities\n\n### Scripts\n- `scripts/log-experiment.mjs`\n- `scripts/experiment-summary.mjs`\n\n### References\n- `references/eval-loop.md`\n- `references/examples.md`\n- `references/decision-rules.md`\n\n### Ops integration\n- Karen QA now treats `partial_keep` as active follow-up debt\n- heartbeat can check unresolved experiment follow-up items\n- daily routine + policy + AGENTS entry updated to make self-improvement part of normal operations\n\n## Why `partial_keep` matters\n\nOne of the most important additions in this release is the idea that a fix can be **directionally correct but still incomplete**.\n\nExample:\n- summary JSON becomes better\n- but the front-end still fails to render the new fields\n\nThat should not be treated as “done.”\nIt should be tracked as `partial_keep` until the paired fix lands.\n\n## Upgrade summary\n\nIf you already used earlier versions of this skill, v0.2.0 adds:\n- a stronger experiment layer\n- clearer decision rules\n- better examples\n- a runtime-friendly summary tool\n- tighter integration with OpenClaw-style QA and heartbeat workflows\n\n## Bottom line\n\n**OpenClaw Self-Improvement v0.2.0 is for teams that want self-improvement to mean more than writing down lessons.**\n\nIt helps you test whether a new workflow rule actually reduces repeated failure, keep what works, and keep unfinished follow-up debt visible until it is truly closed.\n\nFile v0.2.9:releases/v0.2.1-github-release.md\n\n# v0.2.1\n\n- improve README positioning for SEO / GEO\n- add clearer landing-page summary for GitHub and ClawHub readers\n\nFile v0.2.9:releases/v0.2.2-github-release.md\n\n# v0.2.2\n\n- improve README positioning for SEO / GEO\n- add clearer landing-page summary for GitHub and ClawHub readers\n\nArchive v0.2.8: 21 files, 22781 bytes\n\nFiles: package.json (812b), README.md (11770b), references/decision-rules.md (3334b), references/eval-loop.md (3305b), references/examples.md (4367b), references/promotion-guide.md (1175b), references/schema.md (1846b), releases/v0.2.0-clawhub-listing.md (1678b), releases/v0.2.0-github-release.md (3744b), releases/v0.2.1-github-release.md (119b), releases/v0.2.2-github-release.md (119b), releases/v0.2.3-github-release.md (127b), releases/v0.2.4-github-release.md (147b), releases/v0.2.7-github-release.md (147b), releases/v0.2.8-github-release.md (221b), scripts/experiment-summary.mjs (1749b), scripts/log-experiment.mjs (1699b), scripts/log-learning.mjs (1683b), scripts/promote-learning.mjs (1263b), SKILL.md (4677b), _meta.json (144b)\n\nFile v0.2.8:SKILL.md\n\n---\nname: openclaw-self-improvement\ndescription: A reusable self-improving AI agent skill for OpenClaw and ClawLite that turns repeated failures into guardrails, binary eval loops, SOPs, checklists, and proof-based operational improvements.\n---\n\n# OpenClaw / ClawLite Self-Improvement\n\nUse this skill to turn mistakes, corrections, blockers, and better approaches into durable operating knowledge.\n\n## What problem this solves\nAI ops often repeat the same failures because mistakes stay in chat history instead of becoming system rules. This skill creates a lightweight improvement loop:\n- log failures and learnings\n- separate errors from feature requests\n- run small eval-driven experiments on repeated failures\n- promote important patterns into AGENTS.md / TOOLS.md / SOUL.md\n- write operator notes into Obsidian vault\n- support stricter acceptance via Karen / Mission Control\n\n## When to use\nUse this skill when the user asks:\n- \"make the agent improve itself\"\n- \"capture learnings\"\n- \"log mistakes so we do not repeat them\"\n- \"record blockers / corrections / feature gaps\"\n- \"build a self-improving OpenClaw workflow\"\n- \"operationalize lessons learned\"\n- \"test whether this new rule actually helps\"\n- \"run an eval loop on this workflow/skill/SOP\"\n- \"should we keep this new guardrail or discard it\"\n\n## Files this skill uses\n- `.learnings/LEARNINGS.md`\n- `.learnings/ERRORS.md`\n- `.learnings/FEATURE_REQUESTS.md`\n- `.learnings/EXPERIMENTS.md`\n- Obsidian vault note under `ClawLite/Operations/Learnings/`\n\n## Command examples\n```bash\nnode {baseDir}/scripts/log-learning.mjs learning \"Summary\" \"Details\" \"Suggested action\"\nnode {baseDir}/scripts/log-learning.mjs error \"Summary\" \"Error details\" \"Suggested fix\"\nnode {baseDir}/scripts/log-learning.mjs feature \"Capability name\" \"User context\" \"Suggested implementation\"\nnode {baseDir}/scripts/log-learning.mjs experiment \"Target problem\" \"Baseline failure\" \"Single mutation to test\"\nnode {baseDir}/scripts/log-experiment.mjs \"Target problem\" \"Baseline failure\" \"Single mutation\" \"eval1|eval2|eval3\" \"Result summary\" \"testing\"\nnode {baseDir}/scripts/promote-learning.mjs workflow \"Rule text\"\n```\n\n## Categories\n### learning\nUse for:\n- user corrections\n- better recurring workflows\n- tool gotchas\n- operational lessons\n\n### error\nUse for:\n- command failures\n- integration failures\n- runtime blockers\n- broken release / deploy behavior\n\n### feature\nUse for:\n- missing capability requests\n- operator workflow gaps\n- recurring requests that deserve a build item\n\n### experiment\nUse for:\n- repeated failures that need a tested guardrail\n- checklist/SOP/schema changes that should be validated before broad promotion\n- keep/discard decisions on new operating rules\n- binary eval loops for skills, workflows, receipts, summaries, or deploy closeout rules\n\n## Promotion targets\n- `AGENTS.md` → workflow / delegation / execution rules\n- `TOOLS.md` → tool gotchas, secrets locations, environment routing rules\n- `SOUL.md` → behavior / communication / non-negotiable principles\n- Obsidian vault → reusable operator log and content proof asset\n\n## Karen / Mission Control compatibility\nThis skill is designed to work with stricter ops governance:\n- Karen can reference learnings when repeated failures happen\n- Mission Control can treat promoted learnings as new operating rules\n- recurring blockers can be elevated from chat into tracked operational knowledge\n- experiments can test whether a new summary contract, receipt rule, or deploy closeout guardrail actually reduced the failure pattern\n\n## Eval loop rule\nWhen a repeated failure is turning into a new rule/SOP/checklist, do not only log it.\nAlso:\n1. define 3-5 binary evals\n2. record the baseline failure state\n3. change one thing at a time\n4. re-check the same evals\n5. classify the change as keep / discard / partial_keep\n\nUse `{baseDir}/references/eval-loop.md` for the experiment format and examples.\n\n## Output goal\nA good use of this skill should produce one of:\n- a durable learning entry\n- a durable error entry\n- a durable feature request entry\n- a durable experiment entry with binary evals\n- a promoted rule in AGENTS.md / TOOLS.md / SOUL.md\n- an Obsidian vault operations note\n\n## Important limits\n- Logging is not the same as fixing.\n- Do not treat a learning entry as closure for a broken deliverable.\n- Use this skill to reduce repeated mistakes, not to excuse them.\n\n## References\n- `{baseDir}/references/schema.md`\n- `{baseDir}/references/promotion-guide.md`\n- `{baseDir}/references/eval-loop.md`\n- `{baseDir}/references/examples.md`\n- `{baseDir}/references/decision-rules.md`\n\n- `{baseDir}/references/eval-loop.md`\n- `{baseDir}/references/examples.md`\n\nFile v0.2.8:README.md\n\n# OpenClaw Self-Improvement\n\n**OpenClaw Self-Improvement** is a reusable agent skill for turning repeated AI-agent mistakes into durable operational improvements, measurable guardrails, and inspectable workflow upgrades.\n\n## Landing-page summary\n\nMost AI agents do not really improve. They repeat mistakes, hide partial failures behind optimistic language, and leave lessons trapped in chat history.\n\nOpenClaw Self-Improvement gives you a practical operating loop for **self-improving AI agents**:\n- capture repeated failures\n- test one guardrail at a time\n- verify whether it actually reduces failure\n- promote proven fixes into SOPs, checklists, policies, and workflow rules\n\nIf you want **AI agents that get more reliable over time**, **multi-agent workflows that stop repeating the same mistakes**, or **proof-based QA for agent operations**, this skill is built for that exact use case.\n\n## Multilingual summary\n\n**中文：** 这是一个面向 AI agent 自我改进的实战型 skill，用来减少重复犯错、建立 guardrails、验证修复是否真的有效，并把经验沉淀成 SOP、检查清单和可复用规则。\n\n**日本語：** これは自己改善する AI エージェント向けの実践的な skill です。繰り返し発生する失敗を減らし、ガードレールを検証し、改善を SOP・チェックリスト・再利用可能な運用ルールへ昇格させます。\n\n**한국어：** 이 스킬은 스스로 개선하는 AI 에이전트를 위한 운영형 skill입니다. 반복 실수를 줄이고, 가드레일이 실제로 효과가 있는지 검증하며, 개선 사항을 SOP·체크리스트·재사용 가능한 규칙으로 승격합니다.\n\n**Español：** Esta skill está diseñada para agentes de IA que deben mejorar con el tiempo. Ayuda a reducir errores repetidos, validar guardrails operativos y convertir mejoras en SOP, checklists y reglas reutilizables.\n\nIf you want a practical way to deploy OpenClaw with cheaper tokens, BYOK flexibility, and operator control, see **[ClawLite](https://clawlite.ai)**.\n\n## TL;DR\n\nIf you are looking for a practical system for **self-improving AI agents**, **AI workflow optimization**, **multi-agent failure prevention**, **binary eval loops**, or **agent operations QA**, this skill is designed for that exact job.\n\nIt helps OpenClaw / ClawLite operators and agent teams:\n- log recurring failures\n- separate one-off errors from reusable lessons\n- run lightweight **binary eval loops** on new guardrails\n- classify changes as **keep**, **partial_keep**, or **discard**\n- promote proven fixes into SOPs, checklists, workflow rules, and operating policy\n\nIf you care about reducing fake-complete states, tightening QA truth, improving deploy closeout, and making agent learning inspectable, this skill is built for that job.\n\n---\n\n## Why this skill exists\n\nMany AI systems say they \"learn,\" but most only store lessons in chat history or loose notes.\n\nThat is not enough.\n\nOperationally, repeated failures tend to come back in the same forms:\n- delivery gets described as complete before proof exists\n- receipts are missing or too thin\n- back-end fixes never reach the operator-facing surface\n- code-ready states get confused with production-ready states\n- teams add new rules without checking whether those rules actually reduce failure\n\nOpenClaw Self-Improvement gives you a lightweight operating loop for fixing that.\n\n---\n\n## What problem it solves\n\nThis skill helps with:\n- **self-improving AI workflows**\n- **AI operations learning loops**\n- **binary evals for agent guardrails**\n- **Mission Control truth-state improvement**\n- **deploy closeout verification**\n- **receipt / proof completeness**\n- **repeated failure prevention in multi-agent systems**\n- **AI agent reliability engineering**\n- **agent QA systems for OpenClaw, ClawLite, and similar stacks**\n\nIt is especially useful in OpenClaw-style environments where multiple agents, tools, SOPs, and truth surfaces interact.\n\n## Who this is for\n\nThis skill is useful for:\n- OpenClaw operators\n- ClawLite growth / marketing / QA lanes\n- AI agent builders who need durable postmortems instead of vague “reflection”\n- teams running multi-agent workflows with receipts, truth surfaces, and closeout gates\n- anyone trying to reduce repeated AI-agent mistakes in production-like operations\n\n---\n\n## What’s new in v0.2.0\n\n### New capabilities\n- **Experiment mode** for repeated failures\n- **Binary eval loops** for testing whether a new guardrail or SOP actually helps\n- **Keep / partial_keep / discard** decision model\n- **Practical examples** for:\n  - Mission Control summary quality\n  - deploy closeout / production verification\n- **Experiment summary helper** for surfacing unresolved follow-up debt\n- **Decision rules** for when to log, experiment, or promote\n- **Daily routine integration** with heartbeat, Karen QA, AGENTS, and ops policy\n\n### Why it matters\nThis release makes self-improvement more operational.\n\nInstead of stopping at “lesson learned,” you can now:\n1. capture the repeated problem\n2. define a baseline\n3. test one change at a time\n4. evaluate with binary checks\n5. keep, discard, or mark `partial_keep`\n\nThat makes the improvement loop much more auditable and much less hand-wavy.\n\n---\n\n## Core workflow\n\nOpenClaw Self-Improvement now supports a practical loop:\n\n1. **Capture** a learning, error, feature request, or experiment\n2. **Store** it in structured local files\n3. **Experiment** when a repeated failure needs a tested guardrail\n4. **Evaluate** the change with binary checks\n5. **Promote** proven fixes into durable rules, SOPs, or policies\n6. **Track follow-up debt** when a fix is only partial\n\n---\n\n## Best use cases\n\nUse this skill when you want to:\n- capture lessons so agents stop repeating the same mistake\n- log recurring operational errors\n- track feature gaps revealed by repeated work\n- test whether a new workflow rule really improves outcomes\n- run a lightweight eval loop on a skill, SOP, checklist, schema, or handoff rule\n- decide whether a new guardrail should be kept, discarded, or promoted\n- build a self-improving OpenClaw or ClawLite operating loop\n\nTypical targets include:\n- Mission Control summary quality\n- deploy closeout gates\n- receipt requirements\n- QA wording rules\n- truth-surface rendering checks\n- handoff contracts between agents\n\n---\n\n## Files it manages\n\n- `.learnings/LEARNINGS.md`\n- `.learnings/ERRORS.md`\n- `.learnings/FEATURE_REQUESTS.md`\n- `.learnings/EXPERIMENTS.md`\n- Optional Obsidian export directory via `OBSIDIAN_LEARNINGS_DIR`\n- Default local export fallback: `.learnings/exports/obsidian/`\n- Safe by default: no hard-coded external vault path, and `scripts/promote-learning.mjs` prints the resolved write target before writing\n\n---\n\n## Install\n\n```bash\nnpm install\n```\n\n---\n\n## Usage\n\n### Log a learning\n\n```bash\nnode scripts/log-learning.mjs learning \"Summary\" \"Details\" \"Suggested action\"\n```\n\n### Log an error\n\n```bash\nnode scripts/log-learning.mjs error \"Summary\" \"Error details\" \"Suggested fix\"\n```\n\n### Log a feature request\n\n```bash\nnode scripts/log-learning.mjs feature \"Capability name\" \"User context\" \"Suggested implementation\"\n```\n\n### Log a tested experiment\n\n```bash\nnode scripts/log-experiment.mjs \"Target problem\" \"Baseline failure\" \"Single mutation\" \"eval1|eval2|eval3\" \"Result summary\" \"testing\"\n```\n\n### Promote a rule\n\n```bash\nnode scripts/promote-learning.mjs workflow \"Rule text\"\nnode scripts/promote-learning.mjs obsidian \"Reusable learning\" --dry-run\n```\n\n### Summarize experiment outcomes\n\n```bash\nnode scripts/experiment-summary.mjs\n```\n\n---\n\n## Decision model\n\nThis skill uses three levels of action:\n\n### 1. Log only\nUse when:\n- the issue happened once\n- root cause is still unclear\n- there is not enough evidence yet to make a rule\n\n### 2. Experiment\nUse when:\n- the issue repeated 2+ times\n- a new guardrail / SOP / checklist / schema change is being proposed\n- you can define 3–5 binary evals\n\n### 3. Promote\nUse when:\n- the rule is clearly right and low-risk\n- the issue is severe enough that waiting would be irresponsible\n- the rule is about ownership, truth, or a non-negotiable operating principle\n\n---\n\n## Practical examples\n\nThe skill now includes concrete examples for:\n- **Mission Control summary link-complete gates**\n- **ClawLite deploy closeout gates**\n\nThese examples show how to:\n- define the repeated failure\n- capture the baseline\n- propose one mutation\n- evaluate with binary checks\n- classify the outcome as `keep`, `partial_keep`, or `discard`\n\n---\n\n## Promotion targets\n\nPromote proven improvements into:\n- `AGENTS.md` — workflow / delegation / execution rules\n- `TOOLS.md` — tool gotchas and environment routing rules\n- `SOUL.md` — behavior / communication / non-negotiable principles\n- `docs/ops/*.md` — SOPs, policy, and operating contracts\n- Obsidian vault — reusable operator notes and operational memory\n\n---\n\n## Important limits\n\n- Logging is **not** the same as fixing.\n- A learning entry does **not** close a broken deliverable.\n- A back-end-only improvement is not complete if the visible operator-facing surface is still stale.\n- `partial_keep` should be treated as **active follow-up debt**, not as closure.\n\n---\n\n## Repository contents\n\n- `SKILL.md` — agent-facing routing and usage guidance\n- `scripts/log-learning.mjs` — append a learning / error / feature request / experiment\n- `scripts/log-experiment.mjs` — append a structured experiment with binary evals\n- `scripts/experiment-summary.mjs` — summarize keep / partial_keep / discard outcomes and flag follow-up debt\n- `scripts/promote-learning.mjs` — promote a lesson into durable operating rules with explicit path echo and optional `--dry-run`\n- `references/schema.md` — data structure guidance\n- `references/promotion-guide.md` — what to promote and where\n- `references/eval-loop.md` — how to run lightweight binary-eval improvement loops\n- `references/examples.md` — practical examples for summary gates and deploy closeout gates\n- `references/decision-rules.md` — when to log only, run an experiment, or promote immediately\n\n---\n\n## SEO / GEO positioning\n\nThis skill is intentionally legible to both search engines and AI answer engines because it is built around concrete, reusable operational concepts rather than vague “AI reflection” language.\n\n### Primary search themes\n- self-improving AI workflows\n- self-improving agent systems\n- AI agent reliability engineering\n- binary eval loops for agent guardrails\n- repeated failure prevention in multi-agent systems\n- deploy closeout verification\n- proof-based QA for AI operations\n- operational learning loops for OpenClaw / ClawLite style stacks\n\n### Why that matters\nThese phrases map to real operator intent:\n- “how do I stop my AI agents from repeating mistakes?”\n- “how do I build a self-improving agent workflow?”\n- “how do I test whether a new guardrail actually works?”\n- “how do I make AI operations auditable?”\n\nThat makes the skill easier to explain, cite, retrieve, and reuse than generic memory or reflection systems.\n\n---\n\n## Bottom line\n\nIf you want OpenClaw to improve over time instead of repeating the same mistakes across sessions, this repo gives you:\n- an operational memory loop\n- a lightweight eval loop for testing whether a new guardrail actually helps\n- a durable promotion path from mistake -> experiment -> policy -> reusable system asset\n- a decision framework for when to log, experiment, or promote\n- a way to keep unresolved partial improvements visible until they are actually closed\new guardrail actually helps\n- a decision framework for when to log, experiment, or promote\n- a way to keep unresolved partial improvements visible until they are actually closed\n\nFile v0.2.8:_meta.json\n\n{\n  \"ownerId\": \"kn7d88952ey3hbm158x7ejqs1d81zmyq\",\n  \"slug\": \"openclaw-self-improvement\",\n  \"version\": \"0.2.8\",\n  \"publishedAt\": 1776236660417\n}\n\nFile v0.2.8:references/decision-rules.md\n\n# Decision Rules for Self-Improvement\n\nUse this reference to decide whether a new issue should become:\n- a simple learning entry\n- an experiment with binary evals\n- a promoted operating rule\n\n---\n\n## Option 1 — Log only\n\nUse **log only** when:\n- the issue happened once\n- root cause is still unclear\n- there is not enough evidence yet to turn it into a rule\n- the lesson is useful, but not broadly reusable yet\n\nTypical output:\n- `learning`\n- `error`\n- `feature`\n\nExamples:\n- one-off API outage\n- first-time tool glitch with unclear cause\n- user preference that does not affect system-wide ops\n\n---\n\n## Option 2 — Run an experiment\n\nUse an **experiment** when:\n- the same failure happened 2+ times\n- a new guardrail/SOP/checklist/schema change is being proposed\n- you can define 3-5 binary evals\n- you want evidence that the change helped before promoting it broadly\n\nTypical output:\n- `experiment`\n- baseline + mutation + binary evals + keep/discard decision\n\nExamples:\n- Mission Control summaries repeatedly missing links/details\n- deploy closeout repeatedly confusing code-ready with live\n- repeated missing receipts or incomplete proof bundles\n- repeated front-end / operator-surface mismatch after backend fixes\n\n---\n\n## Option 3 — Promote immediately\n\nUse **promote immediately** when:\n- the rule is already obviously correct and low-risk\n- the issue is severe enough that waiting would be irresponsible\n- the required change is a principle or ownership rule, not an uncertain optimization\n- operator review already confirms the new rule should become standard\n\nTypical output:\n- promotion into `AGENTS.md`, `TOOLS.md`, `SOUL.md`, or `docs/ops/*.md`\n\nExamples:\n- deployment owner must be explicit\n- code-ready is not the same as live\n- missing receipt cannot be treated as delivered\n- summary without proof links is not operator-complete\n\n---\n\n## Promote after experiment\n\nUse **experiment first, then promote** when:\n- the rule sounds plausible but may add friction\n- you are not sure whether the added checklist/schema field actually reduces errors\n- the change could create process overhead without improving truth quality\n\nExamples:\n- adding new summary schema fields\n- adding new receipt requirements\n- adding extra verification steps to handoff or QA lanes\n\n---\n\n## Anti-patterns\n\nDo **not** run an experiment when:\n- there is no clear repeated failure\n- the evals would be vague or subjective\n- the issue is really just missing execution, not missing learning\n- the fix requires immediate owner action, not more analysis\n\nDo **not** promote immediately when:\n- the rule is still based on one anecdote\n- the change is likely to create bureaucracy without proof of benefit\n- the actual failure surface is still unclear\n\n---\n\n## Quick decision tree\n\n1. Did this happen only once?\n- Yes → log only\n- No → continue\n\n2. Is the new rule obviously necessary and low-risk?\n- Yes → promote immediately\n- No → continue\n\n3. Can you define 3-5 binary evals for the proposed change?\n- Yes → run an experiment\n- No → log only until the failure is clearer\n\n4. Did the experiment materially improve the failure pattern?\n- Yes → keep and consider promotion\n- No → discard or partial_keep\n\n---\n\n## One-line heuristic\n\n**Single incident = log. Repeated pattern = experiment. Clear principle/ownership rule = promote.**\n\nFile v0.2.8:references/eval-loop.md\n\n# Eval Loop for Self-Improvement\n\nUse this reference when a repeated failure should become a tested operational improvement instead of only a logged lesson.\n\n## Goal\n\nDo not only ask \"what did we learn?\"\nAlso ask:\n- what is the current baseline?\n- what exact guardrail or rule changed?\n- how will we measure whether it helped?\n- should we keep or discard the change?\n\n## Use this loop for\n- repeated Mission Control wording failures\n- missing receipts / missing proof chains\n- deploy closeout failures\n- stale operator-facing surfaces\n- repeated handoff mistakes between agents\n- recurring SOP/checklist changes\n\n## 1. Define the target\n\nState one concrete thing you want to improve.\n\nExamples:\n- Hunter summary should always include concrete links and details\n- ClawLite deploy closeout should never stop at code-ready status\n- Mission Control front-end should render source links from structured fields\n\n## 2. Write 3-5 binary evals\n\nEach eval must be yes/no.\n\nExamples for summary quality:\n- Does the summary include at least one artifact path or URL?\n- Does the summary include evidence links when external proof matters?\n- Does the summary include a detail block describing what actually changed?\n- Does the summary include the next handoff or recovery action?\n- Does the operator-facing surface actually render these fields?\n\nExamples for deploy closeout:\n- Is the deployed commit hash recorded?\n- Is a deployment ref/URL recorded?\n- Was the production page or sitemap actually verified?\n- Was a structured receipt written?\n- Is the final state classified with the correct deploy-state vocabulary?\n\n## 3. Capture baseline\n\nBefore changing the rule/SOP/skill/checklist:\n- record the current failure pattern\n- record which evals currently fail\n- treat this as the baseline state\n\n## 4. Change only one thing\n\nGood changes:\n- one wording rule\n- one new checklist item\n- one schema field\n- one render mapping\n- one validation step\n\nBad changes:\n- rewriting everything at once\n- adding five new rules at once\n- changing wording and schema and code together unless absolutely required\n\n## 5. Re-check and classify\n\nAfter the single change:\n- run the same evals again\n- note which checks improved\n- decide:\n  - KEEP\n  - DISCARD\n  - PARTIAL_KEEP\n\n## 6. Promotion rule\n\nOnly promote broadly reusable changes after they pass the eval loop or after operator review confirms the change materially reduced the failure.\n\n## Suggested experiment entry format\n\n```md\n## [EXP-YYYYMMDD-XXX] experiment\n\n**Logged**: ISO-8601 timestamp\n**Priority**: medium | high | critical\n**Status**: baseline | testing | keep | discard | partial_keep\n**Area**: workflow | tools | product | growth | security | infra | ops\n\n### Target\nWhat repeated problem is being improved\n\n### Baseline\nWhat was failing before the change\n\n### Mutation\nThe single change introduced\n\n### Binary Evals\n- [ ] Eval 1\n- [ ] Eval 2\n- [ ] Eval 3\n\n### Result\nWhat improved / did not improve\n\n### Keep or Discard\nkeep | discard | partial_keep\n\n### Metadata\n- Source: review | postmortem | user_feedback | qa\n- Related Files:\n- Tags:\n```\n\n## Important limit\n\nA logged experiment is not the same as a finished fix. If the production surface or operator-visible truth is still wrong, the experiment remains incomplete even if the local change looks promising.\n\nFile v0.2.8:references/examples.md\n\n# Practical Examples for Self-Improvement Eval Loops\n\nUse these examples when you want to turn a repeated operational mistake into a tested guardrail.\n\n---\n\n## Example 1 — Mission Control summary link-complete gate\n\n### Repeated failure\nMission Control agent summaries were updated at the JSON layer but still too thin for operator use. They lacked concrete links, detail blocks, and next handoff context. In some cases the front-end also failed to render the newly added structured fields.\n\n### Target\nMake Mission Control summaries decision-ready instead of status-only.\n\n### Baseline\nBefore the change:\n- summary text could say \"DELIVERED\" without concrete artifact links\n- evidence existed in underlying files but not in summary fields\n- operator-facing UI could still show \"No source links in current summary\"\n\n### Single mutation example\nAdd a summary contract requiring:\n- at least one artifact path or URL\n- evidence links when external proof matters\n- a detail block\n- next handoff / recovery action\n\n### Binary evals\n- [ ] Does the summary include at least one artifact path or URL?\n- [ ] Does the su\n\nArchive v0.2.7: 20 files, 22060 bytes\n\nFiles: package.json (812b), README.md (11510b), references/decision-rules.md (3334b), references/eval-loop.md (3305b), references/examples.md (4367b), references/promotion-guide.md (747b), references/schema.md (1846b), releases/v0.2.0-clawhub-listing.md (1678b), releases/v0.2.0-github-release.md (3744b), releases/v0.2.1-github-release.md (119b), releases/v0.2.2-github-release.md (119b), releases/v0.2.3-github-release.md (127b), releases/v0.2.4-github-release.md (147b), releases/v0.2.7-github-release.md (147b), scripts/experiment-summary.mjs (1749b), scripts/log-experiment.mjs (1699b), scripts/log-learning.mjs (1683b), scripts/promote-learning.mjs (1030b), SKILL.md (4677b), _meta.json (144b)\n\nArchive v0.2.6: 15 files, 20083 bytes\n\nFiles: package.json (1159b), README.md (8301b), references/decision-rules.md (3334b), references/eval-loop.md (3305b), references/examples.md (4367b), references/promotion-guide.md (747b), references/schema.md (1846b), releases/v0.2.0-clawhub-listing.md (1678b), releases/v0.2.0-github-release.md (3744b), scripts/experiment-summary.mjs (1749b), scripts/log-experiment.mjs (1699b), scripts/log-learning.mjs (1683b), scripts/promote-learning.mjs (1157b), SKILL.md (5289b), _meta.json (144b)\n\nArchive v0.2.5: 15 files, 20437 bytes\n\nFiles: package.json (1045b), README.md (8301b), references/decision-rules.md (3334b), references/eval-loop.md (3305b), references/examples.md (4367b), references/promotion-guide.md (747b), references/schema.md (1846b), releases/v0.2.0-clawhub-listing.md (1678b), releases/v0.2.0-github-release.md (3744b), scripts/experiment-summary.mjs (1749b), scripts/log-experiment.mjs (1699b), scripts/log-learning.mjs (1683b), scripts/promote-learning.mjs (1157b), SKILL.md (6034b), _meta.json (144b)\n\nArchive v0.2.4: 15 files, 20671 bytes\n\nFiles: package.json (1371b), README.md (8301b), references/decision-rules.md (3334b), references/eval-loop.md (3305b), references/examples.md (4367b), references/promotion-guide.md (747b), references/schema.md (1846b), releases/v0.2.0-clawhub-listing.md (1678b), releases/v0.2.0-github-release.md (3744b), scripts/experiment-summary.mjs (1749b), scripts/log-experiment.mjs (1699b), scripts/log-learning.mjs (1683b), scripts/promote-learning.mjs (1157b), SKILL.md (6034b), _meta.json (144b)\n\nArchive v0.2.3: 15 files, 20272 bytes\n\nFiles: package.json (997b), README.md (8301b), references/decision-rules.md (3334b), references/eval-loop.md (3305b), references/examples.md (4367b), references/promotion-guide.md (747b), references/schema.md (1846b), releases/v0.2.0-clawhub-listing.md (1678b), releases/v0.2.0-github-release.md (3744b), scripts/experiment-summary.mjs (1749b), scripts/log-experiment.mjs (1699b), scripts/log-learning.mjs (1683b), scripts/promote-learning.mjs (1157b), SKILL.md (6034b), _meta.json (144b)\n\nArchive v0.2.2: 15 files, 20273 bytes\n\nFiles: package.json (997b), README.md (8301b), references/decision-rules.md (3334b), references/eval-loop.md (3305b), references/examples.md (4367b), references/promotion-guide.md (747b), references/schema.md (1846b), releases/v0.2.0-clawhub-listing.md (1678b), releases/v0.2.0-github-release.md (3744b), scripts/experiment-summary.mjs (1749b), scripts/log-experiment.mjs (1699b), scripts/log-learning.mjs (1683b), scripts/promote-learning.mjs (1157b), SKILL.md (6047b), _meta.json (144b)","readmeExcerpt":"Skill: OpenClaw Self-Improvement Owner: x-rayluan Summary: A reusable operator-guided workflow improvement skill for OpenClaw and ClawLite that turns repeated failures into logged learnings, binary eval loops, SOPs,... Tags: latest:0.2.11 Version history: v0.2.11 | 2026-05-01T08:13:57.502Z | user Add scorecard repair loop with recovery ticket generation. v0.2.10 | 2026-05-01T08:07:07.960Z | user Add harness observabi","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"node {baseDir}/scripts/log-learning.mjs learning \"Summary\" \"Details\" \"Suggested action\"\nnode {baseDir}/scripts/log-learning.mjs error \"Summary\" \"Error details\" \"Suggested fix\"\nnode {baseDir}/scripts/log-learning.mjs feature \"Capability name\" \"User context\" \"Suggested implementation\"\nnode {baseDir}/scripts/log-learning.mjs experiment \"Target problem\" \"Baseline failure\" \"Single mutation to test\"\nnode {baseDir}/scripts/log-experiment.mjs \"Target problem\" \"Baseline failure\" \"Single mutation\" \"eval1|eval2|eval3\" \"Result summary\" \"testing\"\nnode {baseDir}/scripts/promote-learning.mjs workflow \"Rule text\"\nnode {baseDir}/scripts/analyze-openclaw-failures.mjs --output /Users/m1/.openclaw/workspace/memory/harness-backlog-latest.md\nnode {baseDir}/scripts/daily-agent-scorecard.mjs --output /Users/m1/.openclaw/workspace/mission-control/data/delivery-receipts/agent-scorecard-$(date +%F).md\nnode {baseDir}/scripts/daily-agent-scorecard.mjs --repair --output /Users/m1/.openclaw/workspace/mission-control/data/delivery-receipts/agent-scorecard-$(date +%F).md"},{"language":"bash","snippet":"npm install"},{"language":"bash","snippet":"node scripts/log-learning.mjs learning \"Summary\" \"Details\" \"Suggested action\""},{"language":"bash","snippet":"node scripts/log-learning.mjs error \"Summary\" \"Error details\" \"Suggested fix\""},{"language":"bash","snippet":"node scripts/log-learning.mjs feature \"Capability name\" \"User context\" \"Suggested implementation\""},{"language":"bash","snippet":"node scripts/log-experiment.mjs \"Target problem\" \"Baseline failure\" \"Single mutation\" \"eval1|eval2|eval3\" \"Result summary\" \"testing\""}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: openclaw-self-improvement\ndescription: A reusable operator-guided workflow improvement skill for OpenClaw and ClawLite that turns repeated failures into logged learnings, binary eval loops, SOPs, checklists, and proof-based operational improvements.\nmetadata:\n  {\n    \"openclaw\":\n      {\n        \"requires\": { \"bins\": [\"node\"] },\n        \"writes\": [\".learnings/\", \"memory/harness-backlog-latest.md\", \"mission-control/data/delivery-receipts/agent-scorecard-YYYY-MM-DD.md\", \"AGENTS.md\", \"TOOLS.md\", \"SOUL.md\"],\n        \"env\": [\"WORKSPACE\", \"OBSIDIAN_LEARNINGS_DIR\"],\n        \"network\": false,\n        \"notes\": \"Local-file workflow only. Promotion writes should be reviewed and can be previewed with --dry-run.\"\n      }\n  }\n---\n\n# OpenClaw / ClawLite Self-Improvement\n\nUse this skill to turn mistakes, corrections, blockers, and better approaches into durable operating knowledge.\n\n## What problem this solves\nAI ops often repeat the same failures because mistakes stay in chat history instead of becoming system rules. This skill creates a lightweight improvement loop:\n- log failures and learnings\n- separate errors from feature requests\n- run small eval-driven experiments on repeated failures\n- classify harness/runtime failures instead of blaming vague “model issues”\n- generate daily agent scorecards from real evidence chains\n- promote important patterns into AGENTS.md / TOOLS.md / SOUL.md\n- write operator notes into Obsidian vault\n- support stricter acceptance via Karen / Mission Control\n\n## When to use\nUse this skill when the user asks:\n- \"make the agent improve itself\"\n- \"capture learnings\"\n- \"log mistakes so we do not repeat them\"\n- \"record blockers / corrections / feature gaps\"\n- \"build a self-improving OpenClaw workflow\"\n- \"operationalize lessons learned\"\n- \"test whether this new rule actually helps\"\n- \"run an eval loop on this workflow/skill/SOP\"\n- \"should we keep this new guardrail or discard it\"\n- \"why did the agents fail today\"\n- \"why is daily marketing not closing automatically\"\n- \"classify OpenClaw harness failures\"\n- \"generate agent delivery scorecard\"\n\n## Files this skill uses\n- `.learnings/LEARNINGS.md`\n- `.learnings/ERRORS.md`\n- `.learnings/FEATURE_REQUESTS.md`\n- `.learnings/EXPERIMENTS.md`\n- `memory/harness-backlog-latest.md`\n- `mission-control/data/delivery-receipts/agent-scorecard-YYYY-MM-DD.md`\n- Optional export under `.learnings/exports/obsidian/` by default, or `OBSIDIAN_LEARNINGS_DIR` if explicitly configured\n\n## Safety boundaries\n- Local-file workflow only, no network I/O\n- Promotion can append to `AGENTS.md`, `TOOLS.md`, or `SOUL.md`\n- Always review promotion targets first, or run `scripts/promote-learning.mjs ... --dry-run`\n- `OBSIDIAN_LEARNINGS_DIR` should only point at a path you intend to modify\n\n## Command examples\n```bash\nnode {baseDir}/scripts/log-learning.mjs learning \"Summary\" \"Details\" \"Suggested action\"\nnode {baseDir}/scripts/log-learning.mjs error \"Summary\" \"Error details\" \"Suggested fix\"\nnode {baseDir}/scripts/log-le"},{"path":"README.md","content":"# OpenClaw Self-Improvement\n\n**OpenClaw Self-Improvement** is a reusable agent skill for turning repeated AI-agent mistakes into durable operational improvements, measurable guardrails, and inspectable workflow upgrades.\n\n## Landing-page summary\n\nMost AI agents do not really improve. They repeat mistakes, hide partial failures behind optimistic language, and leave lessons trapped in chat history.\n\nOpenClaw Self-Improvement gives you a practical operating loop for **self-improving AI agents**:\n- capture repeated failures\n- test one guardrail at a time\n- verify whether it actually reduces failure\n- promote proven fixes into SOPs, checklists, policies, and workflow rules\n\nIf you want **AI agents that get more reliable over time**, **multi-agent workflows that stop repeating the same mistakes**, or **proof-based QA for agent operations**, this skill is built for that exact use case.\n\n## Multilingual summary\n\n**中文：** 这是一个面向 AI agent 自我改进的实战型 skill，用来减少重复犯错、建立 guardrails、验证修复是否真的有效，并把经验沉淀成 SOP、检查清单和可复用规则。\n\n**日本語：** これは自己改善する AI エージェント向けの実践的な skill です。繰り返し発生する失敗を減らし、ガードレールを検証し、改善を SOP・チェックリスト・再利用可能な運用ルールへ昇格させます。\n\n**한국어：** 이 스킬은 스스로 개선하는 AI 에이전트를 위한 운영형 skill입니다. 반복 실수를 줄이고, 가드레일이 실제로 효과가 있는지 검증하며, 개선 사항을 SOP·체크리스트·재사용 가능한 규칙으로 승격합니다.\n\n**Español：** Esta skill está diseñada para agentes de IA que deben mejorar con el tiempo. Ayuda a reducir errores repetidos, validar guardrails operativos y convertir mejoras en SOP, checklists y reglas reutilizables.\n\nIf you want a practical way to deploy OpenClaw with cheaper tokens, BYOK flexibility, and operator control, see **[ClawLite](https://clawlite.ai)**.\n\n## TL;DR\n\nIf you are looking for a practical system for **self-improving AI agents**, **AI workflow optimization**, **multi-agent failure prevention**, **binary eval loops**, or **agent operations QA**, this skill is designed for that exact job.\n\nIt helps OpenClaw / ClawLite operators and agent teams:\n- log recurring failures\n- separate one-off errors from reusable lessons\n- run lightweight **binary eval loops** on new guardrails\n- classify changes as **keep**, **partial_keep**, or **discard**\n- promote proven fixes into SOPs, checklists, workflow rules, and operating policy\n\nIf you care about reducing fake-complete states, tightening QA truth, improving deploy closeout, and making agent learning inspectable, this skill is built for that job.\n\n---\n\n## Why this skill exists\n\nMany AI systems say they \"learn,\" but most only store lessons in chat history or loose notes.\n\nThat is not enough.\n\nOperationally, repeated failures tend to come back in the same forms:\n- delivery gets described as complete before proof exists\n- receipts are missing or too thin\n- back-end fixes never reach the operator-facing surface\n- code-ready states get confused with production-ready states\n- teams add new rules without checking whether those rules actually reduce failure\n\nOpenClaw Self-Improvement gives you a lightweight operating loop for fixing that.\n\n---\n\n## What problem it solves\n\nT"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7d88952ey3hbm158x7ejqs1d81zmyq\",\n  \"slug\": \"openclaw-self-improvement\",\n  \"version\": \"0.2.11\",\n  \"publishedAt\": 1777623237502\n}"},{"path":"references/decision-rules.md","content":"# Decision Rules for Self-Improvement\n\nUse this reference to decide whether a new issue should become:\n- a simple learning entry\n- an experiment with binary evals\n- a promoted operating rule\n\n---\n\n## Option 1 — Log only\n\nUse **log only** when:\n- the issue happened once\n- root cause is still unclear\n- there is not enough evidence yet to turn it into a rule\n- the lesson is useful, but not broadly reusable yet\n\nTypical output:\n- `learning`\n- `error`\n- `feature`\n\nExamples:\n- one-off API outage\n- first-time tool glitch with unclear cause\n- user preference that does not affect system-wide ops\n\n---\n\n## Option 2 — Run an experiment\n\nUse an **experiment** when:\n- the same failure happened 2+ times\n- a new guardrail/SOP/checklist/schema change is being proposed\n- you can define 3-5 binary evals\n- you want evidence that the change helped before promoting it broadly\n\nTypical output:\n- `experiment`\n- baseline + mutation + binary evals + keep/discard decision\n\nExamples:\n- Mission Control summaries repeatedly missing links/details\n- deploy closeout repeatedly confusing code-ready with live\n- repeated missing receipts or incomplete proof bundles\n- repeated front-end / operator-surface mismatch after backend fixes\n\n---\n\n## Option 3 — Promote immediately\n\nUse **promote immediately** when:\n- the rule is already obviously correct and low-risk\n- the issue is severe enough that waiting would be irresponsible\n- the required change is a principle or ownership rule, not an uncertain optimization\n- operator review already confirms the new rule should become standard\n\nTypical output:\n- promotion into `AGENTS.md`, `TOOLS.md`, `SOUL.md`, or `docs/ops/*.md`\n\nExamples:\n- deployment owner must be explicit\n- code-ready is not the same as live\n- missing receipt cannot be treated as delivered\n- summary without proof links is not operator-complete\n\n---\n\n## Promote after experiment\n\nUse **experiment first, then promote** when:\n- the rule sounds plausible but may add friction\n- you are not sure whether the added checklist/schema field actually reduces errors\n- the change could create process overhead without improving truth quality\n\nExamples:\n- adding new summary schema fields\n- adding new receipt requirements\n- adding extra verification steps to handoff or QA lanes\n\n---\n\n## Anti-patterns\n\nDo **not** run an experiment when:\n- there is no clear repeated failure\n- the evals would be vague or subjective\n- the issue is really just missing execution, not missing learning\n- the fix requires immediate owner action, not more analysis\n\nDo **not** promote immediately when:\n- the rule is still based on one anecdote\n- the change is likely to create bureaucracy without proof of benefit\n- the actual failure surface is still unclear\n\n---\n\n## Quick decision tree\n\n1. Did this happen only once?\n- Yes → log only\n- No → continue\n\n2. Is the new rule obviously necessary and low-risk?\n- Yes → promote immediately\n- No → continue\n\n3. Can you define 3-5 binary evals for the proposed change?\n- Yes → run an exp"},{"path":"references/eval-loop.md","content":"# Eval Loop for Self-Improvement\n\nUse this reference when a repeated failure should become a tested operational improvement instead of only a logged lesson.\n\n## Goal\n\nDo not only ask \"what did we learn?\"\nAlso ask:\n- what is the current baseline?\n- what exact guardrail or rule changed?\n- how will we measure whether it helped?\n- should we keep or discard the change?\n\n## Use this loop for\n- repeated Mission Control wording failures\n- missing receipts / missing proof chains\n- deploy closeout failures\n- stale operator-facing surfaces\n- repeated handoff mistakes between agents\n- recurring SOP/checklist changes\n\n## 1. Define the target\n\nState one concrete thing you want to improve.\n\nExamples:\n- Hunter summary should always include concrete links and details\n- ClawLite deploy closeout should never stop at code-ready status\n- Mission Control front-end should render source links from structured fields\n\n## 2. Write 3-5 binary evals\n\nEach eval must be yes/no.\n\nExamples for summary quality:\n- Does the summary include at least one artifact path or URL?\n- Does the summary include evidence links when external proof matters?\n- Does the summary include a detail block describing what actually changed?\n- Does the summary include the next handoff or recovery action?\n- Does the operator-facing surface actually render these fields?\n\nExamples for deploy closeout:\n- Is the deployed commit hash recorded?\n- Is a deployment ref/URL recorded?\n- Was the production page or sitemap actually verified?\n- Was a structured receipt written?\n- Is the final state classified with the correct deploy-state vocabulary?\n\n## 3. Capture baseline\n\nBefore changing the rule/SOP/skill/checklist:\n- record the current failure pattern\n- record which evals currently fail\n- treat this as the baseline state\n\n## 4. Change only one thing\n\nGood changes:\n- one wording rule\n- one new checklist item\n- one schema field\n- one render mapping\n- one validation step\n\nBad changes:\n- rewriting everything at once\n- adding five new rules at once\n- changing wording and schema and code together unless absolutely required\n\n## 5. Re-check and classify\n\nAfter the single change:\n- run the same evals again\n- note which checks improved\n- decide:\n  - KEEP\n  - DISCARD\n  - PARTIAL_KEEP\n\n## 6. Promotion rule\n\nOnly promote broadly reusable changes after they pass the eval loop or after operator review confirms the change materially reduced the failure.\n\n## Suggested experiment entry format\n\n```md\n## [EXP-YYYYMMDD-XXX] experiment\n\n**Logged**: ISO-8601 timestamp\n**Priority**: medium | high | critical\n**Status**: baseline | testing | keep | discard | partial_keep\n**Area**: workflow | tools | product | growth | security | infra | ops\n\n### Target\nWhat repeated problem is being improved\n\n### Baseline\nWhat was failing before the change\n\n### Mutation\nThe single change introduced\n\n### Binary Evals\n- [ ] Eval 1\n- [ ] Eval 2\n- [ ] Eval 3\n\n### Result\nWhat improved / did not improve\n\n### Keep or Discard\nkeep | discard | partial_keep\n\n### Meta"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"A reusable operator-guided workflow improvement skill for OpenClaw and ClawLite that turns repeated failures into logged learnings, binary eval loops, SOPs,... Skill: OpenClaw Self-Improvement Owner: x-rayluan Summary: A reusable operator-guided workflow improvement skill for OpenClaw and ClawLite that turns repeated failures into logged learnings, binary eval loops, SOPs,... Tags: latest:0.2.11 Version history: v0.2.11 | 2026-05-01T08:13:57.502Z | user Add scorecard repair loop with recovery ticket generation. v0.2.10 | 2026-05-01T08:07:07.960Z | user Add harness observabi","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":907,"uniquenessScore":56,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T08:11:38.169Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T08:11:38.169Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T16:29:32.105Z","emptyReason":null},"items":[{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-10T18:48:31.762Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}