{"id":"5fc9f660-0e43-4149-933d-2e2089052b87","entityType":"agent","slug":"clawhub-leostehlik-proof-loop","name":"Proof Loop Clawhub V030","canonicalUrl":"https://www.xpersona.co/agent/clawhub-leostehlik-proof-loop","canonicalPath":"/agent/clawhub-leostehlik-proof-loop","generatedAt":"2026-10-11T07:39:53.027Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T05:34:21.767Z","emptyReason":null},"description":"Run evidence-gated coding sprints with frozen ACs, separated builder/verifier roles, and durable proof artifacts.","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.1K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s1754kcncc002avbhpwjnqgcr98728et:proof-loop","sourceUrl":"https://clawhub.ai/leostehlik/proof-loop","homepage":"https://clawhub.ai/leostehlik/skills/proof-loop","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/leostehlik/proof-loop","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/leostehlik/skills/proof-loop","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":61,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Proof Loop Clawhub V030 technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T05:34:21.767Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T05:34:21.767Z","emptyReason":null},"stars":null,"forks":null,"downloads":1145,"packageName":null,"latestVersion":"0.3.0","tractionLabel":"1.1K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T05:34:21.695Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T05:34:21.767Z","lastCrawledAt":"2026-10-11T05:34:21.695Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T05:34:21.695Z","lastVerifiedAt":null,"highlights":[{"version":"0.3.0","createdAt":"2026-09-01T01:00:03.579Z","changelog":"v0.3.0 adoption kit: README use-it-today path, docs/adoption-kit.md, compact completed proof artifact under examples/adoption-kit, and test coverage for the new done gate.","fileCount":37,"zipByteSize":35212},{"version":"0.2.4","createdAt":"2026-08-26T10:17:38.144Z","changelog":"Sync ClawHub package with GitHub maintenance refresh, README discovery polish, current skill metadata, and sanitized terminal demo asset.","fileCount":34,"zipByteSize":32873},{"version":"0.2.1","createdAt":"2026-05-24T09:59:51.995Z","changelog":"Clarify explicit activation, repo-local artifact boundaries, least-privilege execution, and approval requirements for publishing, remote validation, permissions, full-access sandboxing, and credentials.","fileCount":34,"zipByteSize":32534},{"version":"0.2.0","createdAt":"2026-05-22T17:05:10.732Z","changelog":"Shorten trigger metadata, add version metadata, and clarify when to use proof-loop.","fileCount":7,"zipByteSize":8110},{"version":"0.1.0","createdAt":"2026-05-20T23:37:24.857Z","changelog":"Initial public Proof Loop skill.","fileCount":7,"zipByteSize":8169}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s1754kcncc002avbhpwjnqgcr98728et:proof-loop","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s1754kcncc002avbhpwjnqgcr98728et:proof-loop` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/leostehlik/proof-loop before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-leostehlik-proof-loop/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-leostehlik-proof-loop/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-leostehlik-proof-loop/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-leostehlik-proof-loop/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-leostehlik-proof-loop/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-leostehlik-proof-loop/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T07:39:53.024Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-leostehlik-proof-loop/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-leostehlik-proof-loop/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-leostehlik-proof-loop/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-leostehlik-proof-loop/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T05:34:21.767Z","emptyReason":null},"readme":"Skill: Proof Loop Clawhub V030\n\nOwner: leostehlik\n\nSummary: Run evidence-gated coding sprints with frozen ACs, separated builder/verifier roles, and durable proof artifacts.\n\nTags: agents:0.2.4, coding:0.2.4, latest:0.3.0, openclaw:0.2.4, verification:0.2.4\n\nVersion history:\n\nv0.3.0 | 2026-09-01T01:00:03.579Z | user\n\nv0.3.0 adoption kit: README use-it-today path, docs/adoption-kit.md, compact completed proof artifact under examples/adoption-kit, and test coverage for the new done gate.\n\nv0.2.4 | 2026-08-26T10:17:38.144Z | user\n\nSync ClawHub package with GitHub maintenance refresh, README discovery polish, current skill metadata, and sanitized terminal demo asset.\n\nv0.2.1 | 2026-05-24T09:59:51.995Z | user\n\nClarify explicit activation, repo-local artifact boundaries, least-privilege execution, and approval requirements for publishing, remote validation, permissions, full-access sandboxing, and credentials.\n\nv0.2.0 | 2026-05-22T17:05:10.732Z | user\n\nShorten trigger metadata, add version metadata, and clarify when to use proof-loop.\n\nv0.1.0 | 2026-05-20T23:37:24.857Z | user\n\nInitial public Proof Loop skill.\n\nArchive index:\n\nArchive v0.3.0: 37 files, 35212 bytes\n\nFiles: assets/proof-loop-terminal-demo.svg (14074b), bin/proof-loop (9901b), bin/proof-loop-check (139b), bin/proof-loop-init (138b), CHANGELOG.md (333b), docs/adoption-kit.md (2176b), examples/adoption-kit/README.md (566b), examples/demo-repo/check_nav_labels.py (826b), examples/demo-repo/nav_labels.json (193b), examples/demo-repo/run_demo.py (1748b), examples/example-task/README.md (557b), examples/README.md (840b), examples/role-briefs/builder.md (514b), examples/role-briefs/fixer.md (505b), examples/role-briefs/orchestrator.md (658b), examples/role-briefs/spec-freezer.md (513b), examples/role-briefs/verifier.md (648b), LICENSE (1068b), Makefile (974b), README.md (13416b), references/artifacts.md (2186b), references/brief-template.md (3128b), references/loopsmith-bridge.md (2535b), references/workflow.md (2986b), schemas/evidence.schema.json (328b), schemas/spec.schema.json (312b), schemas/verdict.schema.json (440b), scripts/check_task.py (2887b), scripts/init_task.py (3445b), skill-card.md (2290b), SKILL.md (3476b), templates/AGENTS.proof-loop.md (237b), templates/CLAUDE.proof-loop.md (189b), templates/hermes.proof-loop.md (183b), templates/opencode.proof-loop.md (161b), tests/test_cli.py (7272b), _meta.json (129b)\n\nFile v0.3.0:SKILL.md\n\n---\nname: proof-loop\ndescription: \"Run evidence-gated coding sprints with frozen ACs, separated builder/verifier roles, and durable proof artifacts.\"\nmetadata:\n  version: \"0.3.0\"\n---\n# Proof Loop\n\nA sprint is not done until every acceptance criterion has a PASS verdict from a fresh verifier session.\n\nRead `references/workflow.md` for the full loop spec.\nRead `references/brief-template.md` for the agent brief format.\nRead `references/artifacts.md` for the artifact schema.\nRead `references/loopsmith-bridge.md` when deciding whether a repeated Proof Loop failure should become a Loopsmith eval case.\n\n\n## Activation and Safety Boundaries\n\nUse this skill only when the user explicitly asks for Proof Loop, proof artifacts, acceptance-criterion verification, fresh verifier separation, or an evidence-gated coding sprint. Do not activate it for ordinary code edits where the user did not request this protocol.\n\nProof Loop may create or update files under `.agent/tasks/<TASK_ID>/` in the current repository. Confirm the task id and repository root before creating artifacts. Do not publish artifacts, run remote validation, change repository permissions, moderate users, request full-access sandboxing, or touch credentials unless the user explicitly asks for that separate action in the current conversation.\n\nRun helper scripts with the least privilege available. If a command could modify source files outside `.agent/tasks/<TASK_ID>/`, ask first and record the command in the evidence artifact.\n\n## The Loop\n\n```\nspec freeze -> build -> evidence -> FRESH verify -> fix -> FRESH verify\n                                         ^                      |\n                                         |______________________|\n                                         (repeat until all ACs = PASS)\n```\n\n## Four Roles — Always Separate\n\n| Role | Does | Never |\n|------|------|-------|\n| **Spec-Freezer** | Writes spec.md with explicit ACs | Edits production code |\n| **Builder** | Implements against frozen spec | Verifies own work |\n| **Verifier** | Fresh session — verdicts each AC | Edits production code |\n| **Fixer** | Minimal fix for what verifier flagged | Signs off on completion |\n\n**The verifier is always a fresh session.** The agent that built cannot judge its own work.\n\n## Acceptance Criteria Format\n\nEvery sprint brief must include explicit ACs before build starts:\n\n```\nAC1: [specific, testable condition — not a task description]\nAC2: [specific, testable condition]\nAC3: [specific, testable condition]\n```\n\nGood: \"AC1: A German-locale user sees all prompt form field labels in German\"\nBad: \"AC1: Translate the form fields\"\n\n## Helper Scripts\n\nUse these when the repository has the `proof-loop` folder available:\n\n```bash\npython3 scripts/init_task.py TASK_ID --title \"Task title\"\npython3 scripts/check_task.py .agent/tasks/TASK_ID\n```\n\n`check_task.py` is the mechanical done gate. It returns success only when the verifier artifacts show every AC as PASS and no open problems remain.\n\n## Sprint is DONE Only When\n\n- Every AC has a PASS verdict in the verifier's `verdict.json`\n- No problems remain in `problems.md`\n- Full regression suite passes (if applicable)\n\n## Artifacts (stored in repo)\n\n```\n.agent/tasks/<TASK_ID>/\n  spec.md         -- frozen ACs + constraints + non-goals\n  verdict.json    -- AC verdicts per phase (PASS/FAIL/UNKNOWN)\n  problems.md     -- specific failures with file/line refs (if any)\n```\n\nSee `references/artifacts.md` for schemas.\n\nFile v0.3.0:examples/adoption-kit/README.md\n\n# Adoption Kit Example\n\nThis is the smallest completed proof folder a repo owner can copy when teaching an agent the Proof Loop shape.\n\nRead:\n\n1. `.agent/tasks/checkout-empty-state-proof/spec.md`\n2. `.agent/tasks/checkout-empty-state-proof/evidence.md`\n3. `.agent/tasks/checkout-empty-state-proof/verdict.json`\n4. `.agent/tasks/checkout-empty-state-proof/problems.md`\n\nRun:\n\n```bash\nbin/proof-loop check examples/adoption-kit/.agent/tasks/checkout-empty-state-proof\nbin/proof-loop report examples/adoption-kit/.agent/tasks/checkout-empty-state-proof --format md\n```\n\nFile v0.3.0:examples/example-task/README.md\n\n# Example Task: UI Language Fix\n\nThis example shows what a completed Proof Loop task looks like inside a repository.\n\nThe task artifacts live under:\n\n```text\n.agent/tasks/ui-language-fix/\n```\n\nRead them in this order:\n\n1. `spec.md` - frozen acceptance criteria and verification plan\n2. `evidence.md` - what was changed and checked\n3. `verdict.json` - structured verifier result\n4. `problems.md` - empty because the final verifier pass found no open issues\n\nThe example is intentionally small. Its job is to prove the artifact shape, not to ship a real app.\n\nFile v0.3.0:examples/README.md\n\n# Proof Loop Examples\n\nStart here if you want proof instead of prose.\n\n| Example | What it shows |\n| --- | --- |\n| [`demo-repo/`](demo-repo/) | A small runnable task with a real file check and passing proof artifacts. |\n| [`adoption-kit/`](adoption-kit/) | A compact completed task folder that shows the copy-paste adoption shape. |\n| [`example-task/`](example-task/) | A completed task folder with spec, verdict, evidence, and problems files. |\n| [`role-briefs/`](role-briefs/) | Copy-paste briefs for orchestrator, spec freezer, builder, verifier, and fixer roles. |\n\nFast path:\n\n```bash\nmake test\nbin/proof-loop check examples/demo-repo/.agent/tasks/nav-labels-proof\nbin/proof-loop check examples/adoption-kit/.agent/tasks/checkout-empty-state-proof\nbin/proof-loop report examples/demo-repo/.agent/tasks/nav-labels-proof --format md\n```\n\nFile v0.3.0:README.md\n\n# Proof Loop\n\n![Tests](https://github.com/LeoStehlik/proof-loop/actions/workflows/test.yml/badge.svg)\n\n**Make AI coding agents prove when work is done.**\n\n**v0.3 focus:** use Proof Loop in a real repo today: copy the adoption kit, freeze ACs, run a fresh verifier, and ship with durable proof artifacts instead of a completion story.\n\nProof Loop is a repo-local verification protocol for AI coding agents. It freezes acceptance criteria before the build, separates builder and verifier roles, records durable proof artifacts in the repo, and refuses to call work done until every acceptance criterion has a fresh PASS verdict.\n\nUse it when an agent, team, or multi-agent sprint needs a clear boundary between “looks done” and verified work. Because the protocol is just files plus role discipline, it works with OpenClaw, Hermes, Codex, OpenCode, Claude Code, or any other harness that can read and write a repository.\n\n## Start Here\n\nRun the current proof gate first:\n\n```bash\ngit clone https://github.com/LeoStehlik/proof-loop.git\ncd proof-loop\nmake test\nbin/proof-loop doctor\n```\n\nStart a proof-tracked task in any repo:\n\n```bash\nbin/proof-loop init maintenance-check --title \"Refresh the README proof\" --root /path/to/repo\nbin/proof-loop check /path/to/repo/.agent/tasks/maintenance-check\n```\n\nThe first check should fail until a fresh verifier records `PASS` for every frozen acceptance criterion and clears `problems.md`. That failure is the point: Proof Loop gives agents a mechanical done gate instead of a confident paragraph.\n\n## Works With\n\nProof Loop is harness-agnostic. Use it with any coding agent that can read files, write files, run commands, and hand verification to a fresh session.\n\nKnown-fit surfaces:\n\n- Codex\n- Claude Code\n- OpenClaw\n- OpenCode\n- Hermes\n- custom multi-agent runners\n\nThe repo includes copy-paste guide templates for several harnesses under `templates/`, plus role briefs under `examples/role-briefs/`.\n\n\n## Activation and Safety\n\nUse Proof Loop when the user explicitly wants an evidence-gated coding sprint, frozen acceptance criteria, fresh verifier separation, or durable proof artifacts. It is not meant to silently wrap every code change.\n\nThe bundled helpers create and check repo-local files under `.agent/tasks/<TASK_ID>/`. Review the task id and repository root before running them. Publishing reports, using remote workers, changing permissions, or running elevated commands are outside this skill unless requested separately.\n\n## Use Cases\n\n- keep AI coding agents honest when they claim a task is done\n- freeze acceptance criteria before implementation starts\n- separate builder and verifier roles in multi-agent coding work\n- leave proof artifacts in the repo for future review\n\n![Animated terminal demo: Proof Loop doctor, check, and report commands](assets/proof-loop-terminal-demo.svg)\n\nProof artifacts and role-brief examples are indexed in [`examples/README.md`](examples/README.md).\n\n## Use It Today\n\nDrop Proof Loop into a task where an agent is about to edit code:\n\n```bash\ngit clone https://github.com/LeoStehlik/proof-loop.git\ncd proof-loop\nmake test\nbin/proof-loop init checkout-empty-state --title \"Fix checkout empty state\" --root /path/to/your/repo\n```\n\nThen freeze `.agent/tasks/checkout-empty-state/spec.md` before implementation starts. After the builder runs, a fresh verifier writes the final `verdict.json`, clears `problems.md`, and the mechanical gate decides whether the work can be called done:\n\n```bash\nbin/proof-loop check /path/to/your/repo/.agent/tasks/checkout-empty-state\nbin/proof-loop report /path/to/your/repo/.agent/tasks/checkout-empty-state --format md\n```\n\nFor copy-paste prompts and a completed repo-local proof folder, start with [`docs/adoption-kit.md`](docs/adoption-kit.md) and [`examples/adoption-kit/`](examples/adoption-kit/).\n\n## 20-second demo\n\n```bash\ngit clone https://github.com/LeoStehlik/proof-loop.git\ncd proof-loop\nmake test\n\ntmp=$(mktemp -d)\nbin/proof-loop-init hn-demo --title \"Prove this task before done\" --root \"$tmp\"\nbin/proof-loop-check \"$tmp/.agent/tasks/hn-demo\"\n```\n\nThe last command fails on purpose because the generated task has not been verified yet. Proof Loop only returns success after a fresh verifier records `PASS` for every acceptance criterion and `problems.md` is empty.\n\nA completed passing example is included:\n\n```bash\nbin/proof-loop-check examples/example-task/.agent/tasks/ui-language-fix\nbin/proof-loop doctor\nbin/proof-loop report examples/demo-repo/.agent/tasks/nav-labels-proof --format md\n```\n\n## Why It Exists\n\nAI coding agents often fail in predictable ways:\n\n- they claim completion without durable proof\n- the same session builds and judges its own work\n- acceptance criteria drift while implementation is underway\n- verification is a prose summary instead of a live check\n- future sessions cannot tell what was actually tested\n\nProof Loop makes completion auditable. A task is done only when a fresh verifier has checked each AC and the repo contains the artifacts to prove it.\n\n## What You Get\n\n- a clear sprint protocol: spec freeze -> build -> evidence -> fresh verify -> fix loop\n- role boundaries for orchestrator, spec-freezer, builder, verifier, and fixer\n- helper scripts to initialize and check task proof folders\n- a complete example task with passing artifacts\n- copy-paste role briefs for OpenClaw, Hermes, Codex, OpenCode, Claude Code, or any agent setup\n- a documented boundary with Loopsmith for recurring behaviour improvement\n\n## CLI\n\n```bash\nbin/proof-loop init TASK_ID --title \"Task title\"\nbin/proof-loop check TASK_ID\nbin/proof-loop status TASK_ID\nbin/proof-loop list\nbin/proof-loop doctor\nbin/proof-loop report TASK_ID --format md\nbin/proof-loop install-guides --dry-run --harness codex --harness claude\n```\n\n## Quick Start\n\nClone the repo or copy it into the project where you want to run the protocol.\n\nCreate a task proof folder from this repo or from another repository:\n\n```bash\nbin/proof-loop-init ui-language-fix --title \"Fix German navigation labels\" --root .\n```\n\nThis creates:\n\n```text\n.agent/tasks/ui-language-fix/\n  spec.md\n  verdict.json\n  problems.md\n  evidence.md\n```\n\nFill `spec.md` with explicit acceptance criteria before implementation starts.\n\nAfter the build and verifier pass, check whether the task is allowed to be called done:\n\n```bash\nbin/proof-loop-check .agent/tasks/ui-language-fix\n```\n\nThe check exits non-zero unless:\n\n- `verdict.json` has `overall: PASS`\n- every AC has `status: PASS`\n- `problems.md` is empty or absent\n\n\n## What This Is Not\n\n- not an agent framework\n- not a benchmark suite\n- not a replacement for tests\n- not tied to one model, vendor, or harness\n\nProof Loop is deliberately small: a protocol, a few files, and a mechanical done gate.\n\n## The Protocol\n\n```text\nspec freeze -> build -> evidence -> fresh verify -> fix -> fresh verify\n                                         ^                    |\n                                         |____________________|\n                                      repeat until all ACs PASS\n```\n\n## Roles\n\n| Role | Does | Never |\n|---|---|---|\n| Orchestrator | Keeps the loop intact and refuses weak completion | Accepts narrative-only proof |\n| Spec-Freezer | Writes frozen `spec.md` with explicit ACs | Edits production code |\n| Builder | Implements against the frozen spec | Verifies own work as final |\n| Verifier | Fresh session that checks each AC | Edits production code |\n| Fixer | Applies minimal fixes for verifier findings | Signs off on completion |\n\nThe verifier must be a fresh session. The agent that built the change does not judge whether the change is done.\n\n## Acceptance Criteria\n\nGood ACs are specific and testable by a third party.\n\n```text\nAC1: A user with locale=de sees all navigation labels in German after saving language preference.\n     Verify: browser check against a German-locale test user.\n\nAC2: The language preference survives page reload.\n     Verify: reload the page and confirm the saved locale and labels remain German.\n\nAC3: Existing English navigation remains unchanged for locale=en.\n     Verify: switch back to English and confirm the original labels render.\n```\n\nWeak ACs are task descriptions, not proof conditions:\n\n```text\nAC1: Translate the UI.\nAC2: Make language switching work.\nAC3: Fix the bugs.\n```\n\n## Artifacts\n\nEvery task stores proof under `.agent/tasks/<TASK_ID>/`.\n\n```text\n.agent/tasks/<TASK_ID>/\n  spec.md       frozen ACs, constraints, non-goals, verification approach\n  evidence.md   build summary and checks run\n  verdict.json  structured verifier result: PASS / FAIL / UNKNOWN per AC\n  problems.md   specific open failures, empty when no problems remain\n```\n\nSee [`references/artifacts.md`](references/artifacts.md) for schemas.\n\n## Real Demo\n\nRun a small failing-to-passing demo:\n\n```bash\nmake demo\n```\n\nThe demo intentionally breaks a tiny navigation-label fixture, shows the check failing, applies the fix, reruns the check, and renders a proof report.\n\n## Examples\n\nA complete passing example lives at:\n\n```text\nexamples/example-task/.agent/tasks/ui-language-fix/\n```\n\nRole prompts live at:\n\n```text\nexamples/role-briefs/\n  orchestrator.md\n  spec-freezer.md\n  builder.md\n  verifier.md\n  fixer.md\n```\n\n## Proof Loop vs Loopsmith\n\nProof Loop governs a single task.\n\nLoopsmith improves repeated agent behaviour over time.\n\nUse Proof Loop when you need a specific task to finish with evidence. Use [Loopsmith](https://github.com/LeoStehlik/loopsmith) when the same failure pattern keeps coming back and you want to improve the agent, prompt, policy, or evaluator itself.\n\nSee [`references/loopsmith-bridge.md`](references/loopsmith-bridge.md).\n\n\n## When To Use Which Repo\n\nUse this repo when a specific coding task needs evidence before anyone is allowed to call it done. Proof Loop freezes the spec, separates builder and verifier roles, requires proof artifacts, and records verdicts in the repo.\n\nUse the neighbouring tools at different points in the workflow:\n\n| Need | Use |\n| --- | --- |\n| Turn a fuzzy request into an executable agent brief | [Brief Master](https://github.com/LeoStehlik/brief-master) |\n| Prove one coding task is actually done | [Proof Loop](https://github.com/LeoStehlik/proof-loop) |\n| Improve repeated agent behaviour with evals | [Loopsmith](https://github.com/LeoStehlik/loopsmith) |\n| Keep source-backed memory for long-running agents | [Sovereign Brain](https://github.com/LeoStehlik/decoupled-agent-memory) |\n| Stop frontend agents producing generic UI sludge | [no-slop-ui](https://github.com/LeoStehlik/no-slop-ui) |\n\nA practical chain looks like this: messy request -> Brief Master brief -> Proof Loop task -> Loopsmith eval if the same failure keeps recurring -> Sovereign Brain records the durable decision.\n\n## Related Tools\n\n- [Loopsmith](https://github.com/LeoStehlik/loopsmith) - use when Proof Loop exposes a repeated agent behaviour problem that should become an eval and promotion loop.\n- [Sovereign Brain](https://github.com/LeoStehlik/decoupled-agent-memory) - source-backed memory for long-running agents; useful when proof artifacts, decisions, and synthesis need durable context.\n- [Brief Master](https://github.com/LeoStehlik/brief-master) - helps write sharper task briefs and acceptance criteria before a Proof Loop starts.\n\n## Installation As A Skill\n\n### OpenClaw\n\nAdd your skills directory to `openclaw.json`:\n\n```json\n{\n  \"skills\": {\n    \"load\": {\n      \"extraDirs\": [\"/path/to/your/skills\"]\n    }\n  }\n}\n```\n\nClone this repo into that directory:\n\n```bash\ngit clone https://github.com/LeoStehlik/proof-loop.git /path/to/your/skills/proof-loop\n```\n\n### Codex / Claude Code\n\nCopy the `proof-loop` folder into your agent skills directory, or reference `SKILL.md` directly in your task brief. For harnesses without a formal skill system, use the README, role briefs, and scripts directly from the repo.\n\n## Repository Map\n\n```text\nproof-loop/\n  SKILL.md                         skill trigger and core operating rules\n  bin/\n    proof-loop                     unified CLI\n    proof-loop-init                compatibility wrapper\n    proof-loop-check               compatibility wrapper\n  scripts/\n    init_task.py                   create .agent/tasks/<TASK_ID>/ skeletons\n    check_task.py                  mechanical done gate\n  schemas/                         JSON schemas for verdict and evidence bundles\n  templates/                       opt-in harness guide templates\n  tests/                           stdlib unittest coverage for CLI behavior\n  .github/workflows/test.yml       CI running make test\n  references/\n    workflow.md                    full phase-by-phase protocol\n    brief-template.md              reusable sprint and role prompts\n    artifacts.md                   artifact schemas\n    loopsmith-bridge.md            when to escalate repeated failures to Loopsmith\n  examples/\n    example-task/                  complete passing proof artifact example\n    role-briefs/                   copy-paste role prompts\n```\n\n## Status\n\nUsable protocol skill and small toolkit. The scripts are intentionally stdlib-only so they can run inside almost any repository without packaging ceremony.\n\n## License\n\nMIT - see [LICENSE](LICENSE).\n\n## Attribution\n\nInspired by [`repo-task-proof-loop`](https://github.com/DenisSergeevitch/repo-task-proof-loop), adapted for practical multi-agent coding work and public agent-operation skills.\n\nFile v0.3.0:_meta.json\n\n{\n  \"ownerId\": \"kn7d3r58cdxk8k0xg6jq6k65gs873c0c\",\n  \"slug\": \"proof-loop\",\n  \"version\": \"0.3.0\",\n  \"publishedAt\": 1788224403579\n}\n\nFile v0.3.0:references/artifacts.md\n\n# Artifact Schemas\n\nAll artifacts live in the repository under `.agent/tasks/<TASK_ID>/`.\n\n---\n\n## spec.md\n\nPlain Markdown. Written by Spec-Freezer before build starts. Never modified after freeze.\n\n```markdown\n# Task: [TASK_ID]\n\n## Task Statement\n[Original task description]\n\n## Acceptance Criteria\n\n**AC1:** [testable condition]\n- Verify: [how to check]\n\n**AC2:** [testable condition]\n- Verify: [how to check]\n\n## Constraints\n- [what must not break]\n\n## Non-Goals\n- [what is out of scope]\n\n## Verification Approach\n[Overall approach to verifying this task]\n```\n\n---\n\n## verdict.json\n\nWritten by Verifier. Updated after each fix loop iteration.\n\n```json\n{\n  \"task_id\": \"sprint-4c\",\n  \"phase\": \"verify\",\n  \"agent\": \"verifier\",\n  \"timestamp\": \"2026-03-30T14:00:00Z\",\n  \"overall\": \"FAIL\",\n  \"criteria\": [\n    {\n      \"id\": \"AC1\",\n      \"status\": \"PASS\",\n      \"note\": \"All nav labels confirmed German in browser test\"\n    },\n    {\n      \"id\": \"AC2\",\n      \"status\": \"FAIL\",\n      \"note\": \"Form field labels still in English — formSchemaTranslated is null in DB\"\n    },\n    {\n      \"id\": \"AC3\",\n      \"status\": \"PASS\",\n      \"note\": \"pnpm test:e2e — all sprint 1-4c specs green\"\n    }\n  ]\n}\n```\n\n**overall** is PASS only when every AC status is PASS.\n**status values:** PASS | FAIL | UNKNOWN\n\n---\n\n## problems.md\n\nWritten by Verifier when overall is not PASS. Specific, actionable.\n\n```markdown\n# Problems — [TASK_ID]\n\n## AC2: Form field labels in English\n\n**File:** `packages/llm/src/translation.ts`\n**Function:** `translatePrompts()`\n**Issue:** All prompt chunks translated in parallel via `Promise.all()`. Under slow local\nmodels, all calls time out simultaneously and fall back to English silently.\n**Evidence:** DB query shows `formSchemaTranslated IS NULL` for all 50 rows.\n**Fix needed:** Change parallel chunk processing to sequential — one prompt at a time.\n```\n\n---\n\n## evidence.md (optional)\n\nProse summary written by Builder in evidence mode. Supplements verdict.json.\n\n```markdown\n# Evidence — [TASK_ID]\n\n## Build Summary\n[What was changed and why]\n\n## AC1 — PASS\n[How it was verified, what was checked]\n\n## AC2 — FAIL\n[What was attempted, what failed, why]\n```\n\nFile v0.3.0:references/brief-template.md\n\n# Agent Brief Template\n\nUse this template when briefing any coding agent on a non-trivial task.\n\n---\n\n## Sprint/Task Brief: [TASK_ID]\n\n### Task Statement\n[One paragraph describing what needs to be built and why]\n\n### Acceptance Criteria\n\n```\nAC1: [specific, testable condition]\n     Verify: [how to check this — command, visual check, API call]\n\nAC2: [specific, testable condition]\n     Verify: [how to check this]\n\nAC3: [specific, testable condition]\n     Verify: [how to check this]\n```\n\n**Good AC examples:**\n- \"AC1: A user with locale=de sees all navigation labels in German after saving language preference\"\n- \"AC2: POST /api/v1/prompts/translate/de returns 200 with translated titles for all 50 prompts\"\n- \"AC3: `pnpm test:e2e` passes all sprint spec files with no failures\"\n\n**Bad AC examples:**\n- \"AC1: Translate the UI\" (not testable)\n- \"AC1: Make it work in German\" (vague)\n- \"AC1: Fix the translation bugs\" (not specific)\n\n### Constraints\n[What must not break — existing features, performance, API contracts]\n\n### Non-Goals\n[What is explicitly OUT of scope for this task]\n\n### Verification Approach\n[How each AC will be verified — automated tests, manual browser check, API call, etc.]\n\n---\n\n## Brief Checklist (orchestrator — run before firing agents)\n\n- [ ] ACs are explicit, testable, and frozen\n- [ ] Each AC has a verification approach\n- [ ] Constraints are listed\n- [ ] Non-goals are listed\n- [ ] Builder role is clear (who builds)\n- [ ] Verifier role is clear (different agent, fresh session)\n- [ ] Fixer role is clear (if needed)\n- [ ] Artifact path is set: `.agent/tasks/[TASK_ID]/`\n\n---\n\n## Prompts by Role\n\n### Spec-Freezer prompt\n```\nYou are the spec freezer for task [TASK_ID].\n\nRead the brief above. Write spec.md to .agent/tasks/[TASK_ID]/spec.md with:\n- Original task statement\n- Acceptance Criteria (AC1, AC2, AC3...) — copy from brief, do not modify\n- Constraints\n- Non-goals\n- Verification approach per AC\n\nDo not edit any production code. Do not start building.\n```\n\n### Builder prompt\n```\nYou are the builder for task [TASK_ID].\n\nRead .agent/tasks/[TASK_ID]/spec.md. Implement the task against the frozen ACs.\nMake the smallest safe change set that satisfies all ACs.\nDo not verify your own work. When done, hand off to the evidence phase.\n```\n\n### Verifier prompt\n```\nYou are the verifier for task [TASK_ID]. This is a fresh session.\n\nRead:\n- .agent/tasks/[TASK_ID]/spec.md (the frozen ACs)\n- .agent/tasks/[TASK_ID]/verdict.json (current verdicts)\n- .agent/tasks/[TASK_ID]/problems.md (known problems)\n\nRun independent checks. For each AC, write your verdict: PASS / FAIL / UNKNOWN.\nUpdate verdict.json. If any AC is not PASS, update problems.md with specific file/line references.\n\nDo not edit production code. Do not sign off on completion.\n```\n\n### Fixer prompt\n```\nYou are the fixer for task [TASK_ID].\n\nRead:\n- .agent/tasks/[TASK_ID]/spec.md\n- .agent/tasks/[TASK_ID]/verdict.json\n- .agent/tasks/[TASK_ID]/problems.md\n\nFix only what the verifier identified. Apply the minimal diff. Regenerate evidence.\nDo not write final sign-off. A fresh verifier will re-run after your fix.\n```\n\nFile v0.3.0:references/loopsmith-bridge.md\n\n# Loopsmith Bridge\n\nProof Loop and Loopsmith solve different parts of the same reliability problem.\n\n## Boundary\n\nUse Proof Loop when you need one task to finish with evidence:\n\n- freeze acceptance criteria\n- separate builder and verifier roles\n- store task-local artifacts\n- block self-certified done claims\n\nUse Loopsmith when the same failure pattern keeps coming back and you want to improve the agent, prompt, policy, or evaluator itself:\n\n- compare baseline vs candidate behaviour\n- run eval packs\n- score recurring failure modes\n- promote or reject changes with a ledger\n\nShort version:\n\n> Proof Loop governs a task. Loopsmith improves the agent system over time.\n\n## When Proof Loop Is Enough\n\nProof Loop is enough when:\n\n- the task is bounded\n- the acceptance criteria are clear\n- the verifier can run concrete checks\n- failures are task-specific, not a repeated agent behaviour problem\n\nExample: a UI language bug has three ACs, a verifier checks them, and all PASS.\n\n## When To Escalate To Loopsmith\n\nEscalate to Loopsmith when Proof Loop reveals a pattern such as:\n\n- builders repeatedly claim done without evidence\n- verifiers return vague narrative instead of AC-by-AC verdicts\n- fixers patch around symptoms and miss regression checks\n- orchestrators write weak or drifting acceptance criteria\n- the same class of failure appears across multiple tasks\n\nAt that point, the task is no longer the only problem. The agent behaviour needs an eval.\n\n## Turning A Proof Loop Failure Into A Loopsmith Case\n\n1. Pick the smallest representative Proof Loop artifact set.\n2. Convert the failure into an eval case:\n   - input: the task brief, spec, evidence, verdict, or problems file\n   - expected behaviour: what a good agent should do\n   - anti-goals: what the old agent did wrong\n3. Add a baseline response that shows the current behaviour.\n4. Add a candidate policy/prompt/evaluator change.\n5. Run Loopsmith and promote only if the candidate improves the evidence.\n\n## Example Mapping\n\n| Proof Loop Artifact | Loopsmith Use |\n|---|---|\n| `spec.md` | eval input for AC quality or builder discipline |\n| `verdict.json` | structural target for verifier verdict discipline |\n| `problems.md` | input for fixer minimality and regression-awareness cases |\n| `evidence.md` | input for evidence quality or false-completion checks |\n\n## Practical Rule\n\nDo not use Loopsmith for every task. That creates ceremony.\n\nUse Proof Loop by default for non-trivial work. Use Loopsmith when a repeated failure deserves a reusable improvement loop.\n\nFile v0.3.0:references/workflow.md\n\n# Proof Loop Workflow\n\n## Phase 0: Spec Freeze\n\n**Who:** Orchestrator (project lead, or whoever is briefing)\n\nBefore a single line of code is written:\n1. Write `spec.md` with explicit acceptance criteria (AC1, AC2, AC3...)\n2. Each AC must be testable by a third party who didn't build it\n3. Include: constraints, non-goals, verification approach\n4. Freeze the spec — ACs cannot change during build\n\n**spec.md must include:**\n- Original task statement\n- Acceptance Criteria (AC1, AC2, ...)\n- Constraints (what must not break)\n- Non-goals (what is explicitly out of scope)\n- Verification approach (how each AC will be checked)\n\n---\n\n## Phase 1: Build\n\n**Who:** Builder agent (fresh session)\n\n- Reads spec.md — implements against frozen ACs only\n- Makes the smallest safe change set that satisfies all ACs\n- Does not verify own work\n- Hands off to evidence phase when implementation is complete\n\n---\n\n## Phase 2: Evidence\n\n**Who:** Builder (same session, switched to evidence mode) or a fresh evidence agent\n\n- Does not change production code\n- Runs checks and records results\n- Writes `evidence.md` (prose) and updates `verdict.json` (structured)\n- Verdict per AC: PASS / FAIL / UNKNOWN\n- If FAIL or UNKNOWN: writes `problems.md` with specific file/line references\n\n**Builder hard constraints in evidence mode:**\n- Must not patch evidence to make it look better\n- Must not change production code\n- UNKNOWN is valid — it means \"could not verify\"\n\n---\n\n## Phase 3: Fresh Verify\n\n**Who:** Verifier — always a NEW session, never the same agent that built\n\n- Reads `spec.md`, `verdict.json`, `problems.md`\n- Runs independent checks against the current codebase\n- Updates `verdict.json` with verifier's findings\n- Updates `problems.md` if issues found\n- **Must not edit production code**\n- **Must not sign off on completion**\n\n---\n\n## Phase 4: Fix (if needed)\n\n**Who:** Fixer — fresh session\n\n- Reads `spec.md` + `verdict.json` + `problems.md`\n- Reconfirms each problem before editing\n- Applies minimal fix — only what the verifier identified\n- Regenerates evidence\n- **Does not write final sign-off** — goes back to Phase 3\n\n---\n\n## The Fix Loop\n\n```\nfix -> fresh verify -> fix -> fresh verify\n```\n\nContinues until verifier writes PASS for every AC.\n\nA single successful PASS does not end the loop — ALL ACs must be PASS.\n\n---\n\n## Done Criteria\n\nA task is complete when:\n- Every AC in `spec.md` has status PASS in `verdict.json`\n- `problems.md` is empty or does not exist\n- Any regression suite required by the spec passes\n\n---\n\n## Common Failure Modes (and how this prevents them)\n\n| Failure | Prevention |\n|---------|-----------|\n| Agent claims done without checking | Verifier is separate, required |\n| ACs drift during build | Spec is frozen before build starts |\n| Later sessions can't tell what was verified | Verdict artifacts stay in repo |\n| Builder judges own work | Fresh verifier is a hard rule |\n| Fix introduces new regression | Verifier reruns after every fix |\n\nFile v0.3.0:CHANGELOG.md\n\n# Changelog\n\n## v0.3.0\n\n- Add a focused adoption kit for repo owners who want to run Proof Loop today.\n- Add a compact completed checkout empty-state proof folder under `examples/adoption-kit/`.\n- Refresh README quick path around copy-paste setup, fresh verifier handoff, and mechanical done gates.\n- Bump skill metadata to `0.3.0`.\n\nFile v0.3.0:docs/adoption-kit.md\n\n# Proof Loop Adoption Kit\n\nUse this when a coding agent is about to do work that needs a hard done gate.\n\n## 1. Create the Task Folder\n\nFrom the Proof Loop repo:\n\n```bash\nbin/proof-loop init TASK_ID --title \"One sentence task title\" --root /path/to/project\n```\n\nThis creates:\n\n```text\n.agent/tasks/TASK_ID/\n  spec.md\n  evidence.md\n  verdict.json\n  problems.md\n```\n\n## 2. Freeze the Acceptance Criteria\n\nEdit `spec.md` before the builder starts. Keep each AC specific enough that a different session can verify it.\n\n```text\nAC1: The checkout empty state shows the saved-cart recovery action when the cart has prior items.\n     Verify: run the browser check against /checkout?cart=empty-with-history.\n\nAC2: A brand-new empty cart still shows the normal continue-shopping action.\n     Verify: run the browser check against /checkout?cart=empty-new.\n\nAC3: Existing checkout totals and payment tests stay green.\n     Verify: pnpm test checkout.\n```\n\n## 3. Give the Builder a Small Brief\n\n```text\nYou are the builder for TASK_ID.\nRead .agent/tasks/TASK_ID/spec.md. Implement only the frozen ACs.\nRecord what changed and the checks you ran in evidence.md. Do not write the final verifier verdict.\n```\n\n## 4. Run a Fresh Verifier\n\n```text\nYou are the verifier for TASK_ID in a fresh session.\nRead spec.md, evidence.md, verdict.json, and problems.md. Run independent checks for each AC.\nWrite PASS / FAIL / UNKNOWN for every AC in verdict.json. If anything is not PASS, write specific failures in problems.md. Do not edit production code.\n```\n\n## 5. Use the Mechanical Done Gate\n\n```bash\nbin/proof-loop check /path/to/project/.agent/tasks/TASK_ID\nbin/proof-loop report /path/to/project/.agent/tasks/TASK_ID --format md\n```\n\nIf `check` fails, send only the verifier findings to a fixer. After the fix, run a fresh verifier again.\n\n## What Good Looks Like\n\nA finished task has:\n\n- frozen ACs in `spec.md`\n- concrete commands or inspection notes in `evidence.md`\n- `overall: PASS` and every criterion `PASS` in `verdict.json`\n- an empty `problems.md`\n- a final `proof-loop check` pass\n\nSee `examples/adoption-kit/.agent/tasks/checkout-empty-state-proof/` for a compact completed example.\n\nArchive v0.2.4: 34 files, 32873 bytes\n\nFiles: assets/proof-loop-terminal-demo.svg (14074b), bin/proof-loop (9901b), bin/proof-loop-check (139b), bin/proof-loop-init (138b), examples/demo-repo/check_nav_labels.py (826b), examples/demo-repo/nav_labels.json (193b), examples/demo-repo/run_demo.py (1748b), examples/example-task/README.md (557b), examples/README.md (644b), examples/role-briefs/builder.md (514b), examples/role-briefs/fixer.md (505b), examples/role-briefs/orchestrator.md (658b), examples/role-briefs/spec-freezer.md (513b), examples/role-briefs/verifier.md (648b), LICENSE (1068b), Makefile (748b), README.md (12353b), references/artifacts.md (2186b), references/brief-template.md (3128b), references/loopsmith-bridge.md (2535b), references/workflow.md (2986b), schemas/evidence.schema.json (328b), schemas/spec.schema.json (312b), schemas/verdict.schema.json (440b), scripts/check_task.py (2887b), scripts/init_task.py (3445b), skill-card.md (2207b), SKILL.md (3476b), templates/AGENTS.proof-loop.md (237b), templates/CLAUDE.proof-loop.md (189b), templates/hermes.proof-loop.md (183b), templates/opencode.proof-loop.md (161b), tests/test_cli.py (7272b), _meta.json (129b)\n\nFile v0.2.4:SKILL.md\n\n---\nname: proof-loop\ndescription: \"Run evidence-gated coding sprints with frozen ACs, separated builder/verifier roles, and durable proof artifacts.\"\nmetadata:\n  version: \"0.2.4\"\n---\n# Proof Loop\n\nA sprint is not done until every acceptance criterion has a PASS verdict from a fresh verifier session.\n\nRead `references/workflow.md` for the full loop spec.\nRead `references/brief-template.md` for the agent brief format.\nRead `references/artifacts.md` for the artifact schema.\nRead `references/loopsmith-bridge.md` when deciding whether a repeated Proof Loop failure should become a Loopsmith eval case.\n\n\n## Activation and Safety Boundaries\n\nUse this skill only when the user explicitly asks for Proof Loop, proof artifacts, acceptance-criterion verification, fresh verifier separation, or an evidence-gated coding sprint. Do not activate it for ordinary code edits where the user did not request this protocol.\n\nProof Loop may create or update files under `.agent/tasks/<TASK_ID>/` in the current repository. Confirm the task id and repository root before creating artifacts. Do not publish artifacts, run remote validation, change repository permissions, moderate users, request full-access sandboxing, or touch credentials unless the user explicitly asks for that separate action in the current conversation.\n\nRun helper scripts with the least privilege available. If a command could modify source files outside `.agent/tasks/<TASK_ID>/`, ask first and record the command in the evidence artifact.\n\n## The Loop\n\n```\nspec freeze -> build -> evidence -> FRESH verify -> fix -> FRESH verify\n                                         ^                      |\n                                         |______________________|\n                                         (repeat until all ACs = PASS)\n```\n\n## Four Roles — Always Separate\n\n| Role | Does | Never |\n|------|------|-------|\n| **Spec-Freezer** | Writes spec.md with explicit ACs | Edits production code |\n| **Builder** | Implements against frozen spec | Verifies own work |\n| **Verifier** | Fresh session — verdicts each AC | Edits production code |\n| **Fixer** | Minimal fix for what verifier flagged | Signs off on completion |\n\n**The verifier is always a fresh session.** The agent that built cannot judge its own work.\n\n## Acceptance Criteria Format\n\nEvery sprint brief must include explicit ACs before build starts:\n\n```\nAC1: [specific, testable condition — not a task description]\nAC2: [specific, testable condition]\nAC3: [specific, testable condition]\n```\n\nGood: \"AC1: A German-locale user sees all prompt form field labels in German\"\nBad: \"AC1: Translate the form fields\"\n\n## Helper Scripts\n\nUse these when the repository has the `proof-loop` folder available:\n\n```bash\npython3 scripts/init_task.py TASK_ID --title \"Task title\"\npython3 scripts/check_task.py .agent/tasks/TASK_ID\n```\n\n`check_task.py` is the mechanical done gate. It returns success only when the verifier artifacts show every AC as PASS and no open problems remain.\n\n## Sprint is DONE Only When\n\n- Every AC has a PASS verdict in the verifier's `verdict.json`\n- No problems remain in `problems.md`\n- Full regression suite passes (if applicable)\n\n## Artifacts (stored in repo)\n\n```\n.agent/tasks/<TASK_ID>/\n  spec.md         -- frozen ACs + constraints + non-goals\n  verdict.json    -- AC verdicts per phase (PASS/FAIL/UNKNOWN)\n  problems.md     -- specific failures with file/line refs (if any)\n```\n\nSee `references/artifacts.md` for schemas.\n\nFile v0.2.4:examples/example-task/README.md\n\n# Example Task: UI Language Fix\n\nThis example shows what a completed Proof Loop task looks like inside a repository.\n\nThe task artifacts live under:\n\n```text\n.agent/tasks/ui-language-fix/\n```\n\nRead them in this order:\n\n1. `spec.md` - frozen acceptance criteria and verification plan\n2. `evidence.md` - what was changed and checked\n3. `verdict.json` - structured verifier result\n4. `problems.md` - empty because the final verifier pass found no open issues\n\nThe example is intentionally small. Its job is to prove the artifact shape, not to ship a real app.\n\nFile v0.2.4:examples/README.md\n\n# Proof Loop Examples\n\nStart here if you want proof instead of prose.\n\n| Example | What it shows |\n| --- | --- |\n| [`demo-repo/`](demo-repo/) | A small runnable task with a real file check and passing proof artifacts. |\n| [`example-task/`](example-task/) | A completed task folder with spec, verdict, evidence, and problems files. |\n| [`role-briefs/`](role-briefs/) | Copy-paste briefs for orchestrator, spec freezer, builder, verifier, and fixer roles. |\n\nFast path:\n\n```bash\nmake test\nbin/proof-loop check examples/demo-repo/.agent/tasks/nav-labels-proof\nbin/proof-loop report examples/demo-repo/.agent/tasks/nav-labels-proof --format md\n```\n\nFile v0.2.4:README.md\n\n# Proof Loop\n\n![Tests](https://github.com/LeoStehlik/proof-loop/actions/workflows/test.yml/badge.svg)\n\n**Make AI coding agents prove when work is done.**\n\nProof Loop is a repo-local verification protocol for AI coding agents. It freezes acceptance criteria before the build, separates builder and verifier roles, records durable proof artifacts in the repo, and refuses to call work done until every acceptance criterion has a fresh PASS verdict.\n\nUse it when an agent, team, or multi-agent sprint needs a clear boundary between “looks done” and verified work. Because the protocol is just files plus role discipline, it works with OpenClaw, Hermes, Codex, OpenCode, Claude Code, or any other harness that can read and write a repository.\n\n## Start Here\n\nRun the current proof gate first:\n\n```bash\ngit clone https://github.com/LeoStehlik/proof-loop.git\ncd proof-loop\nmake test\nbin/proof-loop doctor\n```\n\nStart a proof-tracked task in any repo:\n\n```bash\nbin/proof-loop init maintenance-check --title \"Refresh the README proof\" --root /path/to/repo\nbin/proof-loop check /path/to/repo/.agent/tasks/maintenance-check\n```\n\nThe first check should fail until a fresh verifier records `PASS` for every frozen acceptance criterion and clears `problems.md`. That failure is the point: Proof Loop gives agents a mechanical done gate instead of a confident paragraph.\n\n## Works With\n\nProof Loop is harness-agnostic. Use it with any coding agent that can read files, write files, run commands, and hand verification to a fresh session.\n\nKnown-fit surfaces:\n\n- Codex\n- Claude Code\n- OpenClaw\n- OpenCode\n- Hermes\n- custom multi-agent runners\n\nThe repo includes copy-paste guide templates for several harnesses under `templates/`, plus role briefs under `examples/role-briefs/`.\n\n\n## Activation and Safety\n\nUse Proof Loop when the user explicitly wants an evidence-gated coding sprint, frozen acceptance criteria, fresh verifier separation, or durable proof artifacts. It is not meant to silently wrap every code change.\n\nThe bundled helpers create and check repo-local files under `.agent/tasks/<TASK_ID>/`. Review the task id and repository root before running them. Publishing reports, using remote workers, changing permissions, or running elevated commands are outside this skill unless requested separately.\n\n## Use Cases\n\n- keep AI coding agents honest when they claim a task is done\n- freeze acceptance criteria before implementation starts\n- separate builder and verifier roles in multi-agent coding work\n- leave proof artifacts in the repo for future review\n\n![Animated terminal demo: Proof Loop doctor, check, and report commands](assets/proof-loop-terminal-demo.svg)\n\nProof artifacts and role-brief examples are indexed in [`examples/README.md`](examples/README.md).\n\n## 20-second demo\n\n```bash\ngit clone https://github.com/LeoStehlik/proof-loop.git\ncd proof-loop\nmake test\n\ntmp=$(mktemp -d)\nbin/proof-loop-init hn-demo --title \"Prove this task before done\" --root \"$tmp\"\nbin/proof-loop-check \"$tmp/.agent/tasks/hn-demo\"\n```\n\nThe last command fails on purpose because the generated task has not been verified yet. Proof Loop only returns success after a fresh verifier records `PASS` for every acceptance criterion and `problems.md` is empty.\n\nA completed passing example is included:\n\n```bash\nbin/proof-loop-check examples/example-task/.agent/tasks/ui-language-fix\nbin/proof-loop doctor\nbin/proof-loop report examples/demo-repo/.agent/tasks/nav-labels-proof --format md\n```\n\n## Why It Exists\n\nAI coding agents often fail in predictable ways:\n\n- they claim completion without durable proof\n- the same session builds and judges its own work\n- acceptance criteria drift while implementation is underway\n- verification is a prose summary instead of a live check\n- future sessions cannot tell what was actually tested\n\nProof Loop makes completion auditable. A task is done only when a fresh verifier has checked each AC and the repo contains the artifacts to prove it.\n\n## What You Get\n\n- a clear sprint protocol: spec freeze -> build -> evidence -> fresh verify -> fix loop\n- role boundaries for orchestrator, spec-freezer, builder, verifier, and fixer\n- helper scripts to initialize and check task proof folders\n- a complete example task with passing artifacts\n- copy-paste role briefs for OpenClaw, Hermes, Codex, OpenCode, Claude Code, or any agent setup\n- a documented boundary with Loopsmith for recurring behaviour improvement\n\n## CLI\n\n```bash\nbin/proof-loop init TASK_ID --title \"Task title\"\nbin/proof-loop check TASK_ID\nbin/proof-loop status TASK_ID\nbin/proof-loop list\nbin/proof-loop doctor\nbin/proof-loop report TASK_ID --format md\nbin/proof-loop install-guides --dry-run --harness codex --harness claude\n```\n\n## Quick Start\n\nClone the repo or copy it into the project where you want to run the protocol.\n\nCreate a task proof folder from this repo or from another repository:\n\n```bash\nbin/proof-loop-init ui-language-fix --title \"Fix German navigation labels\" --root .\n```\n\nThis creates:\n\n```text\n.agent/tasks/ui-language-fix/\n  spec.md\n  verdict.json\n  problems.md\n  evidence.md\n```\n\nFill `spec.md` with explicit acceptance criteria before implementation starts.\n\nAfter the build and verifier pass, check whether the task is allowed to be called done:\n\n```bash\nbin/proof-loop-check .agent/tasks/ui-language-fix\n```\n\nThe check exits non-zero unless:\n\n- `verdict.json` has `overall: PASS`\n- every AC has `status: PASS`\n- `problems.md` is empty or absent\n\n\n## What This Is Not\n\n- not an agent framework\n- not a benchmark suite\n- not a replacement for tests\n- not tied to one model, vendor, or harness\n\nProof Loop is deliberately small: a protocol, a few files, and a mechanical done gate.\n\n## The Protocol\n\n```text\nspec freeze -> build -> evidence -> fresh verify -> fix -> fresh verify\n                                         ^                    |\n                                         |____________________|\n                                      repeat until all ACs PASS\n```\n\n## Roles\n\n| Role | Does | Never |\n|---|---|---|\n| Orchestrator | Keeps the loop intact and refuses weak completion | Accepts narrative-only proof |\n| Spec-Freezer | Writes frozen `spec.md` with explicit ACs | Edits production code |\n| Builder | Implements against the frozen spec | Verifies own work as final |\n| Verifier | Fresh session that checks each AC | Edits production code |\n| Fixer | Applies minimal fixes for verifier findings | Signs off on completion |\n\nThe verifier must be a fresh session. The agent that built the change does not judge whether the change is done.\n\n## Acceptance Criteria\n\nGood ACs are specific and testable by a third party.\n\n```text\nAC1: A user with locale=de sees all navigation labels in German after saving language preference.\n     Verify: browser check against a German-locale test user.\n\nAC2: The language preference survives page reload.\n     Verify: reload the page and confirm the saved locale and labels remain German.\n\nAC3: Existing English navigation remains unchanged for locale=en.\n     Verify: switch back to English and confirm the original labels render.\n```\n\nWeak ACs are task descriptions, not proof conditions:\n\n```text\nAC1: Translate the UI.\nAC2: Make language switching work.\nAC3: Fix the bugs.\n```\n\n## Artifacts\n\nEvery task stores proof under `.agent/tasks/<TASK_ID>/`.\n\n```text\n.agent/tasks/<TASK_ID>/\n  spec.md       frozen ACs, constraints, non-goals, verification approach\n  evidence.md   build summary and checks run\n  verdict.json  structured verifier result: PASS / FAIL / UNKNOWN per AC\n  problems.md   specific open failures, empty when no problems remain\n```\n\nSee [`references/artifacts.md`](references/artifacts.md) for schemas.\n\n## Real Demo\n\nRun a small failing-to-passing demo:\n\n```bash\nmake demo\n```\n\nThe demo intentionally breaks a tiny navigation-label fixture, shows the check failing, applies the fix, reruns the check, and renders a proof report.\n\n## Examples\n\nA complete passing example lives at:\n\n```text\nexamples/example-task/.agent/tasks/ui-language-fix/\n```\n\nRole prompts live at:\n\n```text\nexamples/role-briefs/\n  orchestrator.md\n  spec-freezer.md\n  builder.md\n  verifier.md\n  fixer.md\n```\n\n## Proof Loop vs Loopsmith\n\nProof Loop governs a single task.\n\nLoopsmith improves repeated agent behaviour over time.\n\nUse Proof Loop when you need a specific task to finish with evidence. Use [Loopsmith](https://github.com/LeoStehlik/loopsmith) when the same failure pattern keeps coming back and you want to improve the agent, prompt, policy, or evaluator itself.\n\nSee [`references/loopsmith-bridge.md`](references/loopsmith-bridge.md).\n\n\n## When To Use Which Repo\n\nUse this repo when a specific coding task needs evidence before anyone is allowed to call it done. Proof Loop freezes the spec, separates builder and verifier roles, requires proof artifacts, and records verdicts in the repo.\n\nUse the neighbouring tools at different points in the workflow:\n\n| Need | Use |\n| --- | --- |\n| Turn a fuzzy request into an executable agent brief | [Brief Master](https://github.com/LeoStehlik/brief-master) |\n| Prove one coding task is actually done | [Proof Loop](https://github.com/LeoStehlik/proof-loop) |\n| Improve repeated agent behaviour with evals | [Loopsmith](https://github.com/LeoStehlik/loopsmith) |\n| Keep source-backed memory for long-running agents | [Sovereign Brain](https://github.com/LeoStehlik/decoupled-agent-memory) |\n| Stop frontend agents producing generic UI sludge | [no-slop-ui](https://github.com/LeoStehlik/no-slop-ui) |\n\nA practical chain looks like this: messy request -> Brief Master brief -> Proof Loop task -> Loopsmith eval if the same failure keeps recurring -> Sovereign Brain records the durable decision.\n\n## Related Tools\n\n- [Loopsmith](https://github.com/LeoStehlik/loopsmith) - use when Proof Loop exposes a repeated agent behaviour problem that should become an eval and promotion loop.\n- [Sovereign Brain](https://github.com/LeoStehlik/decoupled-agent-memory) - source-backed memory for long-running agents; useful when proof artifacts, decisions, and synthesis need durable context.\n- [Brief Master](https://github.com/LeoStehlik/brief-master) - helps write sharper task briefs and acceptance criteria before a Proof Loop starts.\n\n## Installation As A Skill\n\n### OpenClaw\n\nAdd your skills directory to `openclaw.json`:\n\n```json\n{\n  \"skills\": {\n    \"load\": {\n      \"extraDirs\": [\"/path/to/your/skills\"]\n    }\n  }\n}\n```\n\nClone this repo into that directory:\n\n```bash\ngit clone https://github.com/LeoStehlik/proof-loop.git /path/to/your/skills/proof-loop\n```\n\n### Codex / Claude Code\n\nCopy the `proof-loop` folder into your agent skills directory, or reference `SKILL.md` directly in your task brief. For harnesses without a formal skill system, use the README, role briefs, and scripts directly from the repo.\n\n## Repository Map\n\n```text\nproof-loop/\n  SKILL.md                         skill trigger and core operating rules\n  bin/\n    proof-loop                     unified CLI\n    proof-loop-init                compatibility wrapper\n    proof-loop-check               compatibility wrapper\n  scripts/\n    init_task.py                   create .agent/tasks/<TASK_ID>/ skeletons\n    check_task.py                  mechanical done gate\n  schemas/                         JSON schemas for verdict and evidence bundles\n  templates/                       opt-in harness guide templates\n  tests/                           stdlib unittest coverage for CLI behavior\n  .github/workflows/test.yml       CI running make test\n  references/\n    workflow.md                    full phase-by-phase protocol\n    brief-template.md              reusable sprint and role prompts\n    artifacts.md                   artifact schemas\n    loopsmith-bridge.md            when to escalate repeated failures to Loopsmith\n  examples/\n    example-task/                  complete passing proof artifact example\n    role-briefs/                   copy-paste role prompts\n```\n\n## Status\n\nUsable protocol skill and small toolkit. The scripts are intentionally stdlib-only so they can run inside almost any repository without packaging ceremony.\n\n## License\n\nMIT - see [LICENSE](LICENSE).\n\n## Attribution\n\nInspired by [`repo-task-proof-loop`](https://github.com/DenisSergeevitch/repo-task-proof-loop), adapted for practical multi-agent coding work and public agent-operation skills.\n\nFile v0.2.4:_meta.json\n\n{\n  \"ownerId\": \"kn7d3r58cdxk8k0xg6jq6k65gs873c0c\",\n  \"slug\": \"proof-loop\",\n  \"version\": \"0.2.4\",\n  \"publishedAt\": 1787739458144\n}\n\nFile v0.2.4:references/artifacts.md\n\n# Artifact Schemas\n\nAll artifacts live in the repository under `.agent/tasks/<TASK_ID>/`.\n\n---\n\n## spec.md\n\nPlain Markdown. Written by Spec-Freezer before build starts. Never modified after freeze.\n\n```markdown\n# Task: [TASK_ID]\n\n## Task Statement\n[Original task description]\n\n## Acceptance Criteria\n\n**AC1:** [testable condition]\n- Verify: [how to check]\n\n**AC2:** [testable condition]\n- Verify: [how to check]\n\n## Constraints\n- [what must not break]\n\n## Non-Goals\n- [what is out of scope]\n\n## Verification Approach\n[Overall approach to verifying this task]\n```\n\n---\n\n## verdict.json\n\nWritten by Verifier. Updated after each fix loop iteration.\n\n```json\n{\n  \"task_id\": \"sprint-4c\",\n  \"phase\": \"verify\",\n  \"agent\": \"verifier\",\n  \"timestamp\": \"2026-03-30T14:00:00Z\",\n  \"overall\": \"FAIL\",\n  \"criteria\": [\n    {\n      \"id\": \"AC1\",\n      \"status\": \"PASS\",\n      \"note\": \"All nav labels confirmed German in browser test\"\n    },\n    {\n      \"id\": \"AC2\",\n      \"status\": \"FAIL\",\n      \"note\": \"Form field labels still in English — formSchemaTranslated is null in DB\"\n    },\n    {\n      \"id\": \"AC3\",\n      \"status\": \"PASS\",\n      \"note\": \"pnpm test:e2e — all sprint 1-4c specs green\"\n    }\n  ]\n}\n```\n\n**overall** is PASS only when every AC status is PASS.\n**status values:** PASS | FAIL | UNKNOWN\n\n---\n\n## problems.md\n\nWritten by Verifier when overall is not PASS. Specific, actionable.\n\n```markdown\n# Problems — [TASK_ID]\n\n## AC2: Form field labels in English\n\n**File:** `packages/llm/src/translation.ts`\n**Function:** `translatePrompts()`\n**Issue:** All prompt chunks translated in parallel via `Promise.all()`. Under slow local\nmodels, all calls time out simultaneously and fall back to English silently.\n**Evidence:** DB query shows `formSchemaTranslated IS NULL` for all 50 rows.\n**Fix needed:** Change parallel chunk processing to sequential — one prompt at a time.\n```\n\n---\n\n## evidence.md (optional)\n\nProse summary written by Builder in evidence mode. Supplements verdict.json.\n\n```markdown\n# Evidence — [TASK_ID]\n\n## Build Summary\n[What was changed and why]\n\n## AC1 — PASS\n[How it was verified, what was checked]\n\n## AC2 — FAIL\n[What was attempted, what failed, why]\n```\n\nFile v0.2.4:references/brief-template.md\n\n# Agent Brief Template\n\nUse this template when briefing any coding agent on a non-trivial task.\n\n---\n\n## Sprint/Task Brief: [TASK_ID]\n\n### Task Statement\n[One paragraph describing what needs to be built and why]\n\n### Acceptance Criteria\n\n```\nAC1: [specific, testable condition]\n     Verify: [how to check this — command, visual check, API call]\n\nAC2: [specific, testable condition]\n     Verify: [how to check this]\n\nAC3: [specific, testable condition]\n     Verify: [how to check this]\n```\n\n**Good AC examples:**\n- \"AC1: A user with locale=de sees all navigation labels in German after saving language preference\"\n- \"AC2: POST /api/v1/prompts/translate/de returns 200 with translated titles for all 50 prompts\"\n- \"AC3: `pnpm test:e2e` passes all sprint spec files with no failures\"\n\n**Bad AC examples:**\n- \"AC1: Translate the UI\" (not testable)\n- \"AC1: Make it work in German\" (vague)\n- \"AC1: Fix the translation bugs\" (not specific)\n\n### Constraints\n[What must not break — existing features, performance, API contracts]\n\n### Non-Goals\n[What is explicitly OUT of scope for this task]\n\n### Verification Approach\n[How each AC will be verified — automated tests, manual browser check, API call, etc.]\n\n---\n\n## Brief Checklist (orchestrator — run before firing agents)\n\n- [ ] ACs are explicit, testable, and frozen\n- [ ] Each AC has a verification approach\n- [ ] Constraints are listed\n- [ ] Non-goals are listed\n- [ ] Builder role is clear (who builds)\n- [ ] Verifier role is clear (different agent, fresh session)\n- [ ] Fixer role is clear (if needed)\n- [ ] Artifact path is set: `.agent/tasks/[TASK_ID]/`\n\n---\n\n## Prompts by Role\n\n### Spec-Freezer prompt\n```\nYou are the spec freezer for task [TASK_ID].\n\nRead the brief above. Write spec.md to .agent/tasks/[TASK_ID]/spec.md with:\n- Original task statement\n- Acceptance Criteria (AC1, AC2, AC3...) — copy from brief, do not modify\n- Constraints\n- Non-goals\n- Verification approach per AC\n\nDo not edit any production code. Do not start building.\n```\n\n### Builder prompt\n```\nYou are the builder for task [TASK_ID].\n\nRead .agent/tasks/[TASK_ID]/spec.md. Implement the task against the frozen ACs.\nMake the smallest safe change set that satisfies all ACs.\nDo not verify your own work. When done, hand off to the evidence phase.\n```\n\n### Verifier prompt\n```\nYou are the verifier for task [TASK_ID]. This is a fresh session.\n\nRead:\n- .agent/tasks/[TASK_ID]/spec.md (the frozen ACs)\n- .agent/tasks/[TASK_ID]/verdict.json (current verdicts)\n- .agent/tasks/[TASK_ID]/problems.md (known problems)\n\nRun independent checks. For each AC, write your verdict: PASS / FAIL / UNKNOWN.\nUpdate verdict.json. If any AC is not PASS, update problems.md with specific file/line references.\n\nDo not edit production code. Do not sign off on completion.\n```\n\n### Fixer prompt\n```\nYou are the fixer for task [TASK_ID].\n\nRead:\n- .agent/tasks/[TASK_ID]/spec.md\n- .agent/tasks/[TASK_ID]/verdict.json\n- .agent/tasks/[TASK_ID]/problems.md\n\nFix only what the verifier identified. Apply the minimal diff. Regenerate evidence.\nDo not write final sign-off. A fresh verifier will re-run after your fix.\n```\n\nFile v0.2.4:references/loopsmith-bridge.md\n\n# Loopsmith Bridge\n\nProof Loop and Loopsmith solve different parts of the same reliability problem.\n\n## Boundary\n\nUse Proof Loop when you need one task to finish with evidence:\n\n- freeze acceptance criteria\n- separate builder and verifier roles\n- store task-local artifacts\n- block self-certified done claims\n\nUse Loopsmith when the same failure pattern keeps coming back and you want to improve the agent, prompt, policy, or evaluator itself:\n\n- compare baseline vs candidate behaviour\n- run eval packs\n- score recurring failure modes\n- promote or reject changes with a ledger\n\nShort version:\n\n> Proof Loop governs a task. Loopsmith improves the agent system over time.\n\n## When Proof Loop Is Enough\n\nProof Loop is enough when:\n\n- the task is bounded\n- the acceptance criteria are clear\n- the verifier can run concrete checks\n- failures are task-specific, not a repeated agent behaviour problem\n\nExample: a UI language bug has three ACs, a verifier checks them, and all PASS.\n\n## When To Escalate To Loopsmith\n\nEscalate to Loopsmith when Proof Loop reveals a pattern such as:\n\n- builders repeatedly claim done without evidence\n- verifiers return vague narrative instead of AC-by-AC verdicts\n- fixers patch around symptoms and miss regression checks\n- orchestrators write weak or drifting acceptance criteria\n- the same class of failure appears across multiple tasks\n\nAt that point, the task is no longer the only problem. The agent behaviour needs an eval.\n\n## Turning A Proof Loop Failure Into A Loopsmith Case\n\n1. Pick the smallest representative Proof Loop artifact set.\n2. Convert the failure into an eval case:\n   - input: the task brief, spec, evidence, verdict, or problems file\n   - expected behaviour: what a good agent should do\n   - anti-goals: what the old agent did wrong\n3. Add a baseline response that shows the current behaviour.\n4. Add a candidate policy/prompt/evaluator change.\n5. Run Loopsmith and promote only if the candidate improves the evidence.\n\n## Example Mapping\n\n| Proof Loop Artifact | Loopsmith Use |\n|---|---|\n| `spec.md` | eval input for AC quality or builder discipline |\n| `verdict.json` | structural target for verifier verdict discipline |\n| `problems.md` | input for fixer minimality and regression-awareness cases |\n| `evidence.md` | input for evidence quality or false-completion checks |\n\n## Practical Rule\n\nDo not use Loopsmith for every task. That creates ceremony.\n\nUse Proof Loop by default for non-trivial work. Use Loopsmith when a repeated failure deserves a reusable improvement loop.\n\nFile v0.2.4:references/workflow.md\n\n# Proof Loop Workflow\n\n## Phase 0: Spec Freeze\n\n**Who:** Orchestrator (project lead, or whoever is briefing)\n\nBefore a single line of code is written:\n1. Write `spec.md` with explicit acceptance criteria (AC1, AC2, AC3...)\n2. Each AC must be testable by a third party who didn't build it\n3. Include: constraints, non-goals, verification approach\n4. Freeze the spec — ACs cannot change during build\n\n**spec.md must include:**\n- Original task statement\n- Acceptance Criteria (AC1, AC2, ...)\n- Constraints (what must not break)\n- Non-goals (what is explicitly out of scope)\n- Verification approach (how each AC will be checked)\n\n---\n\n## Phase 1: Build\n\n**Who:** Builder agent (fresh session)\n\n- Reads spec.md — implements against frozen ACs only\n- Makes the smallest safe change set that satisfies all ACs\n- Does not verify own work\n- Hands off to evidence phase when implementation is complete\n\n---\n\n## Phase 2: Evidence\n\n**Who:** Builder (same session, switched to evidence mode) or a fresh evidence agent\n\n- Does not change production code\n- Runs checks and records results\n- Writes `evidence.md` (prose) and updates `verdict.json` (structured)\n- Verdict per AC: PASS / FAIL / UNKNOWN\n- If FAIL or UNKNOWN: writes `problems.md` with specific file/line references\n\n**Builder hard constraints in evidence mode:**\n- Must not patch evidence to make it look better\n- Must not change production code\n- UNKNOWN is valid — it means \"could not verify\"\n\n---\n\n## Phase 3: Fresh Verify\n\n**Who:** Verifier — always a NEW session, never the same agent that built\n\n- Reads `spec.md`, `verdict.json`, `problems.md`\n- Runs independent checks against the current codebase\n- Updates `verdict.json` with verifier's findings\n- Updates `problems.md` if issues found\n- **Must not edit production code**\n- **Must not sign off on completion**\n\n---\n\n## Phase 4: Fix (if needed)\n\n**Who:** Fixer — fresh session\n\n- Reads `spec.md` + `verdict.json` + `problems.md`\n- Reconfirms each problem before editing\n- Applies minimal fix — only what the verifier identified\n- Regenerates evidence\n- **Does not write final sign-off** — goes back to Phase 3\n\n---\n\n## The Fix Loop\n\n```\nfix -> fresh verify -> fix -> fresh verify\n```\n\nContinues until verifier writes PASS for every AC.\n\nA single successful PASS does not end the loop — ALL ACs must be PASS.\n\n---\n\n## Done Criteria\n\nA task is complete when:\n- Every AC in `spec.md` has status PASS in `verdict.json`\n- `problems.md` is empty or does not exist\n- Any regression suite required by the spec passes\n\n---\n\n## Common Failure Modes (and how this prevents them)\n\n| Failure | Prevention |\n|---------|-----------|\n| Agent claims done without checking | Verifier is separate, required |\n| ACs drift during build | Spec is frozen before build starts |\n| Later sessions can't tell what was verified | Verdict artifacts stay in repo |\n| Builder judges own work | Fresh verifier is a hard rule |\n| Fix introduces new regression | Verifier reruns after every fix |\n\nFile v0.2.4:examples/role-briefs/builder.md\n\n# Builder Brief\n\nYou are the builder for task `[TASK_ID]`.\n\nRead `.agent/tasks/[TASK_ID]/spec.md` and implement only what is needed to satisfy the frozen ACs.\n\n## Responsibilities\n\n- Make the smallest safe change set.\n- Respect constraints and non-goals.\n- Update `evidence.md` with what changed and what checks you ran.\n- Leave final verdicting to a fresh verifier.\n\n## Hard Boundaries\n\n- Do not verify your own work as final.\n- Do not mark the task done.\n- Do not change frozen ACs to match your implementation.\n\nFile v0.2.4:examples/role-briefs/fixer.md\n\n# Fixer Brief\n\nYou are the fixer for task `[TASK_ID]`.\n\nRead:\n\n- `.agent/tasks/[TASK_ID]/spec.md`\n- `.agent/tasks/[TASK_ID]/verdict.json`\n- `.agent/tasks/[TASK_ID]/problems.md`\n\n## Responsibilities\n\n- Reproduce or understand each verifier-reported problem.\n- Make the minimal fix for those problems only.\n- Update `evidence.md` with changed files and checks run.\n- Hand back to a fresh verifier.\n\n## Hard Boundaries\n\n- Do not broaden scope.\n- Do not rewrite unrelated code.\n- Do not write final sign-off.\n\nFile v0.2.4:examples/role-briefs/orchestrator.md\n\n# Orchestrator Brief\n\nYou are the orchestrator for task `[TASK_ID]`.\n\nYour job is to keep the Proof Loop intact.\n\n## Responsibilities\n\n- Ensure `.agent/tasks/[TASK_ID]/spec.md` exists before build starts.\n- Confirm all ACs are explicit and testable.\n- Assign separate builder and verifier roles.\n- Refuse final completion until `scripts/check_task.py .agent/tasks/[TASK_ID]` passes.\n\n## Hard Boundaries\n\n- Do not let the builder verify their own work.\n- Do not change ACs mid-build. If the scope changes, create a new task or explicitly revise the frozen spec before implementation continues.\n- Do not accept a narrative summary as proof. Require artifacts.\n\nArchive v0.2.1: 34 files, 32534 bytes\n\nFiles: assets/proof-loop-terminal-demo.svg (14096b), bin/proof-loop (9901b), bin/proof-loop-check (139b), bin/proof-loop-init (138b), examples/demo-repo/check_nav_labels.py (826b), examples/demo-repo/nav_labels.json (193b), examples/demo-repo/run_demo.py (1748b), examples/example-task/README.md (557b), examples/README.md (644b), examples/role-briefs/builder.md (514b), examples/role-briefs/fixer.md (505b), examples/role-briefs/orchestrator.md (658b), examples/role-briefs/spec-freezer.md (513b), examples/role-briefs/verifier.md (648b), LICENSE (1068b), Makefile (748b), README.md (11329b), references/artifacts.md (2186b), references/brief-template.md (3128b), references/loopsmith-bridge.md (2535b), references/workflow.md (2986b), schemas/evidence.schema.json (328b), schemas/spec.schema.json (312b), schemas/verdict.schema.json (440b), scripts/check_task.py (2887b), scripts/init_task.py (3445b), skill-card.md (2118b), SKILL.md (3476b), templates/AGENTS.proof-loop.md (237b), templates/CLAUDE.proof-loop.md (189b), templates/hermes.proof-loop.md (183b), templates/opencode.proof-loop.md (161b), tests/test_cli.py (7272b), _meta.json (129b)\n\nFile v0.2.1:SKILL.md\n\n---\nname: proof-loop\ndescription: \"Run evidence-gated coding sprints with frozen ACs, separated builder/verifier roles, and durable proof artifacts.\"\nmetadata:\n  version: \"0.2.1\"\n---\n# Proof Loop\n\nA sprint is not done until every acceptance criterion has a PASS verdict from a fresh verifier session.\n\nRead `references/workflow.md` for the full loop spec.\nRead `references/brief-template.md` for the agent brief format.\nRead `references/artifacts.md` for the artifact schema.\nRead `references/loopsmith-bridge.md` when deciding whether a repeated Proof Loop failure should become a Loopsmith eval case.\n\n\n## Activation and Safety Boundaries\n\nUse this skill only when the user explicitly asks for Proof Loop, proof artifacts, acceptance-criterion verification, fresh verifier separation, or an evidence-gated coding sprint. Do not activate it for ordinary code edits where the user did not request this protocol.\n\nProof Loop may create or update files under `.agent/tasks/<TASK_ID>/` in the current repository. Confirm the task id and repository root before creating artifacts. Do not publish artifacts, run remote validation, change repository permissions, moderate users, request full-access sandboxing, or touch credentials unless the user explicitly asks for that separate action in the current conversation.\n\nRun helper scripts with the least privilege available. If a command could modify source files outside `.agent/tasks/<TASK_ID>/`, ask first and record the command in the evidence artifact.\n\n## The Loop\n\n```\nspec freeze -> build -> evidence -> FRESH verify -> fix -> FRESH verify\n                                         ^                      |\n                                         |______________________|\n                                         (repeat until all ACs = PASS)\n```\n\n## Four Roles — Always Separate\n\n| Role | Does | Never |\n|------|------|-------|\n| **Spec-Freezer** | Writes spec.md with explicit ACs | Edits production code |\n| **Builder** | Implements against frozen spec | Verifies own work |\n| **Verifier** | Fresh session — verdicts each AC | Edits production code |\n| **Fixer** | Minimal fix for what verifier flagged | Signs off on completion |\n\n**The verifier is always a fresh session.** The agent that built cannot judge its own work.\n\n## Acceptance Criteria Format\n\nEvery sprint brief must include explicit ACs before build starts:\n\n```\nAC1: [specific, testable condition — not a task description]\nAC2: [specific, testable condition]\nAC3: [specific, testable condition]\n```\n\nGood: \"AC1: A German-locale user sees all prompt form field labels in German\"\nBad: \"AC1: Translate the form fields\"\n\n## Helper Scripts\n\nUse these when the repository has the `proof-loop` folder available:\n\n```bash\npython3 scripts/init_task.py TASK_ID --title \"Task title\"\npython3 scripts/check_task.py .agent/tasks/TASK_ID\n```\n\n`check_task.py` is the mechanical done gate. It returns success only when the verifier artifacts show every AC as PASS and no open problems remain.\n\n## Sprint is DONE Only When\n\n- Every AC has a PASS verdict in the verifier's `verdict.json`\n- No problems remain in `problems.md`\n- Full regression suite passes (if applicable)\n\n## Artifacts (stored in repo)\n\n```\n.agent/tasks/<TASK_ID>/\n  spec.md         -- frozen ACs + constraints + non-goals\n  verdict.json    -- AC verdicts per phase (PASS/FAIL/UNKNOWN)\n  problems.md     -- specific failures with file/line refs (if any)\n```\n\nSee `references/artifacts.md` for schemas.\n\nFile v0.2.1:examples/example-task/README.md\n\n# Example Task: UI Language Fix\n\nThis example shows what a completed Proof Loop task looks like inside a repository.\n\nThe task artifacts live under:\n\n```text\n.agent/tasks/ui-language-fix/\n```\n\nRead them in this order:\n\n1. `spec.md` - frozen acceptance criteria and verification plan\n2. `evidence.md` - what was changed and checked\n3. `verdict.json` - structured verifier result\n4. `problems.md` - empty because the final verifier pass found no open issues\n\nThe example is intentionally small. Its job is to prove the artifact shape, not to ship a real app.\n\nFile v0.2.1:examples/README.md\n\n# Proof Loop Examples\n\nStart here if you want proof instead of prose.\n\n| Example | What it shows |\n| --- | --- |\n| [`demo-repo/`](demo-repo/) | A small runnable task with a real file check and passing proof artifacts. |\n| [`example-task/`](example-task/) | A completed task folder with spec, verdict, evidence, and problems files. |\n| [`role-briefs/`](role-briefs/) | Copy-paste briefs for orchestrator, spec freezer, builder, verifier, and fixer roles. |\n\nFast path:\n\n```bash\nmake test\nbin/proof-loop check examples/demo-repo/.agent/tasks/nav-labels-proof\nbin/proof-loop report examples/demo-repo/.agent/tasks/nav-labels-proof --format md\n```\n\nFile v0.2.1:README.md\n\n# Proof Loop\n\n![Tests](https://github.com/LeoStehlik/proof-loop/actions/workflows/test.yml/badge.svg)\n\n**Make AI coding agents prove when work is done.**\n\nProof Loop is a repo-local verification protocol for AI coding agents. It freezes acceptance criteria before the build, separates builder and verifier roles, records durable proof artifacts in the repo, and refuses to call work done until every acceptance criterion has a fresh PASS verdict.\n\nUse it when an agent, team, or multi-agent sprint needs a clear boundary between “looks done” and verified work. Because the protocol is just files plus role discipline, it works with OpenClaw, Hermes, Codex, OpenCode, Claude Code, or any other harness that can read and write a repository.\n\n\n## Activation and Safety\n\nUse Proof Loop when the user explicitly wants an evidence-gated coding sprint, frozen acceptance criteria, fresh verifier separation, or durable proof artifacts. It is not meant to silently wrap every code change.\n\nThe bundled helpers create and check repo-local files under `.agent/tasks/<TASK_ID>/`. Review the task id and repository root before running them. Publishing reports, using remote workers, changing permissions, or running elevated commands are outside this skill unless requested separately.\n\n## Use Cases\n\n- keep AI coding agents honest when they claim a task is done\n- freeze acceptance criteria before implementation starts\n- separate builder and verifier roles in multi-agent coding work\n- leave proof artifacts in the repo for future review\n\n![Animated terminal demo: Proof Loop doctor, check, and report commands](assets/proof-loop-terminal-demo.svg)\n\nProof artifacts and role-brief examples are indexed in [`examples/README.md`](examples/README.md).\n\n## 20-second demo\n\n```bash\ngit clone https://github.com/LeoStehlik/proof-loop.git\ncd proof-loop\nmake test\n\ntmp=$(mktemp -d)\nbin/proof-loop-init hn-demo --title \"Prove this task before done\" --root \"$tmp\"\nbin/proof-loop-check \"$tmp/.agent/tasks/hn-demo\"\n```\n\nThe last command fails on purpose because the generated task has not been verified yet. Proof Loop only returns success after a fresh verifier records `PASS` for every acceptance criterion and `problems.md` is empty.\n\nA completed passing example is included:\n\n```bash\nbin/proof-loop-check examples/example-task/.agent/tasks/ui-language-fix\nbin/proof-loop doctor\nbin/proof-loop report examples/demo-repo/.agent/tasks/nav-labels-proof --format md\n```\n\n## Why It Exists\n\nAI coding agents often fail in predictable ways:\n\n- they claim completion without durable proof\n- the same session builds and judges its own work\n- acceptance criteria drift while implementation is underway\n- verification is a prose summary instead of a live check\n- future sessions cannot tell what was actually tested\n\nProof Loop makes completion auditable. A task is done only when a fresh verifier has checked each AC and the repo contains the artifacts to prove it.\n\n## What You Get\n\n- a clear sprint protocol: spec freeze -> build -> evidence -> fresh verify -> fix loop\n- role boundaries for orchestrator, spec-freezer, builder, verifier, and fixer\n- helper scripts to initialize and check task proof folders\n- a complete example task with passing artifacts\n- copy-paste role briefs for OpenClaw, Hermes, Codex, OpenCode, Claude Code, or any agent setup\n- a documented boundary with Loopsmith for recurring behaviour improvement\n\n## CLI\n\n```bash\nbin/proof-loop init TASK_ID --title \"Task title\"\nbin/proof-loop check TASK_ID\nbin/proof-loop status TASK_ID\nbin/proof-loop list\nbin/proof-loop doctor\nbin/proof-loop report TASK_ID --format md\nbin/proof-loop install-guides --dry-run --harness codex --harness claude\n```\n\n## Quick Start\n\nClone the repo or copy it into the project where you want to run the protocol.\n\nCreate a task proof folder from this repo or from another repository:\n\n```bash\nbin/proof-loop-init ui-language-fix --title \"Fix German navigation labels\" --root .\n```\n\nThis creates:\n\n```text\n.agent/tasks/ui-language-fix/\n  spec.md\n  verdict.json\n  problems.md\n  evidence.md\n```\n\nFill `spec.md` with explicit acceptance criteria before implementation starts.\n\nAfter the build and verifier pass, check whether the task is allowed to be called done:\n\n```bash\nbin/proof-loop-check .agent/tasks/ui-language-fix\n```\n\nThe check exits non-zero unless:\n\n- `verdict.json` has `overall: PASS`\n- every AC has `status: PASS`\n- `problems.md` is empty or absent\n\n\n## What This Is Not\n\n- not an agent framework\n- not a benchmark suite\n- not a replacement for tests\n- not tied to one model, vendor, or harness\n\nProof Loop is deliberately small: a protocol, a few files, and a mechanical done gate.\n\n## The Protocol\n\n```text\nspec freeze -> build -> evidence -> fresh verify -> fix -> fresh verify\n                                         ^                    |\n                                         |____________________|\n                                      repeat until all ACs PASS\n```\n\n## Roles\n\n| Role | Does | Never |\n|---|---|---|\n| Orchestrator | Keeps the loop intact and refuses weak completion | Accepts narrative-only proof |\n| Spec-Freezer | Writes frozen `spec.md` with explicit ACs | Edits production code |\n| Builder | Implements against the frozen spec | Verifies own work as final |\n| Verifier | Fresh session that checks each AC | Edits production code |\n| Fixer | Applies minimal fixes for verifier findings | Signs off on completion |\n\nThe verifier must be a fresh session. The agent that built the change does not judge whether the change is done.\n\n## Acceptance Criteria\n\nGood ACs are specific and testable by a third party.\n\n```text\nAC1: A user with locale=de sees all navigation labels in German after saving language preference.\n     Verify: browser check against a German-locale test user.\n\nAC2: The language preference survives page reload.\n     Verify: reload the page and confirm the saved locale and labels remain German.\n\nAC3: Existing English navigation remains unchanged for locale=en.\n     Verify: switch back to English and confirm the original labels render.\n```\n\nWeak ACs are task descriptions, not proof conditions:\n\n```text\nAC1: Translate the UI.\nAC2: Make language switching work.\nAC3: Fix the bugs.\n```\n\n## Artifacts\n\nEvery task stores proof under `.agent/tasks/<TASK_ID>/`.\n\n```text\n.agent/tasks/<TASK_ID>/\n  spec.md       frozen ACs, constraints, non-goals, verification approach\n  evidence.md   build summary and checks run\n  verdict.json  structured verifier result: PASS / FAIL / UNKNOWN per AC\n  problems.md   specific open failures, empty when no problems remain\n```\n\nSee [`references/artifacts.md`](references/artifacts.md) for schemas.\n\n## Real Demo\n\nRun a small failing-to-passing demo:\n\n```bash\nmake demo\n```\n\nThe demo intentionally breaks a tiny navigation-label fixture, shows the check failing, applies the fix, reruns the check, and renders a proof report.\n\n## Examples\n\nA complete passing example lives at:\n\n```text\nexamples/example-task/.agent/tasks/ui-language-fix/\n```\n\nRole prompts live at:\n\n```text\nexamples/role-briefs/\n  orchestrator.md\n  spec-freezer.md\n  builder.md\n  verifier.md\n  fixer.md\n```\n\n## Proof Loop vs Loopsmith\n\nProof Loop governs a single task.\n\nLoopsmith improves repeated agent behaviour over time.\n\nUse Proof Loop when you need a specific task to finish with evidence. Use [Loopsmith](https://github.com/LeoStehlik/loopsmith) when the same failure pattern keeps coming back and you want to improve the agent, prompt, policy, or evaluator itself.\n\nSee [`references/loopsmith-bridge.md`](references/loopsmith-bridge.md).\n\n\n## When To Use Which Repo\n\nUse this repo when a specific coding task needs evidence before anyone is allowed to call it done. Proof Loop freezes the spec, separates builder and verifier roles, requires proof artifacts, and records verdicts in the repo.\n\nUse the neighbouring tools at different points in the workflow:\n\n| Need | Use |\n| --- | --- |\n| Turn a fuzzy request into an executable agent brief | [Brief Master](https://github.com/LeoStehlik/brief-master) |\n| Prove one coding task is actually done | [Proof Loop](https://github.com/LeoStehlik/proof-loop) |\n| Improve repeated agent behaviour with evals | [Loopsmith](https://github.com/LeoStehlik/loopsmith) |\n| Keep source-backed memory for long-running agents | [Sovereign Brain](https://github.com/LeoStehlik/decoupled-agent-memory) |\n| Stop frontend agents producing generic UI sludge | [no-slop-ui](https://github.com/LeoStehlik/no-slop-ui) |\n\nA practical chain looks like this: messy request -> Brief Master brief -> Proof Loop task -> Loopsmith eval if the same failure keeps recurring -> Sovereign Brain records the durable decision.\n\n## Related Tools\n\n- [Loopsmith](https://github.com/LeoStehlik/loopsmith) - use when Proof Loop exposes a repeated agent behaviour problem that should become an eval and promotion loop.\n- [Sovereign Brain](https://github.com/LeoStehlik/decoupled-agent-memory) - source-backed memory for long-running agents; useful when proof artifacts, decisions, and synthesis need durable context.\n- [Brief Master](https://github.com/LeoStehlik/brief-master) - helps write sharper task briefs and acceptance criteria before a Proof Loop starts.\n\n## Installation As A Skill\n\n### OpenClaw\n\nAdd your skills directory to `openclaw.json`:\n\n```json\n{\n  \"skills\": {\n    \"load\": {\n      \"extraDirs\": [\"/path/to/your/skills\"]\n    }\n  }\n}\n```\n\nClone this repo into that directory:\n\n```bash\ngit clone https://github.com/LeoStehlik/proof-loop.git /path/to/your/skills/proof-loop\n```\n\n### Codex / Claude Code\n\nCopy the `proof-loop` folder into your agent skills directory, or reference `SKILL.md` directly in your task brief. For harnesses without a formal skill system, use the README, role briefs, and scripts directly from the repo.\n\n## Repository Map\n\n```text\nproof-loop/\n  SKILL.md                         skill trigger and core operating rules\n  bin/\n    proof-loop                     unified CLI\n    proof-loop-init                compatibility wrapper\n    proof-loop-check               compatibility wrapper\n  scripts/\n    init_task.py                   create .agent/tasks/<TASK_ID>/ skeletons\n    check_task.py                  mechanical done gate\n  schemas/                         JSON schemas for verdict and evidence bundles\n  templates/                       opt-in harness guide templates\n  tests/                           stdlib unittest coverage for CLI behavior\n  .github/workflows/test.yml       CI running make test\n  references/\n    workflow.md                    full phase-by-phase protocol\n    brief-template.md              reusable sprint and role prompts\n    artifacts.md                   artifact schemas\n    loopsmith-bridge.md            when to escalate repeated failures to Loopsmith\n  examples/\n    example-task/                  complete passing proof artifact example\n    role-briefs/                   copy-paste role prompts\n```\n\n## Status\n\nUsable protocol skill and small toolkit. The scripts are intentionally stdlib-only so they can run inside almost any repository without packaging ceremony.\n\n## License\n\nMIT - see [LICENSE](LICENSE).\n\n## Attribution\n\nInspired by [`repo-task-proof-loop`](https://github.com/DenisSergeevitch/repo-task-proof-loop), adapted for practical multi-agent coding work and public agent-operation skills.\n\nFile v0.2.1:_meta.json\n\n{\n  \"ownerId\": \"kn7d3r58cdxk8k0xg6jq6k65gs873c0c\",\n  \"slug\": \"proof-loop\",\n  \"version\": \"0.2.1\",\n  \"publishedAt\": 1779616791995\n}\n\nFile v0.2.1:references/artifacts.md\n\n# Artifact Schemas\n\nAll artifacts live in the repository under `.agent/tasks/<TASK_ID>/`.\n\n---\n\n## spec.md\n\nPlain Markdown. Written by Spec-Freezer before build starts. Never modified after freeze.\n\n```markdown\n# Task: [TASK_ID]\n\n## Task Statement\n[Original task description]\n\n## Acceptance Criteria\n\n**AC1:** [testable condition]\n- Verify: [how to check]\n\n**AC2:** [testable condition]\n- Verify: [how to check]\n\n## Constraints\n- [what must not break]\n\n## Non-Goals\n- [what is out of scope]\n\n## Verification Approach\n[Overall approach to verifying this task]\n```\n\n---\n\n## verdict.json\n\nWritten by Verifier. Updated after each fix loop iteration.\n\n```json\n{\n  \"task_id\": \"sprint-4c\",\n  \"phase\": \"verify\",\n  \"agent\": \"verifier\",\n  \"timestamp\": \"2026-03-30T14:00:00Z\",\n  \"overall\": \"FAIL\",\n  \"criteria\": [\n    {\n      \"id\": \"AC1\",\n      \"status\": \"PASS\",\n      \"note\": \"All nav labels confirmed German in browser test\"\n    },\n    {\n      \"id\": \"AC2\",\n      \"status\": \"FAIL\",\n      \"note\": \"Form field labels still in English — formSchemaTranslated is null in DB\"\n    },\n    {\n      \"id\": \"AC3\",\n      \"status\": \"PASS\",\n      \"note\": \"pnpm test:e2e — all sprint 1-4c specs green\"\n    }\n  ]\n}\n```\n\n**overall** is PASS only when every AC status is PASS.\n**status values:** PASS | FAIL | UNKNOWN\n\n---\n\n## problems.md\n\nWritten by Verifier when overall is not PASS. Specific, actionable.\n\n```markdown\n# Problems — [TASK_ID]\n\n## AC2: Form field labels in English\n\n**File:** `packages/llm/src/translation.ts`\n**Function:** `translatePrompts()`\n**Issue:** All prompt chunks translated in parallel via `Promise.all()`. Under slow local\nmodels, all calls time out simultaneously and fall back to English silently.\n**Evidence:** DB query shows `formSchemaTranslated IS NULL` for all 50 rows.\n**Fix needed:** Change parallel chunk processing to sequential — one prompt at a time.\n```\n\n---\n\n## evidence.md (optional)\n\nProse summary written by Builder in evidence mode. Supplements verdict.json.\n\n```markdown\n# Evidence — [TASK_ID]\n\n## Build Summary\n[What was changed and why]\n\n## AC1 — PASS\n[How it was verified, what was checked]\n\n## AC2 — FAIL\n[What was attempted, what failed, why]\n```\n\nFile v0.2.1:references/brief-template.md\n\n# Agent Brief Template\n\nUse this template when briefing any coding agent on a non-trivial task.\n\n---\n\n## Sprint/Task Brief: [TASK_ID]\n\n### Task Statement\n[One paragraph describing what needs to be built and why]\n\n### Acceptance Criteria\n\n```\nAC1: [specific, testable condition]\n     Verify: [how to check this — command, visual check, API call]\n\nAC2: [specific, testable condition]\n     Verify: [how to check this]\n\nAC3: [specific, testable condition]\n     Verify: [how to check this]\n```\n\n**Good AC examples:**\n- \"AC1: A user with locale=de sees all navigation labels in German after saving language preference\"\n- \"AC2: POST /api/v1/prompts/translate/de returns 200 with translated titles for all 50 prompts\"\n- \"AC3: `pnpm test:e2e` passes all sprint spec files with no failures\"\n\n**Bad AC examples:**\n- \"AC1: Translate the UI\" (not testable)\n- \"AC1: Make it work in German\" (vague)\n- \"AC1: Fix the translation bugs\" (not specific)\n\n### Constraints\n[What must not break — existing features, performance, API contracts]\n\n### Non-Goals\n[What is explicitly OUT of scope for this task]\n\n### Verification Approach\n[How each AC will be verified — automated tests, manual browser check, API call, etc.]\n\n---\n\n## Brief Checklist (orchestrator — run before firing agents)\n\n- [ ] ACs are explicit, testable, and frozen\n- [ ] Each AC has a verification approach\n- [ ] Constraints are listed\n- [ ] Non-goals are listed\n- [ ] Builder role is clear (who builds)\n- [ ] Verifier role is clear (different agent, fresh session)\n- [ ] Fixer role is clear (if needed)\n- [ ] Artifact path is set: `.agent/tasks/[TASK_ID]/`\n\n---\n\n## Prompts by Role\n\n### Spec-Freezer prompt\n```\nYou are the spec freezer for task [TASK_ID].\n\nRead the brief above. Write spec.md to .agent/tasks/[TASK_ID]/spec.md with:\n- Original task statement\n- Acceptance Criteria (AC1, AC2, AC3...) — copy from brief, do not modify\n- Constraints\n- Non-goals\n- Verification approach per AC\n\nDo not edit any production code. Do not start building.\n```\n\n### Builder prompt\n```\nYou are the builder for task [TASK_ID].\n\nRead .agent/tasks/[TASK_ID]/spec.md. Implement the task against the frozen ACs.\nMake the smallest safe change set that satisfies all ACs.\nDo not verify your own work. When done, hand off to the evidence phase.\n```\n\n### Verifier prompt\n```\nYou are the verifier for task [TASK_ID]. This is a fresh session.\n\nRead:\n- .agent/tasks/[TASK_ID]/spec.md (the frozen ACs)\n- .agent/tasks/[TASK_ID]/verdict.json (current verdicts)\n- .agent/tasks/[TASK_ID]/problems.md (known problems)\n\nRun independent checks. For each AC, write your verdict: PASS / FAIL / UNKNOWN.\nUpdate verdict.json. If any AC is not PASS, update problems.md with specific file/line references.\n\nDo not edit production code. Do not sign off on completion.\n```\n\n### Fixer prompt\n```\nYou are the fixer for task [TASK_ID].\n\nRead:\n- .agent/tasks/[TASK_ID]/spec.md\n- .agent/tasks/[TASK_ID]/verdict.json\n- .agent/tasks/[TASK_ID]/problems.md\n\nFix only what the verifier identified. Apply the minimal diff. Regenerate evidence.\nDo not write final sign-off. A fresh verifier will re-run after your fix.\n```\n\nFile v0.2.1:references/loopsmith-bridge.md\n\n# Loopsmith Bridge\n\nProof Loop and Loopsmith solve different parts of the same reliability problem.\n\n## Boundary\n\nUse Proof Loop when you need one task to finish with evidence:\n\n- freeze acceptance criteria\n- separate builder and verifier roles\n- store task-local artifacts\n- block self-certified done claims\n\nUse Loopsmith when the same failure pattern keeps coming back and you want to improve the agent, prompt, policy, or evaluator itself:\n\n- compare baseline vs candidate behaviour\n- run eval packs\n- score recurring failure modes\n- promote or reject changes with a ledger\n\nShort version:\n\n> Proof Loop governs a task. Loopsmith improves the agent system over time.\n\n## When Proof Loop Is Enough\n\nProof Loop is enough when:\n\n- the task is bounded\n- the acceptance criteria are clear\n- the verifier can run concrete checks\n- failures are task-specific, not a repeated agent behaviour problem\n\nExample: a UI language bug has three ACs, a verifier checks them, and all PASS.\n\n## When To Escalate To Loopsmith\n\nEscalate to Loopsmith when Proof Loop reveals a pattern such as:\n\n- builders repeatedly claim done without evidence\n- verifiers return vague narrative instead of AC-by-AC verdicts\n- fixers patch around symptoms and miss regression checks\n- orchestrators write weak or drifting acceptance criteria\n- the same class of failure appears across multiple tasks\n\nAt that point, the task is no longer the only problem. The agent behaviour needs an eval.\n\n## Turning A Proof Loop Failure Into A Loopsmith Case\n\n1. Pick the smallest representative Proof Loop artifact set.\n2. Convert the failure into an eval case:\n   - input: the task brief, spec, evidence, verdict, or problems file\n   - expected behaviour: what a good agent should do\n   - anti-goals: what the old agent did wrong\n3. Add a baseline response that shows the current behaviour.\n4. Add a candidate policy/prompt/evaluator change.\n5. Run Loopsmith and promote only if the candidate improves the evidence.\n\n## Example Mapping\n\n| Proof Loop Artifact | Loopsmith Use |\n|---|---|\n| `spec.md` | eval input for AC quality or builder discipline |\n| `verdict.json` | structural target for verifier verdict discipline |\n| `problems.md` | input for fixer minimality and regression-awareness cases |\n| `evidence.md` | input for evidence quality or false-completion checks |\n\n## Practical Rule\n\nDo not use Loopsmith for every task. That creates ceremony.\n\nUse Proof Loop by default for non-trivial work. Use Loopsmith when a repeated failure deserves a reusable improvement loop.\n\nFile v0.2.1:references/workflow.md\n\n# Proof Loop Workflow\n\n## Phase 0: Spec Freeze\n\n**Who:** Orchestrator (project lead, or whoever is briefing)\n\nBefore a single line of code is written:\n1. Write `spec.md` with explicit acceptance criteria (AC1, AC2, AC3...)\n2. Each AC must be testable by a third party who didn't build it\n3. Include: constraints, non-goals, verification approach\n4. Freeze the spec — ACs cannot change during build\n\n**spec.md must include:**\n- Original task statement\n- Acceptance Criteria (AC1, AC2, ...)\n- Constraints (what must not break)\n- Non-goals (what is explicitly out of scope)\n- Verification approach (how each AC will be checked)\n\n---\n\n## Phase 1: Build\n\n**Who:** Builder agent (fresh session)\n\n- Reads spec.md — implements against frozen ACs only\n- Makes the smallest safe change set that satisfies all ACs\n- Does not verify own work\n- Hands off to evidence phase when implementation is complete\n\n---\n\n## Phase 2: Evidence\n\n**Who:** Builder (same session, switched to evidence mode) or a fresh evidence agent\n\n- Does not change production code\n- Runs checks and records results\n- Writes `evidence.md` (prose) and updates `verdict.json` (structured)\n- Verdict per AC: PASS / FAIL / UNKNOWN\n- If FAIL or UNKNOWN: writes `problems.md` with specific file/line references\n\n**Builder hard constraints in evidence mode:**\n- Must not patch evidence to make it look better\n- Must not change production code\n- UNKNOWN is valid — it means \"could not verify\"\n\n---\n\n## Phase 3: Fresh Verify\n\n**Who:** Verifier — always a NEW session, never the same agent that built\n\n- Reads `spec.md`, `verdict.json`, `problems.md`\n- Runs independent checks against the current codebase\n- Updates `verdict.json` with verifier's findings\n- Updates `problems.md` if issues found\n- **Must not edit production code**\n- **Must not sign off on completion**\n\n---\n\n## Phase 4: Fix (if needed)\n\n**Who:** Fixer — fresh session\n\n- Reads `spec.md` + `verdict.json` + `problems.md`\n- Reconfirms each problem before editing\n- Applies minimal fix — only what the verifier identified\n- Regenerates evidence\n- **Does not write final sign-off** — goes back to Phase 3\n\n---\n\n## The Fix Loop\n\n```\nfix -> fresh verify -> fix -> fresh verify\n```\n\nContinues until verifier writes PASS for every AC.\n\nA single successful PASS does not end the loop — ALL ACs must be PASS.\n\n---\n\n## Done Criteria\n\nA task is complete when:\n- Every AC in `spec.md` has status PASS in `verdict.json`\n- `problems.md` is empty or does not exist\n- Any regression suite required by the spec passes\n\n---\n\n## Common Failure Modes (and how this prevents them)\n\n| Failure | Prevention |\n|---------|-----------|\n| Agent claims done without checking | Verifier is separate, required |\n| ACs drift during build | Spec is frozen before build starts |\n| Later sessions can't tell what was verified | Verdict artifacts stay in repo |\n| Builder judges own work | Fresh verifier is a hard rule |\n| Fix introduces new regression | Verifier reruns after every fix |\n\nFile v0.2.1:examples/role-briefs/builder.md\n\n# Builder Brief\n\nYou are the builder for task `[TASK_ID]`.\n\nRead `.agent/tasks/[TASK_ID]/spec.md` and implement only what is needed to satisfy the frozen ACs.\n\n## Responsibilities\n\n- Make the smallest safe change set.\n- Respect constraints and non-goals.\n- Update `evidence.md` with what changed and what checks you ran.\n- Leave final verdicting to a fresh verifier.\n\n## Hard Boundaries\n\n- Do not verify your own work as final.\n- Do not mark the task done.\n- Do not change frozen ACs to match your implementation.\n\nFile v0.2.1:examples/role-briefs/fixer.md\n\n# Fixer Brief\n\nYou are the fixer for task `[TASK_ID]`.\n\nRead:\n\n- `.agent/tasks/[TASK_ID]/spec.md`\n- `.agent/tasks/[TASK_ID]/verdict.json`\n- `.agent/tasks/[TASK_ID]/problems.md`\n\n## Responsibilities\n\n- Reproduce or understand each verifier-reported problem.\n- Make the minimal fix for those problems only.\n- Update `evidence.md` with changed files and checks run.\n- Hand back to a fresh verifier.\n\n## Hard Boundaries\n\n- Do not broaden scope.\n- Do not rewrite unrelated code.\n- Do not write final sign-off.\n\nFile v0.2.1:examples/role-briefs/orchestrator.md\n\n# Orchestrator Brief\n\nYou are the orchestrator for task `[TASK_ID]`.\n\nYour job is to keep the Proof Loop intact.\n\n## Responsibilities\n\n- Ensure `.agent/tasks/[TASK_ID]/spec.md` exists before build starts.\n- Confirm all ACs are explicit and testable.\n- Assign separate builder and verifier roles.\n- Refuse final completion until `scripts/check_task.py .agent/tasks/[TASK_ID]` passes.\n\n## Hard Boundaries\n\n- Do not let the builder verify their own work.\n- Do not change ACs mid-build. If the scope changes, create a new task or explicitly revise the frozen spec before implementation continues.\n- Do not accept a narrative summary as proof. Require artifacts.\n\nArchive v0.2.0: 7 files, 8110 bytes\n\nFiles: LICENSE (1068b), README.md (3271b), references/artifacts.md (2186b), references/brief-template.md (3128b), references/workflow.md (2986b), SKILL.md (2276b), _meta.json (129b)\n\nFile v0.2.0:SKILL.md\n\n---\nname: proof-loop\ndescription: \"Run evidence-gated coding sprints with frozen ACs, separated builder/verifier roles, and durable proof artifacts.\"\nmetadata:\n  version: \"0.2.0\"\n---\n\n# Proof Loop\n\nA sprint is not done until every acceptance criterion has a PASS verdict from a fresh verifier session.\n\nRead `references/workflow.md` for the full loop spec.\nRead `references/brief-template.md` for the agent brief format.\nRead `references/artifacts.md` for the artifact schema.\n\nUse for non-trivial coding tasks where completion needs proof from a fresh verifier. Skip for direct questions, tiny one-file edits, and tasks with no meaningful acceptance criteria.\n\n## The Loop\n\n```\nspec freeze -> build -> evidence -> FRESH verify -> fix -> FRESH verify\n                                         ^                      |\n                                         |______________________|\n                                         (repeat until all ACs = PASS)\n```\n\n## Four Roles — Always Separate\n\n| Role | Does | Never |\n|------|------|-------|\n| **Spec-Freezer** | Writes spec.md with explicit ACs | Edits production code |\n| **Builder** | Implements against frozen spec | Verifies own work |\n| **Verifier** | Fresh session — verdicts each AC | Edits production code |\n| **Fixer** | Minimal fix for what verifier flagged | Signs off on completion |\n\n**The verifier is always a fresh session.** The agent that built cannot judge its own work.\n\n## Acceptance Criteria Format\n\nEvery sprint brief must include explicit ACs before build starts:\n\n```\nAC1: [specific, testable condition — not a task description]\nAC2: [specific, testable condition]\nAC3: [specific, testable condition]\n```\n\nGood: \"AC1: A German-locale user sees all prompt form field labels in German\"\nBad: \"AC1: Translate the form fields\"\n\n## Sprint is DONE Only When\n\n- Every AC has a PASS verdict in the verifier's `verdict.json`\n- No problems remain in `problems.md`\n- Full regression suite passes (if applicable)\n\n## Artifacts (stored in repo)\n\n```\n.agent/tasks/<TASK_ID>/\n  spec.md         -- frozen ACs + constraints + non-goals\n  verdict.json    -- AC verdicts per phase (PASS/FAIL/UNKNOWN)\n  problems.md     -- specific failures with file/line refs (if any)\n```\n\nSee `references/artifacts.md` for schemas.\n\nFile v0.2.0:README.md\n\n# proof-loop\n\nMulti-agent sprint protocol. Prevents AI coding agents from declaring done without proof.\n\nThe core idea: spec freeze before build, role-separated agents, explicit acceptance criteria, durable verdict artifacts in the repo. A sprint is not done until every AC has a PASS verdict from a fresh verifier session.\n\n---\n\n## The Problem\n\nLarge multi-agent coding tasks fail in predictable ways:\n\n- The agent claims the job is done without durable proof\n- The same session both implements and judges its own work\n- Acceptance criteria drift during the task\n- A later session cannot tell what was actually verified\n- The verifier approves based on code review, not a live test\n\nProof loop addresses all of these.\n\n---\n\n## The Loop\n\n```\nspec freeze -> build -> evidence -> FRESH verify -> fix -> FRESH verify\n                                         ^                      |\n                                         |______________________|\n                                         (repeat until all ACs = PASS)\n```\n\n---\n\n## Four Roles\n\n| Role | Does | Never |\n|------|------|-------|\n| Spec-Freezer | Writes spec.md with explicit ACs | Edits production code |\n| Builder | Implements against frozen spec | Verifies own work |\n| Verifier | Fresh session, verdicts each AC | Edits production code |\n| Fixer | Minimal fix for what verifier flagged | Signs off on completion |\n\n**The verifier is always a fresh session.** The agent that built cannot judge its own work.\n\n---\n\n## Acceptance Criteria Format\n\n```\nAC1: A user with locale=de sees all navigation labels in German\n     Verify: browser test against demo tenant with German locale\n\nAC2: POST /api/v1/translate/de returns 200 with translated titles\n     Verify: curl command or automated test\n\nAC3: Full regression suite passes\n     Verify: pnpm test:e2e — all spec files green\n```\n\nGood: specific, testable by a third party.\nBad: \"Translate the UI\", \"Make it work in German\", \"Fix the bugs\".\n\n---\n\n## Artifacts (in repo)\n\n```\n.agent/tasks/<TASK_ID>/\n  spec.md         -- frozen ACs + constraints + non-goals\n  verdict.json    -- AC verdicts per phase (PASS/FAIL/UNKNOWN)\n  problems.md     -- specific failures with file/line refs\n  evidence.md     -- prose build summary (optional)\n```\n\n---\n\n## Installation\n\n### OpenClaw\n\n```json\n{\n  \"skills\": {\n    \"load\": {\n      \"extraDirs\": [\"/path/to/your/skills\"]\n    }\n  }\n}\n```\n\n```bash\ngit clone https://github.com/LeoStehlik/proof-loop.git /path/to/your/skills/proof-loop\n```\n\n### Claude Code / Codex\n\nCopy the `proof-loop` folder into `.agents/skills/` or `.claude/skills/`, then invoke with `/proof-loop`.\n\n---\n\n## What's Inside\n\n```\nproof-loop/\n├── SKILL.md                           Core protocol + role table\n└── references/\n    ├── workflow.md                    Full phase-by-phase spec\n    ├── brief-template.md              Copy-paste brief + role prompts\n    └── artifacts.md                   spec.md / verdict.json / problems.md schemas\n```\n\n---\n\n## Inspiration\n\nInspired by [repo-task-proof-loop](https://github.com/DenisSergeevitch/repo-task-proof-loop). Built as our own implementation with a focus on multi-agent team workflows and the lessons from running real sprints.\n\n---\n\n## License\n\nMIT - see [LICENSE](LICENSE)\n\nFile v0.2.0:_meta.json\n\n{\n  \"ownerId\": \"kn7d3r58cdxk8k0xg6jq6k65gs873c0c\",\n  \"slug\": \"proof-loop\",\n  \"version\": \"0.2.0\",\n  \"publishedAt\": 1779469510732\n}\n\nFile v0.2.0:references/artifacts.md\n\n# Artifact Schemas\n\nAll artifacts live in the repository under `.agent/tasks/<TASK_ID>/`.\n\n---\n\n## spec.md\n\nPlain Markdown. Written by Spec-Freezer before build starts. Never modified after freeze.\n\n```markdown\n# Task: [TASK_ID]\n\n## Task Statement\n[Original task description]\n\n## Acceptance Criteria\n\n**AC1:** [testable condition]\n- Verify: [how to check]\n\n**AC2:** [testable condition]\n- Verify: [how to check]\n\n## Constraints\n- [what must not break]\n\n## Non-Goals\n- [what is out of scope]\n\n## Verification Approach\n[Overall approach to verifying this task]\n```\n\n---\n\n## verdict.json\n\nWritten by Verifier. Updated after each fix loop iteration.\n\n```json\n{\n  \"task_id\": \"sprint-4c\",\n  \"phase\": \"verify\",\n  \"agent\": \"verifier\",\n  \"timestamp\": \"2026-03-30T14:00:00Z\",\n  \"overall\": \"FAIL\",\n  \"criteria\": [\n    {\n      \"id\": \"AC1\",\n      \"status\": \"PASS\",\n      \"note\": \"All nav labels confirmed German in browser test\"\n    },\n    {\n      \"id\": \"AC2\",\n      \"status\": \"FAIL\",\n      \"note\": \"Form field labels still in English — formSchemaTranslated is null in DB\"\n    },\n    {\n      \"id\": \"AC3\",\n      \"status\": \"PASS\",\n      \"note\": \"pnpm test:e2e — all sprint 1-4c specs green\"\n    }\n  ]\n}\n```\n\n**overall** is PASS only when every AC status is PASS.\n**status values:** PASS | FAIL | UNKNOWN\n\n---\n\n## problems.md\n\nWritten by Verifier when overall is not PASS. Specific, actionable.\n\n```markdown\n# Problems — [TASK_ID]\n\n## AC2: Form field labels in English\n\n**File:** `packages/llm/src/translation.ts`\n**Function:** `translatePrompts()`\n**Issue:** All prompt chunks translated in parallel via `Promise.all()`. Under slow local\nmodels, all calls time out simultaneously and fall back to English silently.\n**Evidence:** DB query shows `formSchemaTranslated IS NULL` for all 50 rows.\n**Fix needed:** Change parallel chunk processing to sequential — one prompt at a time.\n```\n\n---\n\n## evidence.md (optional)\n\nProse summary written by Builder in evidence mode. Supplements verdict.json.\n\n```markdown\n# Evidence — [TASK_ID]\n\n## Build Summary\n[What was changed and why]\n\n## AC1 — PASS\n[How it was verified, what was checked]\n\n## AC2 — FAIL\n[What was attempted, what failed, why]\n```\n\nFile v0.2.0:references/brief-template.md\n\n# Agent Brief Template\n\nUse this template when briefing any coding agent on a non-trivial task.\n\n---\n\n## Sprint/Task Brief: [TASK_ID]\n\n### Task Statement\n[One paragraph describing what needs to be built and why]\n\n### Acceptance Criteria\n\n```\nAC1: [specific, testable condition]\n     Verify: [how to check this — command, visual check, API call]\n\nAC2: [specific, testable condition]\n     Verify: [how to check this]\n\nAC3: [specific, testable condition]\n     Verify: [how to check this]\n```\n\n**Good AC examples:**\n- \"AC1: A user with locale=de sees all navigation labels in German after saving language preference\"\n- \"AC2: POST /api/v1/prompts/translate/de returns 200 with translated titles for all 50 prompts\"\n- \"AC3: `pnpm test:e2e` passes all sprint spec files with no failures\"\n\n**Bad AC examples:**\n- \"AC1: Translate the UI\" (not testable)\n- \"AC1: Make it work in German\" (vague)\n- \"AC1: Fix the translation bugs\" (not specific)\n\n### Constraints\n[What must not break — existing features, performance, API contracts]\n\n### Non-Goals\n[What is explicitly OUT of scope for this task]\n\n### Verification Approach\n[How each AC will be verified — automated tests, manual browser check, API call, etc.]\n\n---\n\n## Brief Checklist (orchestrator — run before firing agents)\n\n- [ ] ACs are explicit, testable, and frozen\n- [ ] Each AC has a verification approach\n- [ ] Constraints are listed\n- [ ] Non-goals are listed\n- [ ] Builder role is clear (who builds)\n- [ ] Verifier role is clear (different agent, fresh session)\n- [ ] Fixer role is clear (if needed)\n- [ ] Artifact path is set: `.agent/tasks/[TASK_ID]/`\n\n---\n\n## Prompts by Role\n\n### Spec-Freezer prompt\n```\nYou are the spec freezer for task [TASK_ID].\n\nRead the brief above. Write spec.md to .agent/tasks/[TASK_ID]/spec.md with:\n- Original task statement\n- Acceptance Criteria (AC1, AC2, AC3...) — copy from brief, do not modify\n- Constraints\n- Non-goals\n- Verification approach per AC\n\nDo not edit any production code. Do not start building.\n```\n\n### Builder prompt\n```\nYou are the builder for task [TASK_ID].\n\nRead .agent/tasks/[TASK_ID]/spec.md. Implement the task against the frozen ACs.\nMake the smallest safe change set that satisfies all ACs.\nDo not verify your own work. When done, hand off to the evidence phase.\n```\n\n### Verifier prompt\n```\nYou are the verifier for task [TASK_ID]. This is a fresh session.\n\nRead:\n- .agent/tasks/[TASK_ID]/spec.md (the frozen ACs)\n- .agent/tasks/[TASK_ID]/verdict.json (current verdicts)\n- .agent/tasks/[TASK_ID]/problems.md (known problems)\n\nRun independent checks. For each AC, write your verdict: PASS / FAIL / UNKNOWN.\nUpdate verdict.json. If any AC is not PASS, update problems.md with specific file/line references.\n\nDo not edit production code. Do not sign off on completion.\n```\n\n### Fixer prompt\n```\nYou are the fixer for task [TASK_ID].\n\nRead:\n- .agent/tasks/[TASK_ID]/spec.md\n- .agent/tasks/[TASK_ID]/verdict.json\n- .agent/tasks/[TASK_ID]/problems.md\n\nFix only what the verifier identified. Apply the minimal diff. Regenerate evidence.\nDo not write final sign-off. A fresh verifier will re-run after your fix.\n```\n\nFile v0.2.0:references/workflow.md\n\n# Proof Loop Workflow\n\n## Phase 0: Spec Freeze\n\n**Who:** Orchestrator (project lead, or whoever is briefing)\n\nBefore a single line of code is written:\n1. Write `spec.md` with explicit acceptance criteria (AC1, AC2, AC3...)\n2. Each AC must be testable by a third party who didn't build it\n3. Include: constraints, non-goals, verification approach\n4. Freeze the spec — ACs cannot change during build\n\n**spec.md must include:**\n- Original task statement\n- Acceptance Criteria (AC1, AC2, ...)\n- Constraints (what must not break)\n- Non-goals (what is explicitly out of scope)\n- Verification approach (how each AC will be checked)\n\n---\n\n## Phase 1: Build\n\n**Who:** Builder agent (fresh session)\n\n- Reads spec.md — implements against frozen ACs only\n- Makes the smallest safe change set that satisfies all ACs\n- Does not verify own work\n- Hands off to evidence phase when implementation is complete\n\n---\n\n## Phase 2: Evidence\n\n**Who:** Builder (same session, switched to evidence mode) or a fresh evidence agent\n\n- Does not change production code\n- Runs checks and records results\n- Writes `evidence.md` (prose) and updates `verdict.json` (structured)\n- Verdict per AC: PASS / FAIL / UNKNOWN\n- If FAIL or UNKNOWN: writes `problems.md` with specific file/line references\n\n**Builder hard constraints in evidence mode:**\n- Must not patch evidence to make it look better\n- Must not change production code\n- UNKNOWN is valid — it means \"could not verify\"\n\n---\n\n## Phase 3: Fresh Verify\n\n**Who:** Verifier — always a NEW session, never the same agent that built\n\n- Reads `spec.md`, `verdict.json`, `problems.md`\n- Runs independent checks against the current codebase\n- Updates `verdict.json` with verifier's findings\n- Updates `problems.md` if issues found\n- **Must not edit production code**\n- **Must not sign off on completion**\n\n---\n\n## Phase 4: Fix (if needed)\n\n**Who:** Fixer — fresh session\n\n- Reads `spec.md` + `verdict.json` + `problems.md`\n- Reconfirms each problem before editing\n- Applies minimal fix — only what the verifier identified\n- Regenerates evidence\n- **Does not write final sign-off** — goes back to Phase 3\n\n---\n\n## The Fix Loop\n\n```\nfix -> fresh verify -> fix -> fresh verify\n```\n\nContinues until verifier writes PASS for every AC.\n\nA single successful PASS does not end the loop — ALL ACs must be PASS.\n\n---\n\n## Done Criteria\n\nA task is complete when:\n- Every AC in `spec.md` has status PASS in `verdict.json`\n- `problems.md` is empty or does not exist\n- Any regression suite required by the spec passes\n\n---\n\n## Common Failure Modes (and how this prevents them)\n\n| Failure | Prevention |\n|---------|-----------|\n| Agent claims done without checking | Verifier is separate, required |\n| ACs drift during build | Spec is frozen before build starts |\n| Later sessions can't tell what was verified | Verdict artifacts stay in repo |\n| Builder judges own work | Fresh verifier is a hard rule |\n| Fix introduces new regression | Verifier reruns after every fix |\n\nFile v0.2.0:LICENSE\n\nMIT License\n\nCopyright (c) 2026 Leo Stehlik\n\nPermission is hereby granted, free of charge, to any person obtaining a copy\nof this software and associated documentation files (the \"Software\"), to deal\nin the Software without restriction, including without limitation the rights\nto use, copy, modify, merge, publish, distribute, sublicense, and/or sell\ncopies of the Software, and to permit persons to whom the Software is\nfurnished to do so, subject to the following conditions:\n\nThe above copyright notice and this permission notice shall be included in all\ncopies or substantial portions of the Software.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR\nIMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,\nFITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE\nAUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER\nLIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,\nOUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE\nSOFTWARE.\n\nArchive v0.1.0: 7 files, 8169 bytes\n\nFiles: LICENSE (1068b), README.md (3271b), references/artifacts.md (2186b), references/brief-template.md (3128b), references/workflow.md (2986b), SKILL.md (2423b), _meta.json (129b)\n\nFile v0.1.0:SKILL.md\n\n---\nname: proof-loop\ndescription: Multi-agent sprint protocol that prevents AI coding agents from declaring done without proof. Enforces spec freeze before build, role-separated subagents (builder never verifies own work), explicit acceptance criteria (AC1, AC2...), and durable verdict artifacts in the repo. Use when briefing agents on any non-trivial coding task, sprint, or feature where you need verifiable proof of completion. Works with Codex, Claude Code, OpenClaw subagents, or any multi-agent setup.\n---\n\n# Proof Loop\n\nA sprint is not done until every acceptance criterion has a PASS verdict from a fresh verifier session.\n\nRead `references/workflow.md` for the full loop spec.\nRead `references/brief-template.md` for the agent brief format.\nRead `references/artifacts.md` for the artifact schema.\n\n## The Loop\n\n```\nspec freeze -> build -> evidence -> FRESH verify -> fix -> FRESH verify\n                                         ^                      |\n                                         |______________________|\n                                         (repeat until all ACs = PASS)\n```\n\n## Four Roles — Always Separate\n\n| Role | Does | Never |\n|------|------|-------|\n| **Spec-Freezer** | Writes spec.md with explicit ACs | Edits production code |\n| **Builder** | Implements against frozen spec | Verifies own work |\n| **Verifier** | Fresh session — verdicts each AC | Edits production code |\n| **Fixer** | Minimal fix for what verifier flagged | Signs off on completion |\n\n**The verifier is always a fresh session.** The agent that built cannot judge its own work.\n\n## Acceptance Criteria Format\n\nEvery sprint brief must include explicit ACs before build starts:\n\n```\nAC1: [specific, testable condition — not a task description]\nAC2: [specific, testable condition]\nAC3: [specific, testable condition]\n```\n\nGood: \"AC1: A German-locale user sees all prompt form field labels in German\"\nBad: \"AC1: Translate the form fields\"\n\n## Sprint is DONE Only When\n\n- Every AC has a PASS verdict in the verifier's `verdict.json`\n- No problems remain in `problems.md`\n- Full regression suite passes (if applicable)\n\n## Artifacts (stored in repo)\n\n```\n.agent/tasks/<TASK_ID>/\n  spec.md         -- frozen ACs + constraints + non-goals\n  verdict.json    -- AC verdicts per phase (PASS/FAIL/UNKNOWN)\n  problems.md     -- specific failures with file/line refs (if any)\n```\n\nSee `references/artifacts.md` for schemas.\n\nFile v0.1.0:README.md\n\n# proof-loop\n\nMulti-agent sprint protocol. Prevents AI coding agents from declaring done without proof.\n\nThe core idea: spec freeze before build, role-separated agents, explicit acceptance criteria, durable verdict artifacts in the repo. A sprint is not done until every AC has a PASS verdict from a fresh verifier session.\n\n---\n\n## The Problem\n\nLarge multi-agent coding tasks fail in predictable ways:\n\n- The agent claims the job is done without durable proof\n- The same session both implements and judges its own work\n- Acceptance criteria drift during the task\n- A later session cannot tell what was actually verified\n- The verifier approves based on code review, not a live test\n\nProof loop addresses all of these.\n\n---\n\n## The Loop\n\n```\nspec freeze -> build -> evidence -> FRESH verify -> fix -> FRESH verify\n                                         ^                      |\n                                         |______________________|\n                                         (repeat until all ACs = PASS)\n```\n\n---\n\n## Four Roles\n\n| Role | Does | Never |\n|------|------|-------|\n| Spec-Freezer | Writes spec.md with explicit ACs | Edits production code |\n| Builder | Implements against frozen spec | Verifies own work |\n| Verifier | Fresh session, verdicts each AC | Edits production code |\n| Fixer | Minimal fix for what verifier flagged | Signs off on completion |\n\n**The verifier is always a fresh session.** The agent that built cannot judge its own work.\n\n---\n\n## Acceptance Criteria Format\n\n```\nAC1: A user with locale=de sees all navigation labels in German\n     Verify: browser test against demo tenant with German locale\n\nAC2: POST /api/v1/translate/de returns 200 with translated titles\n     Verify: curl command or automated test\n\nAC3: Full regression suite passes\n     Verify: pnpm test:e2e — all spec files green\n```\n\nGood: specific, testable by a third party.\nBad: \"Translate the UI\", \"Make it work in German\", \"Fix the bugs\".\n\n---\n\n## Artifacts (in repo)\n\n```\n.agent/tasks/<TASK_ID>/\n  spec.md         -- frozen ACs + constraints + non-goals\n  verdict.json    -- AC verdicts per phase (PASS/FAIL/UNKNOWN)\n  problems.md     -- specific failures with file/line refs\n  evidence.md     -- prose build summary (optional)\n```\n\n---\n\n## Installation\n\n### OpenClaw\n\n```json\n{\n  \"skills\": {\n    \"load\": {\n      \"extraDirs\": [\"/path/to/your/skills\"]\n    }\n  }\n}\n```\n\n```bash\ngit clone https://github.com/LeoStehlik/proof-loop.git /path/to/your/skills/proof-loop\n```\n\n### Claude Code / Codex\n\nCopy the `proof-loop` folder into `.agents/skills/` or `.claude/skills/`, then invoke with `/proof-loop`.\n\n---\n\n## What's Inside\n\n```\nproof-loop/\n├── SKILL.md                           Core protocol + role table\n└── references/\n    ├── workflow.md                    Full phase-by-phase spec\n    ├── brief-template.md              Copy-paste brief + role prompts\n    └── artifacts.md                   spec.md / verdict.json / problems.md schemas\n```\n\n---\n\n## Inspiration\n\nInspired by [repo-task-proof-loop](https://github.com/DenisSergeevitch/repo-task-proof-loop). Built as our own implementation with a focus on multi-agent team workflows and the lessons from running real sprints.\n\n---\n\n## License\n\nMIT - see [LICENSE](LICENSE)\n\nFile v0.1.0:_meta.json\n\n{\n  \"ownerId\": \"kn7d3r58cdxk8k0xg6jq6k65gs873c0c\",\n  \"slug\": \"proof-loop\",\n  \"version\": \"0.1.0\",\n  \"publishedAt\": 1779320244857\n}\n\nFile v0.1.0:references/artifacts.md\n\n# Artifact Schemas\n\nAll artifacts live in the repository under `.agent/tasks/<TASK_ID>/`.\n\n---\n\n## spec.md\n\nPlain Markdown. Written by Spec-Freezer before build starts. Never modified after freeze.\n\n```markdown\n# Task: [TASK_ID]\n\n## Task Statement\n[Original task description]\n\n## Acceptance Criteria\n\n**AC1:** [testable condition]\n- Verify: [how to check]\n\n**AC2:** [testable condition]\n- Verify: [how to check]\n\n## Constraints\n- [what must not break]\n\n## Non-Goals\n- [what is out of scope]\n\n## Verification Approach\n[Overall approach to verifying this task]\n```\n\n---\n\n## verdict.json\n\nWritten by Verifier. Updated after each fix loop iteration.\n\n```json\n{\n  \"task_id\": \"sprint-4c\",\n  \"phase\": \"verify\",\n  \"agent\": \"verifier\",\n  \"timestamp\": \"2026-03-30T14:00:00Z\",\n  \"overall\": \"FAIL\",\n  \"criteria\": [\n    {\n      \"id\": \"AC1\",\n      \"status\": \"PASS\",\n      \"note\": \"All nav labels confirmed German in browser test\"\n    },\n    {\n      \"id\": \"AC2\",\n      \"status\": \"FAIL\",\n      \"note\": \"Form field labels still in English — formSchemaTranslated is null in DB\"\n    },\n    {\n      \"id\": \"AC3\",\n      \"status\": \"PASS\",\n      \"note\": \"pnpm test:e2e — all sprint 1-4c specs green\"\n    }\n  ]\n}\n```\n\n**overall** is PASS only when every AC status is PASS.\n**status values:** PASS | FAIL | UNKNOWN\n\n---\n\n## problems.md\n\nWritten by Verifier when overall is not PASS. Specific, actionable.\n\n```markdown\n# Problems — [TASK_ID]\n\n## AC2: Form field labels in English\n\n**File:** `packages/llm/src/translation.ts`\n**Function:** `translatePrompts()`\n**Issue:** All prompt chunks translated in parallel via `Promise.all()`. Under slow local\nmodels, all calls time out simultaneously and fall back to English silently.\n**Evidence:** DB query shows `formSchemaTranslated IS NULL` for all 50 rows.\n**Fix needed:** Change parallel chunk processing to sequential — one prompt at a time.\n```\n\n---\n\n## evidence.md (optional)\n\nProse summary written by Builder in evidence mode. Supplements verdict.json.\n\n```markdown\n# Evidence — [TASK_ID]\n\n## Build Summary\n[What was changed and why]\n\n## AC1 — PASS\n[How it was verified, what was checked]\n\n## AC2 — FAIL\n[What was attempted, what failed, why]\n```\n\nFile v0.1.0:references/brief-template.md\n\n# Agent Brief Template\n\nUse this template when briefing any coding agent on a non-trivial task.\n\n---\n\n## Sprint/Task Brief: [TASK_ID]\n\n### Task Statement\n[One paragraph describing what needs to be built and why]\n\n### Acceptance Criteria\n\n```\nAC1: [specific, testable condition]\n     Verify: [how to check this — command, visual check, API call]\n\nAC2: [specific, testable condition]\n     Verify: [how to check this]\n\nAC3: [specific, testable condition]\n     Verify: [how to check this]\n```\n\n**Good AC examples:**\n- \"AC1: A user with locale=de sees all navigation labels in German after saving language preference\"\n- \"AC2: POST /api/v1/prompts/translate/de returns 200 with translated titles for all 50 prompts\"\n- \"AC3: `pnpm test:e2e` passes all sprint spec files with no failures\"\n\n**Bad AC examples:**\n- \"AC1: Translate the UI\" (not testable)\n- \"AC1: Make it work in German\" (vague)\n- \"AC1: Fix the translation bugs\" (not specific)\n\n### Constraints\n[What must not break — existing features, performance, API contracts]\n\n### Non-Goals\n[What is explicitly OUT of scope for this task]\n\n### Verification Approach\n[How each AC will be verified — automated tests, manual browser check, API call, etc.]\n\n---\n\n## Brief Checklist (orchestrator — run before firing agents)\n\n- [ ] ACs are explicit, testable, and frozen\n- [ ] Each AC has a verification approach\n- [ ] Constraints are listed\n- [ ] Non-goals are listed\n- [ ] Builder role is clear (who builds)\n- [ ] Verifier role is clear (different agent, fresh session)\n- [ ] Fixer role is clear (if needed)\n- [ ] Artifact path is set: `.agent/tasks/[TASK_ID]/`\n\n---\n\n## Prompts by Role\n\n### Spec-Freezer prompt\n```\nYou are the spec freezer for task [TASK_ID].\n\nRead the brief above. Write spec.md to .agent/tasks/[TASK_ID]/spec.md with:\n- Original task statement\n- Acceptance Criteria (AC1, AC2, AC3...) — copy from brief, do not modify\n- Constraints\n- Non-goals\n- Verification approach per AC\n\nDo not edit any production code. Do not start building.\n```\n\n### Builder prompt\n```\nYou are the builder for task [TASK_ID].\n\nRead .agent/tasks/[TASK_ID]/spec.md. Implement the task against the frozen ACs.\nMake the smallest safe change set that satisfies all ACs.\nDo not verify your own work. When done, hand off to the evidence phase.\n```\n\n### Verifier prompt\n```\nYou are the verifier for task [TASK_ID]. This is a fresh session.\n\nRead:\n- .agent/tasks/[TASK_ID]/spec.md (the frozen ACs)\n- .agent/tasks/[TASK_ID]/verdict.json (current verdicts)\n- .agent/tasks/[TASK_ID]/problems.md (known problems)\n\nRun independent checks. For each AC, write your verdict: PASS / FAIL / UNKNOWN.\nUpdate verdict.json. If any AC is not PASS, update problems.md with specific file/line references.\n\nDo not edit production code. Do not sign off on completion.\n```\n\n### Fixer prompt\n```\nYou are the fixer for task [TASK_ID].\n\nRead:\n- .agent/tasks/[TASK_ID]/spec.md\n- .agent/tasks/[TASK_ID]/verdict.json\n- .agent/tasks/[TASK_ID]/problems.md\n\nFix only what the verifier identified. Apply the minimal diff. Regenerate evidence.\nDo not write final sign-off. A fresh verifier will re-run after your fix.\n```\n\nFile v0.1.0:references/workflow.md\n\n# Proof Loop Workflow\n\n## Phase 0: Spec Freeze\n\n**Who:** Orchestrator (project lead, or whoever is briefing)\n\nBefore a single line of code is written:\n1. Write `spec.md` with explicit acceptance criteria (AC1, AC2, AC3...)\n2. Each AC must be testable by a third party who didn't build it\n3. Include: constraints, non-goals, verification approach\n4. Freeze the spec — ACs cannot change during build\n\n**spec.md must include:**\n- Original task statement\n- Acceptance Criteria (AC1, AC2, ...)\n- Constraints (what must not break)\n- Non-goals (what is explicitly out of scope)\n- Verification approach (how each AC will be checked)\n\n---\n\n## Phase 1: Build\n\n**Who:** Builder agent (fresh session)\n\n- Reads spec.md — implements against frozen ACs only\n- Makes the smallest safe change set that satisfies all ACs\n- Does not verify own work\n- Hands off to evidence phase when implementation is complete\n\n---\n\n## Phase 2: Evidence\n\n**Who:** Builder (same session, switched to evidence mode) or a fresh evidence agent\n\n- Does not change production code\n- Runs checks and records results\n- Writes `evidence.md` (prose) and updates `verdict.json` (structured)\n- Verdict per AC: PASS / FAIL / UNKNOWN\n- If FAIL or UNKNOWN: writes `problems.md` with specific file/line references\n\n**Builder hard constraints in evidence mode:**\n- Must not patch evidence to make it look better\n- Must not change production code\n- UNKNOWN is valid — it means \"could not verify\"\n\n---\n\n## Phase 3: Fresh Verify\n\n**Who:** Verifier — always a NEW session, never the same agent that built\n\n- Reads `spec.md`, `verdict.json`, `problems.md`\n- Runs independent checks against the current codebase\n- Updates `verdict.json` with verifier's findings\n- Updates `problems.md` if issues found\n- **Must not edit production code**\n- **Must not sign off on completion**\n\n---\n\n## Phase 4: Fix (if needed)\n\n**Who:** Fixer — fresh session\n\n- Reads `spec.md` + `verdict.json` + `problems.md`\n- Reconfirms each problem before editing\n- Applies minimal fix — only what the verifier identified\n- Regenerates evidence\n- **Does not write final sign-off** — goes back to Phase 3\n\n---\n\n## The Fix Loop\n\n```\nfix -> fresh verify -> fix -> fresh verify\n```\n\nContinues until verifier writes PASS for every AC.\n\nA single successful PASS does not end the loop — ALL ACs must be PASS.\n\n---\n\n## Done Criteria\n\nA task is complete when:\n- Every AC in `spec.md` has status PASS in `verdict.json`\n- `problems.md` is empty or does not exist\n- Any regression suite required by the spec passes\n\n---\n\n## Common Failure Modes (and how this prevents them)\n\n| Failure | Prevention |\n|---------|-----------|\n| Agent claims done without checking | Verifier is separate, required |\n| ACs drift during build | Spec is frozen before build starts |\n| Later sessions can't tell what was verified | Verdict artifacts stay in repo |\n| Builder judges own work | Fresh verifier is a hard rule |\n| Fix introduces new regression | Verifier reruns after every fix |\n\nFile v0.1.0:LICENSE\n\nMIT License\n\nCopyright (c) 2026 Leo Stehlik\n\nPermission is hereby granted, free of charge, to any person obtaining a copy\nof this software and associated documentation files (the \"Software\"), to deal\nin the Software without restriction, including without limitation the rights\nto use, copy, modify, merge, publish, distribute, sublicense, and/or sell\ncopies of the Software, and to permit persons to whom the Software is\nfurnished to do so, subject to the following conditions:\n\nThe above copyright notice and this permission notice shall be included in all\ncopies or substantial portions of the Software.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR\nIMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,\nFITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE\nAUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER\nLIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,\nOUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE\nSOFTWARE.","readmeExcerpt":"Skill: Proof Loop Clawhub V030 Owner: leostehlik Summary: Run evidence-gated coding sprints with frozen ACs, separated builder/verifier roles, and durable proof artifacts. Tags: agents:0.2.4, coding:0.2.4, latest:0.3.0, openclaw:0.2.4, verification:0.2.4 Version history: v0.3.0 | 2026-09-01T01:00:03.579Z | user v0.3.0 adoption kit: README use-it-today path, docs/adoption-kit.md, compact completed proof artifact under","codeSnippets":[],"executableExamples":[{"language":"text","snippet":"spec freeze -> build -> evidence -> FRESH verify -> fix -> FRESH verify\n                                         ^                      |\n                                         |______________________|\n                                         (repeat until all ACs = PASS)"},{"language":"text","snippet":"AC1: [specific, testable condition — not a task description]\nAC2: [specific, testable condition]\nAC3: [specific, testable condition]"},{"language":"bash","snippet":"python3 scripts/init_task.py TASK_ID --title \"Task title\"\npython3 scripts/check_task.py .agent/tasks/TASK_ID"},{"language":"text","snippet":".agent/tasks/<TASK_ID>/\n  spec.md         -- frozen ACs + constraints + non-goals\n  verdict.json    -- AC verdicts per phase (PASS/FAIL/UNKNOWN)\n  problems.md     -- specific failures with file/line refs (if any)"},{"language":"bash","snippet":"bin/proof-loop check examples/adoption-kit/.agent/tasks/checkout-empty-state-proof\nbin/proof-loop report examples/adoption-kit/.agent/tasks/checkout-empty-state-proof --format md"},{"language":"text","snippet":".agent/tasks/ui-language-fix/"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: proof-loop\ndescription: \"Run evidence-gated coding sprints with frozen ACs, separated builder/verifier roles, and durable proof artifacts.\"\nmetadata:\n  version: \"0.3.0\"\n---\n# Proof Loop\n\nA sprint is not done until every acceptance criterion has a PASS verdict from a fresh verifier session.\n\nRead `references/workflow.md` for the full loop spec.\nRead `references/brief-template.md` for the agent brief format.\nRead `references/artifacts.md` for the artifact schema.\nRead `references/loopsmith-bridge.md` when deciding whether a repeated Proof Loop failure should become a Loopsmith eval case.\n\n\n## Activation and Safety Boundaries\n\nUse this skill only when the user explicitly asks for Proof Loop, proof artifacts, acceptance-criterion verification, fresh verifier separation, or an evidence-gated coding sprint. Do not activate it for ordinary code edits where the user did not request this protocol.\n\nProof Loop may create or update files under `.agent/tasks/<TASK_ID>/` in the current repository. Confirm the task id and repository root before creating artifacts. Do not publish artifacts, run remote validation, change repository permissions, moderate users, request full-access sandboxing, or touch credentials unless the user explicitly asks for that separate action in the current conversation.\n\nRun helper scripts with the least privilege available. If a command could modify source files outside `.agent/tasks/<TASK_ID>/`, ask first and record the command in the evidence artifact.\n\n## The Loop\n\n```\nspec freeze -> build -> evidence -> FRESH verify -> fix -> FRESH verify\n                                         ^                      |\n                                         |______________________|\n                                         (repeat until all ACs = PASS)\n```\n\n## Four Roles — Always Separate\n\n| Role | Does | Never |\n|------|------|-------|\n| **Spec-Freezer** | Writes spec.md with explicit ACs | Edits production code |\n| **Builder** | Implements against frozen spec | Verifies own work |\n| **Verifier** | Fresh session — verdicts each AC | Edits production code |\n| **Fixer** | Minimal fix for what verifier flagged | Signs off on completion |\n\n**The verifier is always a fresh session.** The agent that built cannot judge its own work.\n\n## Acceptance Criteria Format\n\nEvery sprint brief must include explicit ACs before build starts:\n\n```\nAC1: [specific, testable condition — not a task description]\nAC2: [specific, testable condition]\nAC3: [specific, testable condition]\n```\n\nGood: \"AC1: A German-locale user sees all prompt form field labels in German\"\nBad: \"AC1: Translate the form fields\"\n\n## Helper Scripts\n\nUse these when the repository has the `proof-loop` folder available:\n\n```bash\npython3 scripts/init_task.py TASK_ID --title \"Task title\"\npython3 scripts/check_task.py .agent/tasks/TASK_ID\n```\n\n`check_task.py` is the mechanical done gate. It returns success only when the verifier artifacts show every AC as PASS and no open problems remain.\n\n## S"},{"path":"examples/adoption-kit/README.md","content":"# Adoption Kit Example\n\nThis is the smallest completed proof folder a repo owner can copy when teaching an agent the Proof Loop shape.\n\nRead:\n\n1. `.agent/tasks/checkout-empty-state-proof/spec.md`\n2. `.agent/tasks/checkout-empty-state-proof/evidence.md`\n3. `.agent/tasks/checkout-empty-state-proof/verdict.json`\n4. `.agent/tasks/checkout-empty-state-proof/problems.md`\n\nRun:\n\n```bash\nbin/proof-loop check examples/adoption-kit/.agent/tasks/checkout-empty-state-proof\nbin/proof-loop report examples/adoption-kit/.agent/tasks/checkout-empty-state-proof --format md\n```"},{"path":"examples/example-task/README.md","content":"# Example Task: UI Language Fix\n\nThis example shows what a completed Proof Loop task looks like inside a repository.\n\nThe task artifacts live under:\n\n```text\n.agent/tasks/ui-language-fix/\n```\n\nRead them in this order:\n\n1. `spec.md` - frozen acceptance criteria and verification plan\n2. `evidence.md` - what was changed and checked\n3. `verdict.json` - structured verifier result\n4. `problems.md` - empty because the final verifier pass found no open issues\n\nThe example is intentionally small. Its job is to prove the artifact shape, not to ship a real app."},{"path":"examples/README.md","content":"# Proof Loop Examples\n\nStart here if you want proof instead of prose.\n\n| Example | What it shows |\n| --- | --- |\n| [`demo-repo/`](demo-repo/) | A small runnable task with a real file check and passing proof artifacts. |\n| [`adoption-kit/`](adoption-kit/) | A compact completed task folder that shows the copy-paste adoption shape. |\n| [`example-task/`](example-task/) | A completed task folder with spec, verdict, evidence, and problems files. |\n| [`role-briefs/`](role-briefs/) | Copy-paste briefs for orchestrator, spec freezer, builder, verifier, and fixer roles. |\n\nFast path:\n\n```bash\nmake test\nbin/proof-loop check examples/demo-repo/.agent/tasks/nav-labels-proof\nbin/proof-loop check examples/adoption-kit/.agent/tasks/checkout-empty-state-proof\nbin/proof-loop report examples/demo-repo/.agent/tasks/nav-labels-proof --format md\n```"},{"path":"README.md","content":"# Proof Loop\n\n![Tests](https://github.com/LeoStehlik/proof-loop/actions/workflows/test.yml/badge.svg)\n\n**Make AI coding agents prove when work is done.**\n\n**v0.3 focus:** use Proof Loop in a real repo today: copy the adoption kit, freeze ACs, run a fresh verifier, and ship with durable proof artifacts instead of a completion story.\n\nProof Loop is a repo-local verification protocol for AI coding agents. It freezes acceptance criteria before the build, separates builder and verifier roles, records durable proof artifacts in the repo, and refuses to call work done until every acceptance criterion has a fresh PASS verdict.\n\nUse it when an agent, team, or multi-agent sprint needs a clear boundary between “looks done” and verified work. Because the protocol is just files plus role discipline, it works with OpenClaw, Hermes, Codex, OpenCode, Claude Code, or any other harness that can read and write a repository.\n\n## Start Here\n\nRun the current proof gate first:\n\n```bash\ngit clone https://github.com/LeoStehlik/proof-loop.git\ncd proof-loop\nmake test\nbin/proof-loop doctor\n```\n\nStart a proof-tracked task in any repo:\n\n```bash\nbin/proof-loop init maintenance-check --title \"Refresh the README proof\" --root /path/to/repo\nbin/proof-loop check /path/to/repo/.agent/tasks/maintenance-check\n```\n\nThe first check should fail until a fresh verifier records `PASS` for every frozen acceptance criterion and clears `problems.md`. That failure is the point: Proof Loop gives agents a mechanical done gate instead of a confident paragraph.\n\n## Works With\n\nProof Loop is harness-agnostic. Use it with any coding agent that can read files, write files, run commands, and hand verification to a fresh session.\n\nKnown-fit surfaces:\n\n- Codex\n- Claude Code\n- OpenClaw\n- OpenCode\n- Hermes\n- custom multi-agent runners\n\nThe repo includes copy-paste guide templates for several harnesses under `templates/`, plus role briefs under `examples/role-briefs/`.\n\n\n## Activation and Safety\n\nUse Proof Loop when the user explicitly wants an evidence-gated coding sprint, frozen acceptance criteria, fresh verifier separation, or durable proof artifacts. It is not meant to silently wrap every code change.\n\nThe bundled helpers create and check repo-local files under `.agent/tasks/<TASK_ID>/`. Review the task id and repository root before running them. Publishing reports, using remote workers, changing permissions, or running elevated commands are outside this skill unless requested separately.\n\n## Use Cases\n\n- keep AI coding agents honest when they claim a task is done\n- freeze acceptance criteria before implementation starts\n- separate builder and verifier roles in multi-agent coding work\n- leave proof artifacts in the repo for future review\n\n![Animated terminal demo: Proof Loop doctor, check, and report commands](assets/proof-loop-terminal-demo.svg)\n\nProof artifacts and role-brief examples are indexed in [`examples/README.md`](examples/README.md).\n\n## Use It Today\n\nDrop Proof Loop into a task where an ag"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":1417,"uniquenessScore":41,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T05:34:21.767Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T05:34:21.767Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T07:39:53.027Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}