{"id":"ff341601-1e5b-48f4-ad6e-67fb276df00a","entityType":"agent","slug":"clawhub-chpomob-adversarial-code-loop","name":"adversarial-code-loop","canonicalUrl":"https://www.xpersona.co/agent/clawhub-chpomob-adversarial-code-loop","canonicalPath":"/agent/clawhub-chpomob-adversarial-code-loop","generatedAt":"2026-10-09T16:06:24.296Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-09T13:35:10.043Z","emptyReason":null},"description":"BUILD → REVIEW → (FIX → VERIFY)^N → ARBITER on isolated git branches. Git-native: each loop runs on its own branch, changes are committed, reviews inspect git diffs. Skill: adversarial-code-loop Owner: chpomob Summary: BUILD → REVIEW → (FIX → VERIFY)^N → ARBITER on isolated git branches. Git-native: each loop runs on its own branch, changes are committed, reviews inspect git diffs. Tags: latest:0.1.0 Version history: v0.1.0 | 2026-08-03T18:12:20.116Z | auto Adversarial Code Loop v4 is a major update making the workflow fully git-native. - Each code-review loop runs on its own iso","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 2.6K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s17435m3chty5jmw4jhpkyhnb58brn8g:adversarial-code-loop","sourceUrl":"https://clawhub.ai/chpomob/adversarial-code-loop","homepage":"https://clawhub.ai/chpomob/skills/adversarial-code-loop","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/chpomob/adversarial-code-loop","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/chpomob/skills/adversarial-code-loop","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":68,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"BUILD → REVIEW → (FIX → VERIFY)^N → ARBITER on isolated git branches. Git-native: each loop runs on its own branch, changes are committed, reviews inspect git d"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T13:35:10.043Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T13:35:10.043Z","emptyReason":null},"stars":null,"forks":null,"downloads":2555,"packageName":null,"latestVersion":"0.1.0","tractionLabel":"2.6K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T13:35:10.043Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T13:35:10.043Z","lastCrawledAt":"2026-10-09T13:35:10.043Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T13:35:10.043Z","lastVerifiedAt":null,"highlights":[{"version":"0.1.0","createdAt":"2026-08-03T18:12:20.116Z","changelog":"Adversarial Code Loop v4 is a major update making the workflow fully git-native. - Each code-review loop runs on its own isolated git branch; changes are committed and reviewed via real git diffs, not concatenated files. - All phases (BUILD, REVIEW, FIX, VERIFY, ARBITER) are implemented as shell-invocable steps, with phase responsibilities clarified. - Adds optional gates: custom build/test commands can now block progress at key stages. - Robust resume support: interrupted sessions can be continued with a single flag. - All merge/reject/arbitrate outcomes are precisely tracked by commit/message and machine-readable artifacts. - CLI flags and environment variable handling explicitly defined and streamlined for better configuration.","fileCount":113,"zipByteSize":215642}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17435m3chty5jmw4jhpkyhnb58brn8g:adversarial-code-loop","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-chpomob-adversarial-code-loop/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-chpomob-adversarial-code-loop/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-chpomob-adversarial-code-loop/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-chpomob-adversarial-code-loop/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-chpomob-adversarial-code-loop/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-chpomob-adversarial-code-loop/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-09T16:06:24.295Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-chpomob-adversarial-code-loop/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-chpomob-adversarial-code-loop/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-chpomob-adversarial-code-loop/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-chpomob-adversarial-code-loop/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-09T13:35:10.043Z","emptyReason":null},"readme":"Skill: adversarial-code-loop\n\nOwner: chpomob\n\nSummary: BUILD → REVIEW → (FIX → VERIFY)^N → ARBITER on isolated git branches. Git-native: each loop runs on its own branch, changes are committed, reviews inspect git diffs.\n\nTags: latest:0.1.0\n\nVersion history:\n\nv0.1.0 | 2026-08-03T18:12:20.116Z | auto\n\nAdversarial Code Loop v4 is a major update making the workflow fully git-native.\n\n- Each code-review loop runs on its own isolated git branch; changes are committed and reviewed via real git diffs, not concatenated files.\n- All phases (BUILD, REVIEW, FIX, VERIFY, ARBITER) are implemented as shell-invocable steps, with phase responsibilities clarified.\n- Adds optional gates: custom build/test commands can now block progress at key stages.\n- Robust resume support: interrupted sessions can be continued with a single flag.\n- All merge/reject/arbitrate outcomes are precisely tracked by commit/message and machine-readable artifacts.\n- CLI flags and environment variable handling explicitly defined and streamlined for better configuration.\n\nArchive index:\n\nArchive v0.1.0: 113 files, 215642 bytes\n\nFiles: _retrospective (0b), _retrospective/ISSUES.md (9288b), .gitignore (248b), IMPLEMENTATION_PLAN.md (9881b), LICENSE (665b), README.md (1973b), references (0b), references/batch-splitting-strategy.md (2837b), references/bug-fix-after-review-workflow.md (1834b), references/builder-permission-stall.md (2714b), references/cascade-fix-pattern.md (1864b), references/claude-p-migration-pattern.md (3921b), references/claude-tmux-adversarial-review.md (3832b), references/claude-tmux-fixes-2026-07-14.md (2658b), references/claude-tmux-plan-challenge-compat.md (2111b), references/claude-tmux-prompt-hygiene.md (3775b), references/codex-AGY-pattern.md (7947b), references/codex-deepseek-pattern.md (4250b), references/codex-dev-pi-zai-review.md (1660b), references/delegate-task-review-timeout.md (1572b), references/direct-stdin-pipeline.md (2632b), references/fable5-reviewer-pattern.md (2950b), references/fable5-timeout-recovery.md (3373b), references/fable5-usage-limit.md (1340b), references/failure-modes.md (2900b), references/fixer-sandbox-verifier-deadlock.md (1757b), references/fork-as-live-install.md (2376b), references/full-pipeline-validated.md (2202b), references/git-history-cleanup.md (2002b), references/git-workflow-v4.md (4721b), references/github-secret-scanning-bypass.md (3455b), references/glm-dev-claude-review-pairing.md (1356b), references/glm5-pi-prose-behavior.md (2124b), references/god-module-refactor-workflow.md (1879b), references/host-testing-embedded-c.md (2541b), references/implementation-plan-v4.md (1652b), references/monitoring-long-running-loops.md (1415b), references/multi-repo-planning.md (1958b), references/partial-merge-gap-fill.md (5040b), references/phase-challenge-prompt-reduction.md (2123b), references/pi-auth-setup.md (1998b), references/pi-sentinel-limitation.md (1425b), references/pi-wrong-repo-cwd.md (2911b), references/plan-format-constraints.md (1025b), references/post-loop-extraction-workflow.md (2472b), references/pre-pr-cleanup.md (7384b), references/prompt-injection-threat-model.md (4957b), references/quota-aware-orchestration.md (8108b), references/quota-aware-provider-registry.md (4296b), references/rebase-pr-onto-upstream.md (5080b), references/review-on-committed-code.md (1817b), references/review-verify-timeout-patch.md (2743b), references/sequential-step-pattern.md (2360b), references/single-file-fix-workflow.md (3294b), references/spec-writing-for-adversarial-loops.md (4564b), references/step-chaining-workflow.md (4719b), references/timeout-recovery-workflow.md (2287b), references/tui-terminal-safety.md (3293b), references/ui-port-workflow.md (2061b), references/wrapper-failures.md (8111b), scripts (0b), scripts/__init__.py (477b), scripts/adversarial_loop_v4.py (71596b), scripts/adversarial_loop.py (1043b), scripts/claude-tmux-fix-spec.md (7630b), scripts/install.sh (1968b), scripts/phases (0b), scripts/phases/__init__.py (350b), scripts/phases/integration_gate.py (6906b), scripts/phases/phase_arbiter.py (6093b), scripts/phases/phase_build.py (3586b), scripts/phases/phase_fix.py (3194b), scripts/phases/phase_git.py (3786b), scripts/phases/phase_review.py (11190b), scripts/phases/phase_verify.py (7754b), scripts/phases/runtime.py (2381b), scripts/phases/test_phases.py (13068b), scripts/repos.manifest.yaml (676b), scripts/test_loop_fixes.py (20990b), skill-card.md (2664b)\n\nFile v0.1.0:SKILL.md\n\n---\nname: adversarial-code-loop\ndescription: \"BUILD → REVIEW → (FIX → VERIFY)^N → ARBITER on isolated git branches. Git-native: each loop runs on its own branch, changes are committed, reviews inspect git diffs.\"\nversion: 4.2.0\nauthor: Hermes Agent\nlicense: 0BSD\nplatforms: [linux, macos]\nmetadata:\n  hermes:\n    tags: [adversarial, code-review, multi-model, sequential, loop, persona, git]\n    related_skills: [adversarial-code-review, triangle-code-review, claude-tmux-wrapper]\n---\n\n# Adversarial Code Loop v4\n\n**BUILD → REVIEW → (FIX → VERIFY)^N → ARBITER.** A sequential pipeline where one model\nwrites code, another critiques the git diff, the first fixes, the second validates, and\nan optional arbiter resolves the last disagreement. Every loop runs on its own git\nbranch; each BUILD/FIX is a commit; reviews inspect real git diffs; the result is squash-\nmerged into the parent branch (or marked `[REJECTED]`).\n\n> **Rule: the orchestrator never writes code directly.** This skill delegates code to\n> DEV/FIXER agents (codex, claude-tmux, pi). The orchestrator writes the spec, launches\n> the pipeline, and interprets the results. Never use `patch`/`write`/`bash` to edit code\n> inside a task covered by this skill — always go through the DEV role. If no DEV agent is\n> configured explicitly, use `pi` with the current model.\n\nBased on Multi-Persona adversarial debate (Smit et al., ICML 2024): each role gets a\ndistinct persona, which improves quality even when both roles share the same model.\n\n**When to use:** code that must be **reviewed by another model** before delivery\n(breaking the echo chamber), critical code (security, auth, money), and well-scoped\nmulti-file refactors (up to ~15 files with a structured spec). Not for simple questions,\ntrivial 1-file changes, or open-ended design exploration.\n\n## Installation\n\nRequires the `adversarial-common` sibling repo (shared engine). One-line install:\n\ncurl -fsSL https://raw.githubusercontent.com/chpomob/adversarial-code-loop/main/scripts/install.sh | bash\n\nor, from an existing checkout:\n\nbash scripts/install.sh\n\nBoth place adversarial-code-loop and adversarial-common side by side under `~/.hermes/skills` (override the target with `$1` or `$HERMES_HOME`).\n\n## Overview — what's new in v4\n\nv4 is **git-native**. Where v3 wrote files directly to the worktree and reviewed a stdin\nconcatenation of file contents, v4 isolates every loop on a dedicated branch and reviews\nreal diffs.\n\n| Concern | v3 | v4 |\n|---------|----|----|\n| Isolation | none — writes to live worktree | dedicated branch `loop/<feature>/<N>` |\n| Review input | concatenated file contents (stdin) | `git diff <branch-point>..HEAD` |\n| BUILD/FIX output | prose/JSON the orchestrator extracts | files committed by the model |\n| Recovery on failure | manual file salvage | `git reset`/`git checkout` to restore |\n| Merge | manual `git add -A` | squash-merge into parent branch |\n| Rejection | exit code only | `[REJECTED]` marker commit + branch preserved |\n| Resume | not supported | `--resume` from `state.json` |\n| JSON robustness | strict `json.loads` | `strip_json_wrapper` parses markdown-fenced JSON |\n| Gates | none | optional `--build-cmd` / `--test-cmd` |\n| Result contract | `final.json` + exit code | `final.json` + exit code (unchanged, enriched) |\n\n## Workflow\n\n```\nPHASE 0 ──→ GIT SETUP   (detect/init repo, stash dirty tree, record branch-point,\n                          create loop/<feature>/<N>, bootstrap git identity, gitignore)\nPHASE 1 ──→ BUILD        (DEV writes code, orchestrator stages + commits \"build: ...\")\n                          [optional --build-cmd gate]\nPHASE 2 ──→ REVIEW       (model on git diff <branch-point>..HEAD → JSON findings)\nPHASE 3 ──→ FIX          (DEV addresses findings, orchestrator commits \"fix: ... (round N)\")\nPHASE 4 ──→ VERIFY       (model checks each finding resolved | rejected | disputed)\n   loop 3-4 until APPROVED or --max-loops reached\nPHASE 5 ──→ ARBITER      (optional; resolves disputes after max-loops)\n                          [optional --test-cmd gate]\nMERGE     ──→ squash-merge into parent + evidence tag (APPROVED / ARBITRATED)\n              or [REJECTED] marker commit, loop branch preserved (REJECT)\n```\n\nPHASE 0–5 and the merge are implemented as thin wrappers in `scripts/phases/`\n(`phase_git`, `phase_build`, `phase_review`, `phase_fix`, `phase_verify`,\n`phase_arbiter`). The shared engine — subprocess runner, JSON I/O, provider detection,\ngit operations — lives in the `adversarial-common` sibling skill.\n\n## CLI flags\n\nResolution order per command role: **CLI flag > env var > built-in default**. The\nbuilt-in defaults name specific tools/models (see table) but are overridable; set the\nenv vars or flags to point at your own DEV/REVIEW/ARBITER CLIs.\n\n| Flag | Env | Default | Description |\n|------|-----|---------|-------------|\n| `--spec` | — | *(required)* | Specification file to implement |\n| `--workdir` | — | `.` | Working directory (subprocess cwd, base of `--out`) |\n| `--dev-cmd` | `ACL_DEV_CMD` | `codex exec --skip-git-repo-check --sandbox workspace-write` | DEV (BUILDER/FIXER) command |\n| `--review-cmd` | `ACL_REVIEW_CMD` | `pi --provider zai --model glm-5.2` | REVIEW (CRITIC/VERIFIER) command |\n| `--arbiter-cmd` | `ACL_ARBITER_CMD` | — *(unset = no arbiter)* | ARBITER (JUDGE) command, optional |\n| `--max-loops` | — | `3` | Max FIX/VERIFY cycles |\n| `--no-arbiter` | — | off | Skip arbitration; REJECT instead |\n| `--timeout` | — | `600` | Per-subprocess timeout (s) |\n| `--build-cmd` | — | — | Build gate run after BUILD (e.g. `cargo build`) |\n| `--test-cmd` | — | — | Test gate run before merge (e.g. `cargo test`) |\n| `--no-merge` | — | off | On approval, leave the loop branch unmerged |\n| `--feature` | — | spec filename | Feature name used for branch + artifact dir |\n| `--out` | — | `.adversarial-loop` | Artifact output directory (under `--workdir` if relative) |\n| `--resume` | — | off | Resume from `state.json` |\n| `--provider-config` | — | `~/.config/adversarial/providers.yaml` | External provider config for quota-aware provider selection (see \"Quota-aware provider selection\" section) |\n| `--force` | — | off | Bypass all quota checks for all roles, use first configured provider regardless of state |\n| `--force-provider` | — | — | Repeatable: `--force-provider <role>:<alias>` bypasses quota for a single role (e.g. `--force-provider review:deepseek`). Other roles still check quotas normally |\n\n> **Env-var support is limited by design.** As of v4.0.0 the orchestrator honors only the\n> three command env vars above (`ACL_DEV_CMD`, `ACL_REVIEW_CMD`, `ACL_ARBITER_CMD`).\n> `ACL_WORKDIR`, `ACL_MAX_LOOPS`, `ACL_TIMEOUT`, and `ACL_OUT_DIR` are **not** read by the\n> current code — pass those values via flags. (The names are reserved so future releases\n> can wire them without breaking existing invocations.)\n\nREVIEW/VERIFY commands are passed through privilege reduction (the pipeline strips\nknown dangerous CLI flags like `--dangerously-bypass-approvals-and-sandbox` and\n`--yolo` from review commands), but the pipeline does NOT enforce OS-level\ncontainment (no kernel sandbox, no network cutoff, no filesystem jail). The\n`SandboxMode` enum in adversarial_common is advisory metadata, not a security\nboundary. Reviewers SHOULD use the least-privilege sandbox their CLI provides.\n\n## Exit codes\n\n| Code | Meaning |\n|------|---------|\n| `0` | **APPROVED** — squash-merged into the parent branch |\n| `1` | **Infrastructure failure** — phase crash, timeout, git error, interrupt |\n| `2` | **Usage error** — bad flag, missing/unreadable `--spec`, missing/bad `--workdir` |\n| `3` | **REJECT** — findings unresolved after `--max-loops`, or `--build-cmd`/`--test-cmd` gate failed, or empty BUILD diff. Loop branch is preserved. |\n| `4` | **ARBITRATED** — arbiter approved; conditions recorded in `final.json` |\n\nOrchestrators consuming the pipeline should read `final.json` (the machine-readable\ncontract), not the exit code.\n\n## Findings JSON schema\n\nREVIEW output (one model call, validated by `phase_review._validate`; retried once on\nmalformed JSON):\n\n```json\n{\n  \"findings\": [\n    {\"id\": \"A1\",\n     \"severity\": \"blocker|major|minor|nit\",\n     \"file\": \"path/to/file.rs\",\n     \"line\": 42,\n     \"summary\": \"Short title\",\n     \"evidence\": \"Why it matters, referencing real code in the diff\"}\n  ],\n  \"verdict\": \"REQUEST_CHANGES|APPROVE|REJECT\"\n}\n```\n\n`line` must be an integer (numeric strings are tolerated). Findings lacking an `id`\nreceive a deterministic `auto_<hash>` id so VERIFY can track them across rounds.\n\nVERIFY output (validates each finding's resolution against the current diff):\n\n```json\n{\n  \"results\": [\n    {\"id\": \"A1\", \"status\": \"resolved|rejected|disputed\"}\n  ],\n  \"verdict\": \"APPROVE|REJECT\"\n}\n```\n\n- `resolved` — the problematic code is gone or corrected.\n- `rejected` — the verifier disagrees with the original finding (it was wrong).\n- `disputed` — unclear; stays open for the next round or the arbiter.\n\nApproval requires `verdict == APPROVE` **and** every finding settled (`resolved` or\n`rejected`). A finding the verifier `rejected` does not block approval.\n\n## Artifacts\n\nEmitted under `<--out>/<feature>/` (auto-appended to `.gitignore` so they never merge):\n\n| File | Phase | Contents |\n|------|-------|----------|\n| `state.json` | 0 | Resumability: completed phases, current loop, branch, branch-point SHA, stash id, findings |\n| `00_spec.txt` | 1 | Spec verbatim |\n| `01_build.json` | 1 | BUILD result + commit SHA |\n| `01_build_gate.json` | 1 | `--build-cmd` gate (if set) |\n| `02_review.json` | 2 | Findings + verdict |\n| `03_fix_<N>.json` | 3 | FIX round *N* result (one per loop) |\n| `04_verdict_<N>.json` | 4 | VERIFY round *N* results + verdict (one per loop) |\n| `05_arbiter.json` | 5 | Arbiter verdict + conditions (if run) |\n| `06_test_gate.json` | 6 | `--test-cmd` gate (if set) |\n| `final.md` | end | Human-readable summary (also the evidence-tag annotation) |\n| `final.json` | end | **Machine-readable contract** — `verdict`, `reason`, `loops`, `branch`, `merged`, `conditions`, `arbitrated`, `artifacts_dir` |\n\n## Git workflow\n\n**Auto-init.** If `gitops.detect_enclosing_repo(workdir)` finds a parent repo, it is used\nas-is. Otherwise `gitops.auto_init` initializes one (initial branch pinned to `main`).\nThe parent branch is the current branch (or `main` after auto-init).\n\n**Dirty working tree.** `gitops.stash_dirty` runs `git stash push -u` at PHASE 0 and\nrecords `stash@{0}` in `state.json`. The stash is popped on **every** exit path\n(success, reject, interrupt) via `_restore`. If `git stash pop` hits a conflict (the\nparent branch advanced and touched the same lines), the loop aborts with exit 1 and a\nhuman must resolve — the stash is preserved, nothing is lost.\n\n**Branch naming.** `loop/<sanitized-feature>/<N>`, where *N* is one more than the highest\nexisting `N` under that prefix (starts at 1). `--feature` is sanitized to a\nbranch-safe slug; default is the `--spec` filename stem.\n\n**Commits.** BUILD commits `build: <feature> — <summary>`; each FIX round commits\n`fix: <feature> — address finding(s) (round N)`. An empty BUILD diff is still committed\n(empty commit allowed) but triggers an `EMPTY_DIFF` REJECT at REVIEW. Git identity\n(`user.name`/`user.email`) is bootstrapped on the loop branch if unset.\n\n**Reviews on diffs.** REVIEW and VERIFY receive `git diff <branch-point>..HEAD`, so they\nsee the cumulative change since the branch point — every BUILD + all FIX rounds — and\neach finding must reference code that actually appears in the diff.\n\n**Merge (APPROVED / ARBITRATED).** `gitops.squash_merge` checks out the parent branch,\nruns `git merge --squash <loop-branch>`, commits `squash: <feature> — adversarial\napproved`, and drops the loop branch. A merge conflict aborts with exit 1 and keeps the\nloop branch. Before merging, `tag_with_evidence` creates an annotated tag\n`<loop-branch>-approved` carrying `final.md` (best-effort — a missing file never blocks\nthe merge). `--no-merge` skips the merge and leaves the loop branch for human review.\n\n**Reject (REJECT).** `gitops.reject_marker` records an empty\n`[REJECTED] <feature> — <verdict>` commit on the loop branch. The branch is **not**\ndeleted and **not** merged, so the rejected work is recoverable.\n\n## Language discipline\n\nAll internal pipeline text is **English**: spec files, auto-generated commit messages,\npersonas (`builder.md`, `critic.md`, …), findings JSON, verdicts, synthesis reports, and\ncode comments. User-facing summaries (what the orchestrator prints to you) stay in your\nconversation language. Do not language-switch between roles inside the pipeline — it\nconfuses the model, especially in FIX where it receives an English persona + English\nreview + possibly a non-English spec.\n\n## Personas\n\nBUILDER / CRITIC / FIXER / VERIFIER / JUDGE live as text files in\n`~/.hermes/skills/adversarial-common/personas/` (single source of truth, editable without\ntouching Python). All v4 personas are **git-aware**: BUILD produces committed code,\nREVIEW inspects a diff, FIX commits a new round, VERIFY checks findings against the diff.\nInjection is provider-aware: `pi` is detected and selects `builder-pi.md`/`fixer-pi.md`\n(tool-based writes instead of markdown/JSON code output — mitigates the prose-overwrite\nfailure, pitfall #6).\n\n## Examples (validated)\n\n```bash\n# Basic — Codex DEV + GLM-5.2 REVIEW, default flags.\npython3 ~/.hermes/skills/adversarial-code-loop/scripts/adversarial_loop.py \\\n  --spec /tmp/spec.md --workdir /path/to/project\n\n# Claude-as-DEV via claude-tmux (Fable 5 / Opus). Use ABSOLUTE paths — `~` expands\n# relative to --workdir, not $HOME (pitfall #11). Extended thinking runs 8-12 min,\n# so push --timeout up and keep the inner --hard-timeout >= the loop timeout.\npython3 ~/.hermes/skills/adversarial-code-loop/scripts/adversarial_loop.py \\\n  --spec /tmp/spec.md --workdir /path/to/project \\\n  --dev-cmd \"python3 /path/to/claude-tmux.py --model best --timeout 900 --hard-timeout 2400 --max-turns 20\" \\\n  --timeout 2400\n\n# GLM-5.2 DEV + DeepSeek REVIEW (thinking high on both). No Claude quota needed.\npython3 ~/.hermes/skills/adversarial-code-loop/scripts/adversarial_loop.py \\\n  --spec /tmp/spec.md --workdir /path/to/project \\\n  --dev-cmd  \"pi -p --provider zai --model glm-5.2 --thinking high\" \\\n  --review-cmd \"pi -p --provider deepseek --model deepseek-v4-pro --thinking high\" \\\n  --max-loops 2 --no-arbiter --timeout 1200\n\n# With build + test gates and a named feature (Rust project).\npython3 ~/.hermes/skills/adversarial-code-loop/scripts/adversarial_loop.py \\\n  --spec /tmp/spec.md --workdir /path/to/project --feature peer-auth \\\n  --build-cmd \"cargo build\" --test-cmd \"cargo test\" \\\n  --max-loops 3 --timeout 1800\n\n# Arbiter on, no merge (human reviews the loop branch first).\npython3 ~/.hermes/skills/adversarial-code-loop/scripts/adversarial_loop.py \\\n  --spec /tmp/spec.md --workdir /path/to/project \\\n  --arbiter-cmd \"pi -p --provider gemini --model gemini-3-pro\" --no-merge\n\n# Resume after an interrupt (reads state.json under --out/<feature>/).\npython3 ~/.hermes/skills/adversarial-code-loop/scripts/adversarial_loop.py \\\n  --spec /tmp/spec.md --workdir /path/to/project --resume\n```\n\n> **Wrapper compatibility:** claude-tmux wrapper v1 rejects `--yolo` with argparse\n> exit code 2. Omit that flag; its permission-bypass behavior is already the default.\n\n**Model pairing notes:** Codex is a fast first-choice DEV; GLM-5.2 (pi) reviews\nthoroughly and reliably returns JSON; DeepSeek REVIEW is slower but finds more findings;\nClaude (via tmux) is the most thorough reviewer but slowest and quota-bound. Codex\nFIX often *cascades* beyond spec scope (migrates consumers, fixes adjacent bugs) — check\n`git diff --stat` after every loop before assuming REJECT means the code is wrong.\n\n**Quota-aware provider selection (available):** The pipeline now supports\n`--provider-config`, `--force`, and `--force-provider <role>:<alias>` flags.\nWhen a provider config is loaded, each phase checks real-time quotas and auto-selects\nthe best available command per role, with fallback chains defined externally. The\nprovider config lives in the user's `~/.config/adversarial/providers.yaml` by default.\nSee the spec at `adversarial-spec`'s `spec.md` for the full design.\n\n## Pitfalls\n\n1. **`--plan` mode is NOT wired into `adversarial_loop_v4.py`'s argparse.** The actual Python code has no `--plan` argument. The `adversarial_loop.py` entry point (which re-exports v4) only accepts `--spec`; `phase_plan.py` is not imported and has no CLI entry point. **Symptom:** passing `--plan` still reports that `--spec` is required. **Fix:** run each step as a separate code loop with `--spec` pointed at a focused spec. Do NOT rely on `--plan`; it is not implemented as of 2026-07-15.\n\n1b. **Codex `--sandbox read-only` vs `--dangerously-bypass-approvals-and-sandbox`.** When\n   Codex is the REVIEWER and you add `--dangerously-bypass-approvals-and-sandbox`, it\n   silently overrides `--sandbox read-only` to `--sandbox danger-full-access`, giving the\n   reviewer write access — the opposite of what you want. **Fix:** for read-only review\n   use `--sandbox read-only` **without** the bypass flag (interactive approval only);\n   for a writing DEV use `--sandbox danger-full-access --dangerously-bypass-approvals-and-sandbox`\n   together. In non-interactive mode the approval flag is required or Codex hangs.\n2. **Bound the loop** with `--max-loops`. The arbiter settles the last disagreement; it\n   does not extend the loop.\n3. **GLM JSON wrapped in markdown — parsed in v4, not v3.** v3 used strict `json.loads`\n   and choked on `` ```json ``-fenced output. v4's `jsonio.strip_json_wrapper` strips\n   fences and extracts the largest JSON object, so GLM-5.2 / Claude markdown-wrapped JSON\n   is now parsed. Since P6, every phase (including VERIFY) routes through the shared\n   3-strategy parser `adversarial_common.jsonio.parse_json_output`, which tries:\n   (1) markdown stripping, (2) extracting `{...}` via\n   `text.find('{')`..`rfind('}')`, (3) extracting `[...]` for raw arrays. This makes the\n   pipeline **model-agnostic** — the same code works regardless of whether the model\n   returns raw JSON, markdown-wrapped JSON, text + JSON, or a JSON array. REVIEW/VERIFY\n   still retry once on malformed JSON. If a model returns prose with no JSON object at\n   all, the phase fails (exit 1) — check the captured stdout in the artifact.\n4. **Claude extended thinking runs 8-12 min (Fable 5).** Pass\n   `--timeout 900 --hard-timeout 2400` *inside* the claude-tmux command and keep the\n   loop's `--timeout >= 2400`. The inner `--timeout` controls tmux pane inactivity\n   detection — if Claude goes silent for more than this period, the pane is killed.\n   With extended thinking, Claude can be silent for 12+ minutes even on small codebases\n   (validated 2026-07-13 on a ~2580-line plugin: first review timed out at 600s).\n   **Never use `--timeout 600` or lower** for Fable 5 REVIEW — the first pass always\n   has the longest thinking burst as it reads the full diff and project structure. Set\n   `--hard-timeout 2400` (40 min) to survive Verifier passes that require multiple file\n   reads. If Claude repeatedly times out, switch `--review-cmd` to GLM-5.2\n   (`pi -p --provider zai --model glm-5.2 --thinking high`), which is faster (no\n   extended thinking) and equally reliable for JSON output. See\n   `references/wrapper-failures.md`, `references/fable5-timeout-recovery.md`, and the\n   `claude-tmux-wrapper` skill.\n5. **Dirty working tree must be committed or stashed.** v4 auto-stashes at PHASE 0 and\n   restores on every exit path, so a dirty tree no longer blocks startup. The remaining\n   risk is a **stash-pop conflict**: if the parent branch advanced and touched the same\n   lines you had stashed, `git stash pop` fails and the loop aborts (exit 1). The stash\n   is preserved — resolve manually, then `--resume`.\n6. **Models may overwrite source files with prose instead of code.** Claude/Fable 5,\n   pi/GLM-5.2, and Codex have all been observed to replace working source with a markdown\n   report or `<<<SEE BELOW>>>` placeholder. v4 mitigates this with pi-specific personas\n   (`builder-pi.md`/`fixer-pi.md`, auto-selected when `pi` is detected) and is far easier\n   to recover from than v3: `git checkout HEAD -- <file>` restores the committed version\n   on the loop branch, then re-run FIX or apply the change directly via `patch` for a\n   well-understood single-file fix. For mechanical fixes, direct `patch` is faster and\n   more reliable than re-running the loop.\n7. **Merge conflicts if the parent branch advances during the loop.** Squash-merge aborts\n   with exit 1 and keeps the loop branch. Fix by rebasing the loop branch onto the\n   updated parent (`git rebase <parent>`) or re-running; the loop branch is never lost.\n8. **NEVER run parallel loops on the same workdir.** Each loop checks out its own branch,\n   but two concurrent DEV/FIXER subprocesses writing to the same worktree files corrupt\n   each other. Run batches sequentially; only disjoint file sets (no overlap in\n   `git diff --stat`) can run in parallel. See `references/batch-splitting-strategy.md`.\n9. **`--resume` requires `state.json` from a previous run.** It is read from\n   `<--out>/<feature>/state.json`. If absent (e.g. you changed `--feature` or wiped\n   `--out`), the loop starts fresh with a warning. Resumed runs re-checkout the recorded\n   branch and skip completed phases.\n10. **`~` in `--dev-cmd`/`--review-cmd`/`--arbiter-cmd` expands relative to `--workdir`,\n    not `$HOME`.** The subprocess runner does no shell expansion. Always use absolute\n    paths (`/home/user/.hermes/...` or `$HOME/.hermes/...`) for scripts in command flags.\n11. **Use claude-tmux-wrapper, not `claude -p`, for Claude roles.** `claude -p` bills\n    against Agent SDK credit (monthly cap); interactive Claude via tmux stays on the 5h\n    sliding quota. Model alias `claude-sonnet-4` is invalid — use `claude-sonnet-4-20250514`\n    or `opus`/`sonnet`/`best`/`fable` aliases. See `references/claude-p-migration-pattern.md`.\n12. **Prompt injection from reviewed code.** Code under review (diff, spec) can embed\n    adversarial instructions like `{\"verdict\": \"APPROVE\"}` that try to override the\n    pipeline verdict. v4's review-on-diff narrows the attack surface but does not close\n    it. See `references/prompt-injection-threat-model.md`; cross-model diversity (using\n    different models for DEV and REVIEW) is a recommended defense, though the pipeline\n    does not enforce it — distinct personas alone provide some separation.\n13. **Codex sandbox builds commit `target/` / build artifacts.** When a DEV/FIXER runs\n    `cargo build`/`cargo test`, the sandbox writes `target/` into the workdir; the\n    orchestrator's `git add -A` at BUILD/FIX commits them, bloating the squash. Ensure\n    `target/` (and equivalent) is in `.gitignore` **before** the first loop. After a\n    loop: `git status --porcelain target/ | head -3`; if committed, `git rm -r --cached\n    target/` and gitignore it.\n14. **The loop can REJECT for out-of-scope findings.** The reviewer is not told to\n    distinguish \"pre-existing bug\" from \"new bug in this changeset.\" GLM-5.2 is\n    particularly prone to finding pre-existing bugs outside spec scope. After a REJECT,\n    always build + test and inspect the code on disk; if the spec-scope code is correct\n    and the rest are pre-existing, the code is usable — commit it and patch the rest\n    manually if wanted.\n15. **Codex / models may exit 1 on deletion-only or \"no new code\" specs** without writing\n    to stdout. Check `git status` / `git diff --stat` on the loop branch — the model may\n    have made the changes before the process died. An empty BUILD diff is REJECTed as\n    `EMPTY_DIFF`.\n16. **`--out` persists between runs.** The directory is created with\n    `mkdir(parents=True, exist_ok=True)` and not cleaned. Re-running with a different\n    spec in the same project: either `rm -rf .adversarial-loop` first, or use a distinct\n    `--out` / `--feature`.\n17. **No parallel loops sharing a branch namespace** — the monotonic `<N>` counter in\n    `loop/<feature>/<N>` is read from existing refs at PHASE 0; two concurrent starts can\n    pick the same *N* and clobber each other. Sequential launches are safe.\n18. **Codex / OpenAI quota exhaustion kills REVIEW silently.** Codex has usage limits,\n    especially on free/Plus tiers. When exhausted (`ERROR: You've hit your usage limit`),\n    the review phase exits 1 with no useful output. **Detection:** before a long loop,\n    check quota with a quick CODE-only call (no reasoning). **Fallback:** switch\n    `--review-cmd` to a non-OpenAI provider (GLM-5.2, DeepSeek, Claude). If Codex is the\n    only reviewer configured, prepare a fallback inline or skip the review pass. Codex\n    quota resets at the start of each month (OpenAI billing cycle). See\n    `references/ai-quota-apis.md`.\n19. **User preference: never say \"I'll check back in X minutes\" without actually doing\n    it.** When monitoring a long-running loop, use an explicit polling loop\n    (`for i in 1..N; do sleep 30; ls artifacts/; done`) or rely on\n    `notify_on_complete=true`. Passive promises without follow-through frustrate\n    the user. Either monitor actively with a polling loop, or say nothing and let the\n    notification fire. See `references/monitoring-long-running-loops.md`.\n    **Validated 2026-07-14:** the user called out the agent twice in one session for saying \\\"I'll check back\\\" without doing it. The agent said \\\"je revérifie dans 3 min\\\" and the reply was \\\"tu as encore menti\\\". This is a hard constraint: either launch a real polling loop now, or use notify_on_complete and stay silent. Never end a turn with a future-monitoring promise.\n\n    **Concrete pattern that was validated:** launch with\n    `terminal(background=true, notify_on_complete=true)` and do other work. When\n    mid-run progress checks are needed, use a compact `for` loop with `sleep 30`\n    that checks for specific artifact files (`02_review.json`, `loop_1_04_verdict.json`,\n    `final.json`).\n20. **DeepSeek via pi requires `~/.pi/agent/auth.json`.** Hermes stores the DeepSeek API\n    key in `~/.hermes/.env` but does NOT export it to subprocesses. To use DeepSeek\n    through `pi`, create `~/.pi/agent/auth.json` with: `{\"deepseek\": {\"type\": \"api_key\",\n    \"key\": \"<key>\"}}`. Extract the key from Hermes via `grep DEEPSEEK_API_KEY\n    ~/.hermes/.env` (the file has the actual key — Hermes masks it in terminal output\n    but the file is readable by Python). Set permissions to `0600`. See\n    `references/pi-auth-setup.md`.\n21. **`terminal(background=true)` with `notify_on_complete=true` is the recommended\n    monitoring pattern.** Long loops (5+ minutes per phase) should run in the background.\n    The preferred approach: launch the loop with `background=true` +\n    `notify_on_complete=true`, then work on other tasks. The notification fires\n    automatically on completion. If you must monitor mid-run, use a compact polling\n    loop: `for i in 1..N; do sleep 30; ls artifacts/; done`. Avoid idle waiting —\n    do other work while the loop runs.\n22. **DeepSeek V4 Pro VERIFY JSON can be malformed.** DeepSeek with `--thinking high`\n    occasionally wraps JSON in additional markdown or text, causing\n    `strip_json_wrapper` to fail extraction. The retry also fails because the model\n    repeats the same wrapping. **Symptoms:** REVIEW succeeds (findings parsed), but\n    VERIFY fails with \"invalid JSON after retry\". **Mitigation:** switch `--review-cmd`\n    to a model that reliably outputs raw JSON (GLM-5.2 is more reliable for VERIFY).\n    Or check the code on the loop branch manually — BUILD and FIX commits are correct\n    even when VERIFY fails. Validated 2026-07-06 with GLM+DeepSeek pairing.\n23. **Review prompt no longer concatenates code — model reads files directly from\n    the loop branch checkout.** The review prompt is under 1K tokens. The reviewer\n    runs `git diff HEAD~1..HEAD` to see changes and reads files with `cat`/`grep`\n    for context. See `references/review-on-committed-code.md`.\n24. **GLM-5.2 quota is 80 prompts per rolling 5h (Z.AI Lite).** HTTP 429 after 2-3\n    heavy loops. Recovery: switch to DeepSeek V4 Pro (`pi -p --provider deepseek --model deepseek-v4-pro --thinking high`) for DEV, or Claude Sonnet for REVIEW. If all providers exhausted, wait 5h for GLM reset.\n25. **User preference — monitor actively or stay silent.** Use polling loops or\n    `notify_on_complete=true`. Never promise to \"check back\" without following through.\n26. **User preference — quality over speed.** Always use `--thinking high`. Set generous\n    timeouts (`--timeout 2400`). Accept 10-15 min BUILD times.\n27. **Pipeline workdir == Hermes Agent install directory (fork-as-live-install).** When\n    `--workdir` points at the Hermes Agent repo and Hermes is *running that checkout*, the\n    pipeline's git operations (branch creation, checkout, squash-merge) operate on the live\n    codebase. A squash-merge into the parent branch (typically `main`) without `--no-merge`\n    commits the loop output directly into your running Hermes install — which can leave the\n    install in an inconsistent state mid-change. **Always use `--no-merge`** so the loop\n    branch stays isolated for human review and manual merge. After review, merge deliberately:\n    `git checkout main && git merge --squash <loop-branch>`. Also, auto-stash of dirty trees\n    (pitfall #5) is riskier here: a stash-pop conflict during the pipeline aborts with exit 1\n    and leaves the working tree in a mixed state while Hermes is trying to run from those same\n    files. Pre-commit or stash manually before launching. See `references/fork-as-live-install.md`.\n\n28. **`.gitignore` auto-modification leaks into upstream PRs.** The pipeline's PHASE 0\n    appends `--out` patterns (`.adversarial-loop/` by default) to `.gitignore` so artifacts\n    never get tracked. This is correct for local development, but the `.gitignore` change\n    ends up in every BUILD commit (via `git add -A`) and propagates into the squash merge.\n    When the loop output is destined for an upstream PR, **drop the `.gitignore` delta before\n    pushing**. After squash-merge into the parent branch: check with\n    `git diff HEAD~1..HEAD -- .gitignore`; if it shows artifact patterns, restore the\n    upstream version with `git checkout HEAD -- .gitignore` and amend:\n    `git commit --amend --no-edit`. For `--no-merge` loops: inspect `.gitignore` before the\n    manual merge — the upstream `.gitignore` likely already has `target/` etc., so a diff\n    showing only `.adversarial-loop/`, `*.orig`, `*.rej` is the signal.\n    See `references/pre-pr-cleanup.md`.\n\n29. **Keep REVIEW/VERIFY timeout propagation wired end to end.** `run_review()` and\n    `run_verify()` accept a `timeout` parameter and pass it to `providers.run_cmd()`;\n    both call sites in `adversarial_loop.py` pass `timeout=args.timeout`. This makes the\n    pipeline's `--timeout` apply to all five phases. Preserve all three links when\n    changing phase signatures or dispatch. A regression causes Claude Fable 5 REVIEW or\n    VERIFY to fail with `exit code 124: TIMEOUT after 600s` even when the caller passed\n    `--timeout 2400`. See `references/fable5-timeout-recovery.md` for the validated\n    reproduction and implementation details.\n\n30. **claude-tmux wrapper must NOT modify the pipeline prompt.** The wrapper exists to\n    capture output via tmux instead of `claude -p`. Its only addition to the pipeline's\n    stdin is the output-capture instruction; never add behavioral modifiers that duplicate\n    or contradict the pipeline prompt. See\n    `references/claude-tmux-prompt-hygiene.md` for the validated wrapper pattern.\n\n31. **Squash commit naming for upstream PRs.** The pipeline's merge commit message format is\n    `\"squash: <feature> — adversarial approved\"` (individual BUILD commits use\n    `\"build: <feature> — <summary>\"`, FIX commits use\n    `\"fix: <feature> — address finding(s) (round N)\"`). These are pipeline-internal names\n    that don't follow conventional commits, and upstream reviewers will flag them. After\n    squash-merge into the parent branch, rewrite the squash commit:\n    `git commit --amend -m \"feat(cli): add on_status_bar_render hook to narrow width tier\"`.\n    For several squash commits stacked together, either rebase and reword each,\n    or squash them all into one conventional-format commit before pushing the branch upstream.\n    Always verify the final commit message with `git log --oneline -1` before `git push`.\n    See `references/pre-pr-cleanup.md`.\n\n32. **`pi` (GLM-5.2) can review the wrong git repo despite correct `cwd`.** Although `pi` runs inside `subprocess.Popen(cwd=workdir)`, its internal file-access tools may navigate to a different repository. **Symptom:** the REVIEW finding references a commit hash and file paths that don't exist in `--workdir` (e.g., from `hermes-agent` instead of a plugin repo), claiming an empty diff. **Diagnosis:** check `02_review.json` — if the `\"file\"` field says `\"(commit 92ce650...)\"` instead of a real file path in your project, pi is in the wrong repo. **Workaround:** merge the BUILD manually (`git merge --squash <loop-branch>`); the code on the loop branch is correct, only the review was misdirected. This was validated 2026-07-13 on a 320-line keyring-hardening step where GLM reviewed the hermes-agent repo instead of a plugin repo. After manual merge the code compiled and all 149 tests passed.\n\n33. **Fable 5 has its own usage limit separate from Claude Pro's 5h sliding quota.** The model can be blocked even when regular Claude Pro quota is green. **Symptom:** claude-tmux starts, bypasses permissions, reads the prompt, then displays \"You've reached your Fable 5 limit\" and stops. **Fix:** switch to `--model sonnet` or `--model opus`. Sonnet is preferred for plan-challenger and code-loop REVIEW because it has no extended thinking (faster response, no 12-min silence), reliable JSON output, and lower token cost. See `references/fable5-usage-limit.md`. **Validated:** 2026-07-15 — Fable 5 hit limit mid-challenge; Sonnet completed in ~2 min.\n\n34. Codex FIX phase hangs on stdin when the spec is small or findings are minor. Codex prints Reading prompt from stdin... and blocks forever when its generated input does not constitute a complete code-generation request. Symptom: BUILD succeeds, REVIEW returns findings, but FIX exits 1 with Reading prompt from stdin... as the only output. Root cause: the FIX phase embeds findings into a prompt Codex expects to be a full coding task; narrow specs with minor findings can leave Codex waiting. Mitigation (validated 2026-07-15): switch --dev-cmd from Codex to GLM-5.2 (pi -p --provider zai --model glm-5.2 --thinking high) for the problematic step. GLM reliably handles FIX prompts without stdin hang. Permanent fix: ensure the FIX prompt always includes a concrete code-generation request with file paths and expected diff pattern.\n\n35. **Untracked spec files (step-P*.md) can be lost across sequential code loops.** PHASE 0 runs `git stash push -u` which captures untracked files. If you prepare multiple step specs in the same repo and run sequential loops, each loop stashes the untracked spec files from the previous loop's stash pop. After a squash-merge, the stash is dropped — and the untracked spec files are gone. **Symptom:** a code loop with `--spec step-P3-spec.md` that ran fine earlier now fails with \"Spec not found\" because the file was stashed and dropped by a previous loop. **Fix:** add step spec patterns to `.gitignore` (e.g. `step-P*-spec.md`) so they are never stashed. Or keep step specs outside the repo's workdir and reference them by absolute path in `--spec`.\n\n36. **`hermes update` breaks after loops run on the live Hermes install (fork-as-live-install fallout).** When the pipeline workdir is the Hermes Agent git install (`~/.hermes/hermes-agent`, origin=fork, upstream=NousResearch) and loops left the repo sitting on a feature branch with local `main` carrying squash commits, `hermes update` fails. Hermes update (fork installs) switches to `main` and runs `git pull --ff-only origin main`; with `main` diverged (local adversarial squashes) and the current branch 1000s of commits behind upstream, the pull can't fast-forward and the update dies on the conflict. **Validated 2026-07-31:** install sat on `feat/status-bar-hook-all-widths` (4 commits, 4393 behind upstream/main) with `main` carrying 3 adversarial squash commits → `hermes update` failed. **Recovery (full recipe with exact commands in `references/rebase-pr-onto-upstream.md`):**\n    1. Diagnose: `git status`, `git branch -vv`, `git log --oneline upstream/main..HEAD`, `gh pr view <branch> --repo <upstream-repo> --json mergeable,mergeStateStatus` (CONFLICTING confirms the drift).\n    2. Backup: `git branch backup/<name> <parent-branch>` + `git tag backup/<feature>-pre-rebase HEAD`.\n    3. `git fetch upstream main` then `git rebase upstream/main` on the feature branch (commits replay one at a time; each may conflict).\n    4. Conflict resolution: **keep BOTH sides** — upstream's new code AND the feature's code (validated: upstream had added `_status_bar_goal_segment`/`battery_prefix`/`focus_label` in the same status-bar function the feature was extending with `_get_status_bar_plugin_values`; the correct merge keeps both methods and folds the feature's `parts`/`plugin_values` into upstream's new narrow-width branch).\n    5. Non-interactive `rebase --continue`: **`GIT_EDITOR=true git rebase --continue`** — a plain `git rebase --continue` fails with \"There was a problem with the editor\" when stdin isn't a TTY (agent terminal). The `GIT_EDITOR=true` trick applies to any non-interactive rebase/commit.\n    6. Verify: `python3 -m py_compile <touched files>`, targeted `venv/bin/python -m pytest <test files>`, then the repo's canonical runner (`scripts/run_tests.sh <test files>` — CI-parity, hermetic env).\n    7. `git push --force-with-lease origin <branch>` → PR flips to MERGEABLE (BLOCKED = review required, normal).\n    8. Realign `main` + sync fork: `git branch -f main upstream/main && git checkout main`, then `git push origin main` (fast-forward safe — check `git merge-base --is-ancestor origin/main upstream/main` first).\n    9. Confirm the fix: `hermes update --check` → \"Already up to date.\"\n    10. Leave the install on `main` (healthy for updates); the feature branch stays for PR work. Only `main` should track upstream; feature/loop work lives on branches.\n\n37. **The loop can APPROVE without satisfying textual acceptance criteria.** Validated 2026-07-31 twice on the same step (remove challenge-prompt embedding in adversarial-plan): both loops were APPROVED while `phase_challenge.py` was left untouched — the BUILD only adjusted tests/docs to match the existing code, the reviewer/verifier accepted the diff, and pytest stayed green. The spec's AC was a grep-level invariant the pipeline never checks mechanically. **Fix:** when an acceptance criterion is a grep/source invariant (e.g. \"no `plan_text` in phase_challenge.py\", \"no `adversarial_loop_v3`\"), run the grep yourself AFTER the loop reports APPROVED — do not trust the verdict. Prefer ACs that are enforced by a test the BUILD must add (sentinel-absence assertions) over prose ACs, and say so in the spec (\"a test must assert X\").\n38. **REVIEW/VERIFY timeout propagation regression (pitfall #34, re-fixed).** After the v4.1.0 fix, the `timeout=args.timeout` argument was dropped again from `run_review()` (adversarial_loop_v4.py:886) and the normal-path `run_verify()` (:1074) call sites — only the concurrent verify path kept it. Symptom: `REVIEW exited 124: TIMEOUT after 600s` even when `--timeout 2400` is passed, on large diffs (e.g. 5000+ line re-tracking diffs). Fixed in 6e7bbed by re-adding `timeout=args.timeout` to both call sites. When adding phase-parameter plumbing, grep all call sites of `phase_review.run_review` / `phase_verify.run_verify` — the concurrent path is a separate call site.\n\n39. **The DEV can write OUTSIDE the loop workdir — check sibling repos after every loop.** Validated 2026-08-01: the P16b loop (workdir = adversarial-spec) was APPROVED and squash-merged, but the DEV had implemented part of the fix (stash `on_pushed` plumbing) in the SHARED repo `adversarial-common` via a relative path (`../adversarial-common/...`), leaving 4 uncommitted files there that the spec-repo loop never committed and the spec squash didn't include. The loop's git isolation only protects the workdir's branch; any path reachable from the workdir is writable. **Fix:** after each loop that touches shared modules (adversarial-common), run `git status --porcelain` on ALL sibling repos before declaring the step done — and commit any out-of-workdir changes with a message naming the originating step.\n\n## Retrospective logging\n\nEvery pipeline failure is automatically logged to `_retrospective/ISSUES.md` with:\n\n- Phase name, branch, error message, and last 200 chars of stdout\n- Date/time of failure\n- Feature name from the spec\n\nReview issues before planning v5:\n\n```bash\ncat ~/.hermes/skills/adversarial-code-loop/_retrospective/ISSUES.md\n```\n\nTo manually add a note about a limitation you noticed, add an entry at the top of\n`_retrospective/ISSUES.md` following the same format:\n\n```markdown\n### YYYY-MM-DD — Short title\n\n- **Model combo:** GLM/DeepSeek/Claude/Codex + (role)\n- **Symptom:** What went wrong\n- **Root cause:** Why it happened\n- **Fix/workaround:** How you worked around it\n- **Would fix in v5 by:** Concrete design change\n```\n\n## Changelog\n\n- **v4.1.3** (2026-07-31): Added pitfall #37 (loop can APPROVE without satisfying textual acceptance criteria — verify grep-level invariants yourself) and pitfall #38 (reviewer timeout propagation regression, re-fixed in 6e7bbed).\n\n- **v4.1.2** (2026-07-31): Reconciled the implemented single-spec v4 workflow across 17 files: removed unsupported `--plan` workflows, obsolete v3 entrypoint/migration documentation, and invalid claude-tmux `--yolo` examples; clarified that `--plan` is not implemented; and renumbered the retained pitfalls consecutively.\n- **v4.1.1** (2026-07-31): Added pitfall #42 (now #36; `hermes update` breaks after loops on the live install — fork-as-live-install fallout) and reference `rebase-pr-onto-upstream.md` (validated recipe: backup → rebase onto upstream/main → keep-both-sides conflict resolution → `GIT_EDITOR=true git rebase --continue` → canonical `scripts/run_tests.sh` → `--force-with-lease` push → realign `main` + sync fork).\n- **v4.1.0** (2026-07-14): Added pitfalls #25b (multi-repo parent repo auto-init), #26 (claude-tmux --cwd required), #34 (REVIEW/VERIFY timeout propagation), #35 (mid-pipeline model fallback), #36 (plan resume after partial completion), and #38 (pi wrong repo review), using the numbering at release. Added references: fable5-timeout-recovery.md, github-secret-scanning-bypass.md, delegate-task-review-timeout.md. Fixed `run_review()` and `run_verify()` timeout propagation bug (timeout defaulted to 600s despite pipeline `--timeout`). Fixed `_fail_phase` → `fail_phase` bug in adversarial_review.py (underscore prefix leaked from runner module).\n\nFile v0.1.0:README.md\n\n# adversarial-code-loop\n\n**BUILD → REVIEW → (FIX → VERIFY)^N → ARBITER.** A git-native adversarial development pipeline where one model writes code, another critiques the real `git diff`, the first fixes, the second validates, and an optional arbiter resolves deadlocks.\n\nFor Hermes Agent, Claude Code, Codex, or any LLM CLI.\n\n## How it works\n\nEvery loop runs on an isolated git branch (`loop/<feature>/<N>`):\n\n```\nPHASE 0 ──→ GIT SETUP   (branch, stash, identity, gitignore)\nPHASE 1 ──→ BUILD        (DEV model writes code, commits)\nPHASE 2 ──→ REVIEW       (CRITIC model inspects `git diff <branch>..HEAD`)\nPHASE 3 ──→ FIX          (DEV addresses findings, commits)\nPHASE 4 ──→ VERIFY       (CRITIC checks each finding resolved)\n   └── loop 3-4 until APPROVED or max-loops\nPHASE 5 ──→ ARBITER      (resolves last dispute, optional)\nMERGE     ──→ squash-merge into parent, or [REJECTED] marker\n```\n\n## Comparison\n\n| Feature | adversarial-code-loop | claude-wizard | opencode-spec-kit |\n|---------|----------------------|---------------|-------------------|\n| Git-native (reviews real diffs) | ✅ | ❌ | ❌ |\n| Multi-model (Codex DEV + Claude REVIEW) | ✅ | ❌ Single model | ❌ |\n| Per-step plan mode | ❌ (manual) | ❌ | ❌ |\n| Resume on interrupt | ✅ `--resume` from `state.json` | ❌ | ❌ |\n| Build/test gates | ✅ `--build-cmd` / `--test-cmd` | ❌ | ❌ |\n\n## Quick start\n\n```bash\npython3 scripts/adversarial_loop.py \\\n  --spec /path/to/spec.md \\\n  --workdir /path/to/project \\\n  --dev-cmd \"codex exec --sandbox workspace-write\" \\\n  --review-cmd \"pi -p --provider zai --model glm-5.2 --thinking high\"\n```\n\nSee `SKILL.md` for full CLI reference and 30+ validated pitfalls.\n\n## Dependencies\n\n- Python ≥ 3.11\n- Git ≥ 2.5\n- A DEV CLI (codex, pi, claude-tmux, …)\n- A REVIEW CLI (pi, claude-tmux, …)\n\nUses `adversarial-common` as the shared engine.\n\n## License\n\n0BSD — see [LICENSE](LICENSE).\n\nFile v0.1.0:_meta.json\n\n{\n  \"ownerId\": \"kn7e26az9x7m8bgwfwg90q1wkh8bsqw0\",\n  \"slug\": \"adversarial-code-loop\",\n  \"version\": \"0.1.0\",\n  \"publishedAt\": 1785780740116\n}\n\nFile v0.1.0:references/batch-splitting-strategy.md\n\n# Batch Splitting Strategy for Adversarial Dev Loops\n\nHow to go from an adversarial-code-review report (17 findings) to fixed code via adversarial dev loops, without stepping on your own feet.\n\n## The Pipeline\n\n1. **adversarial-code-review** → finds bugs, classifies by severity, cross-validates\n2. **Group findings into batches** by file dependency (see below)\n3. **Write a spec per batch** covering the fixes, files, and test requirements\n4. **Run adversarial dev loops sequentially** — NEVER parallel on overlapping files\n5. **Commit after each batch** before starting the next\n\n## Why Sequential?\n\nThe DEV/FIXER role writes files to the workdir. Two simultaneous loops that touch the same file will overwrite each other. The review batches may touch disjoint files but the dev loop writes what the spec asks for — and specs often overlap on shared files (e.g. `routes_api.py`, `gui.py`).\n\n**Rule:** only parallelize when `git diff --stat` between batches shows zero file overlap. In practice, this almost never happens — just run sequentially.\n\n## Batch Grouping Rules\n\n1. **Tightly coupled bugs → same batch.** If fixing bug A enables bug B (e.g. callback fix enables refresh-loop fix), batch them together.\n2. **Same file modified → same batch if possible.** Two bugs in `routes_api.py` should be one batch.\n3. **Independent fixes → separate batches.** Config atomicity (routes_api.py) and version sync (__init__.py) can be separate if they don't touch the same function.\n4. **Size: 3-6 bugs per batch max.** More than 6 and the spec gets too long; the DEV may skip items.\n\n## Validated Example (pz-save-manager, 2026-06-16)\n\n17 adversarial-review findings, 10 fixed across 3 batches:\n\n| Batch | Bugs | Files | Cycles | Files touched |\n|-------|------|-------|--------|---------------|\n| 1 | A2/B1, B2, B3, C1, C2 | 5 | 1 | gui.py, routes_api.py, watcher.py, index.html, test_gui.py |\n| 2 | B5, B6, B7, B8 | 4 | 1 | routes_api.py, backup.py, index.html, __init__.py |\n| 3 | A4 | 1 | 2 | gui.py |\n\nBatch 1 and 2 both touched `routes_api.py` and `index.html` → must be sequential. Batch 3 touched `gui.py` (also touched by batch 1) → must be sequential.\n\n**Total: ~30 min for 3 batches, 4 cycles, all APPROVED, Codex DEV + Claude Opus REVIEW.**\n\n## Pitfalls\n\n- **Don't batch security hardening with functional fixes.** The review's A1 (loopback gate) is a cross-cutting concern touching every route — it deserves its own batch with careful testing.\n- **The adversarial-code-review cross-review was one-directional** (A reviewed B only). Single-reviewer findings (A1-A9) have lower confidence. Prioritize cross-validated (A2/B1) and consensus (B2-B8) findings first.\n- **Disputed findings (A7/B4) need human adjudication** before entering a dev loop. Don't automate a fix for something the reviewers disagreed on.\n\nFile v0.1.0:references/bug-fix-after-review-workflow.md\n\n# Bug Fix After Review — Workflow Pattern\n\nWhen a code review produces 5+ findings across multiple modules,\norganize fixes into small adversarial-loop-sized specs rather than\na single large spec.\n\n## The Pattern\n\n1. **Taxonomy pass**: Group findings by module/severity. Each group\n   becomes one spec (e.g. F1a = wifi_csi data races, F1b = ble_rssi\n   data races, etc.).\n\n2. **Batch by risk**:\n   - Lot 1 (HIGH/CRITICAL): data races, memory safety, init bugs\n   - Lot 2 (MEDIUM): dead code, edge cases, missing init\n   - Lot 3 (LOW): cosmetic, performance, style\n\n3. **Each spec targets 1-3 files max** and adds zero or very few tests.\n   The review already validated the tests — the fix just needs to\n   compile and pass.\n\n4. **Launch order**: most critical first, simplest first. This maximizes\n   the chance that every fix gets done even if quota runs out.\n\n## Example: OmniSense Bug Fix Session\n\n```\nLot 1 — Data races (4 specs, 4 files total)\n  F1a  wifi_csi atomics\n  F1b  ble_rssi atomics\n  F1c  subghz atomics\n  F1d  fusion volatile -> _Atomic\n\nLot 2 — Logic bugs (3 specs, 4 files)\n  F2a  config SD init in setup()\n  F2b  VHCI stale-response race\n  F2c  call fusion_set_band_snr()\n\nLot 3 — Edge cases (2 specs, 2 files)\n  F3a  millis() wraparound\n  F3b  csi_doppler min subcarriers\n```\n\n## When FIX Times Out\n\nIf the FIX phase times out (common with Codex sandbox at 300s\non integration specs):\n\n1. Kill the process\n2. `git diff --stat` — verify the BUILD wrote the core changes\n3. Read `02_review.json` — check if the findings are critical\n4. Apply critical fixes manually\n5. `make all && pio run` — if it compiles and tests pass, commit\n6. Document skipped findings as technical debt\n\nIf Codex wrote files during BUILD (sandbox mode), they're always\nrecoverable via `git diff --stat` even if FIX never ran.\n\nFile v0.1.0:references/builder-permission-stall.md\n\n# BUILDER Permission Stall — Deadlock Pattern\n\n## Symptom\n\nThe adversarial loop runs but:\n- BUILDER produces a conversational message asking for write/edit permission (not code)\n- CRITIC correctly flags \"no reviewable code provided\" (F1: blocker)\n- FIXER acknowledges the finding but cannot fix because there's no code to fix\n- VERIFIER rejects every cycle — still no code\n- ARBITER rules in favor of the reviewer: \"BUILDER never produced code despite N rounds\"\n\nNet result: ~25 LLM calls, 3 loops, 0 lines of code.\n\n## Root Cause\n\nCodex CLI, when asked to \"produce the COMPLETE modified source code for ALL\nfiles\", enters a meta-discussion loop:\n\n1. It detects that write commands would be blocked by sandbox/permissions\n2. Instead of generating code inline (in its stdout), it asks the user to\n   \"authorize Write/Edit permissions\"\n3. Across the FIX rounds, it produces hundreds of lines of procedural\n   discussion and 0 lines of code — even though inline generation was always\n   possible\n\n**Spec-size trigger:** the pattern is far more likely on large specs (the\nobserved case was a 6.5KB spec asking for a 12-file migration). A spec\nrequiring >3-4 files in one pass paralyzes the BUILDER — split into\nincremental specs.\n\n## Prevention\n\nAdd to the spec when targeting Codex as BUILDER:\n\n```\nIMPORTANT: Produce ALL code inline in your response using ```python code blocks.\nDo NOT attempt to write to disk. Do NOT ask for permissions. Just output the code.\n```\n\nAlso consider Claude (not Codex) as BUILDER for large refactoring tasks —\nClaude is more willing to generate code inline in stdout.\n\n## Detection\n\nAfter the BUILD step, check `01_code.md` size:\n\n```bash\nif [ $(wc -c < .adversarial-loop/01_code.md) -lt 500 ]; then\n    echo \"EMPTY BUILDER — aborting pipeline\"\nfi\n```\n\nIf the content is conversational (not code), the BUILDER stalled. Kill the\npipeline — it will not recover by looping.\n\n## Resolution\n\nDo NOT re-run the loop:\n\n1. Read the ARBITER's decision in `05_arbitrage.md` — its CODE_NEEDS_FIXES\n   rationale tells you what to do\n2. Implement the fix directly (the orchestrating agent takes over where the\n   pipeline failed). There is no code to extract because none was generated.\n\n## Fallback strategies\n\n- **Option A**: re-run BUILDER with Claude instead of Codex (swap roles)\n- **Option B**: have the orchestrating agent write the code directly (fastest)\n- **Option C**: split the spec into smaller pieces and retry\n\n## Lesson\n\nThe adversarial loop is NOT suitable for large multi-file refactors where the\nBUILDER can't write to disk. Use it for: single-file code generation, bug\nfixes one file at a time, review of existing code. NOT for 12-file\ncross-module refactoring.\n\nFile v0.1.0:references/cascade-fix-pattern.md\n\n# Cascade Fix Pattern (Codex FIX behavior)\n\n## What it is\n\nWhen Codex executes a FIX phase in the adversarial loop, it often **cascades** beyond the spec scope — migrating consumers, adding minor features, fixing pre-existing bugs it discovers during analysis.\n\n## Observed examples\n\n| Project | Spec scope | Actual FIX scope |\n|---------|-----------|------------------|\n| chatter Rust P2-T2 | Peer.authenticated field (1 change, 1 file) | Completed P2-T3 (auth gate), P2-T4 (deadlock fix), P3-T1 (typed responses), P2-T5 (tests) — 5 files, ~600 lines |\n| omnisense firmware Step 3 | 6-line fix in main.cpp | 13 files, 524 insertions across the entire codebase |\n| pz-save-manager DI refactor | 3 files | 15 files, 7 new, 512 insertions, 1065 deletions |\n\n## Why it happens\n\nCodex's FIX prompt receives the full review with findings. It explores the codebase and finds opportunities to improve code quality. This is usually **productive** (accelerates multi-step plans by 2-3x) but can trigger REJECT when GLM-5.2 or Claude find out-of-scope issues.\n\n## Implications\n\n- After an APPROVED cascade, **always check `git diff --stat`** to see what actually changed\n- A cascade may commit `target/` build artifacts (pitfall #25) — verify `.gitignore`\n- If the next step in your plan targets files that Codex already touched, **verify current state before writing the next spec** — the file may already be different from what the plan assumes\n- REJECT after a cascade is often **out-of-scope** (pitfall #24): check if the spec-scope code is correct, commit anyway\n\n## Response\n\n```bash\n# After cascade APPROVED:\ngit diff --stat                    # see actual scope\ngit log --oneline -1               # review commit message\ngrep -n \"target/\" .gitignore       # verify gitignore\ngit status --porcelain target/ 2>/dev/null | head -3  # check no target/ leaks\n```\n\nFile v0.1.0:references/claude-p-migration-pattern.md\n\n# Migration: `claude -p` → `claude-tmux.py` Pipe Pattern\n\n**Context:** `claude -p` (headless print mode) draws from Agent SDK credit ($20/mo Pro) since June 2026. Interactive `claude` via tmux wrapper stays on the 5h sliding quota. All skills must use claude-tmux.py instead.\n\n## Simple pipe replacement\n\n### Before (BANNED)\n```bash\necho \"context\" | claude -p \"instruction\" --model opus --max-turns 10 --output-format json > output.json\n```\n\n### After\n```bash\necho \"context\" | python3 /home/chpo/.hermes/skills/autonomous-ai-agents/hermes-agent/scripts/claude-tmux.py --prompt \"instruction\" --model opus --timeout 600 --hard-timeout 1200 --max-turns 20 > output.json 2>output.err\n```\n\nKey changes:\n- `claude -p` → `python3 <abs-path>/claude-tmux.py`\n- Do not add `--yolo`; claude-tmux wrapper v1 rejects it (permission bypass is already the default)\n- Add `--timeout 600 --hard-timeout 1200` (inactivity + absolute caps)\n- Add `--prompt \"instruction\"` to prepend the instruction before stdin context\n- Drop `--output-format json` (not available in interactive mode; prompt the model to output JSON instead)\n- 2>stderr redirect recommended for debugging (model switches, timeout warnings)\n\n## Multi-line instruction (bash pipe)\n\nWhen the instruction is a multi-line string that was previously embedded in `claude -p \"...\"`:\n\n### Before (BANNED)\n```bash\npython3 -c \"...\" | claude -p \"Produis un RAPPORT en markdown:\n# Title\n## Section A\n## Section B\nInclus les insights.\" --model opus --max-turns 10 > output.md\n```\n\n### After\n```bash\npython3 -c \"...\" | python3 /home/chpo/.hermes/skills/autonomous-ai-agents/hermes-agent/scripts/claude-tmux.py --prompt \"Produis un RAPPORT en markdown: # Title ## Section A ## Section B Inclus les insights.\" --model opus --timeout 600 --hard-timeout 1200 --max-turns 20 > output.md\n```\n\nThe prompt string sits on one line (no literal `\\n` needed — markdown headers render fine with spaces between them via `--prompt`).\n\n## Multi-line context + instruction (echo pipe)\n\nWhen both context and instruction are in an echo string piped to claude:\n\n### Before (BANNED)\n```bash\necho \"$SAFETY\n<UNTRUSTED_CTX>\n$CTX\n</UNTRUSTED_CTX>\nDo the task.\" | claude -p \"$(cat)\" --model opus --max-turns 15 --output-format json > output.json\n```\n\n### After\n```bash\necho \"$SAFETY\n<UNTRUSTED_CTX>\n$CTX\n</UNTRUSTED_CTX>\nDo the task.\" | python3 /home/chpo/.hermes/skills/autonomous-ai-agents/hermes-agent/scripts/claude-tmux.py --model opus --timeout 600 --hard-timeout 1200 --max-turns 20 > output.json 2>output.err\n```\n\nHere the full prompt (both instruction and context) is piped via stdin. No `--prompt` needed.\n\n## Claude with `--allowedTools` and `--workdir`\n\n### Before (BANNED)\n```bash\necho \"...\" | claude -p \"Write code...\" --model opus --max-turns 15 --allowedTools \"Read,Write,Edit,Bash\" --workdir \"$REPO\" > output.json\n```\n\n### After\n```bash\necho \"...\" | python3 /home/chpo/.hermes/skills/autonomous-ai-agents/hermes-agent/scripts/claude-tmux.py --model opus --timeout 600 --hard-timeout 1200 --max-turns 20 --allowedTools \"Read,Write,Edit,Bash\" --workdir \"$REPO\" > output.json 2>output.err\n```\n\n`--allowedTools` and `--workdir` are passed through to `claude` by the wrapper — they work identically in interactive mode.\n\n## Verifying a migration\n\n1. Run the command and check exit code (0 = success)\n2. Check that output file is non-empty and contains the expected content\n3. Check stderr for `MODEL_SWITCHED` warnings (model fallback)\n4. If output is prose instead of JSON, the prompt needs to explicitly demand JSON format\n\n## References\n\n- claude-tmux-wrapper skill for wrapper details\n- adversarial-code-loop pitfall #11 for background on why claude -p is banned\n- Skills patched in this migration (2026-06-16): triangle-code-analyze, triangle-code-architect, triangle-code-develop, triangle-code-review, triangle-agents, hermes-agent-skill-authoring, claude-code, ponytail-audit-workflow\n\nFile v0.1.0:references/claude-tmux-adversarial-review.md\n\n# claude-tmux adversarial review findings (2026-07-14)\n\nThese findings were produced by an adversarial review (Codex Architect + GLM-5.2 Inspector)\nof `claude-tmux.py` on 2026-07-14. All have been fixed.\n\n## Fixed findings\n\n### 1. Hard timeout returned 0 instead of 3 (blocker)\n**Problem:** When `--hard-timeout` expired, `hard_timeout_hit=True` was set but never consumed.\nExecution fell through to the success path and returned 0 with partial output — indistinguishable\nfrom a clean completion.\n\n**Fix:** Added `if hard_timeout_hit: print(\"hard-timeout\", file=sys.stderr); return 3` before\nthe success return path. The soft-timeout path (--timeout) already returned 3.\n\n### 2. Hardcoded username `chpo@` in pane-heuristic finish detection (blocker)\n**Problem:** `wait_for_pane_text(session, \"chpo@\", ...)` would fail for any user other than `chpo`.\n\n**Fix:** Removed `\"chpo@\"` from the pane fallback pattern. The `\\u276f` (❯) prompt indicator alone\nis sufficient. Additionally added `re.search(r\"[#$%>]\\s*\\Z\", prompt_line)` as a generic shell-prompt\ndetector for covering other prompt styles.\n\n### 3. Output file read before done.sentinel confirmation (major)\n**Problem:** `output.txt` was returned as success whenever nonempty, even before `done.sentinel`\nexisted. This meant partial output from a still-running Claude could count as complete.\n\n**Fix:** The wait loop now checks `done_sentinel` BEFORE the `hard_deadline` check:\n```python\nif os.path.exists(done_sentinel):\n    break\nif hard_deadline ...:\n    hard_timeout_hit = True\n    break\n```\nThe pane heuristic was also demoted to a last-resort fallback (`pane_fallback_ready` flag)\nrather than breaking the loop immediately.\n\n### 4. Trust-dialog detection matched bare word `\"trust\"` (major)\n**Problem:** `wait_for_pane_text(session, \"trust\", timeout=10)` would fire on any pane content\ncontaining \"trust\" — not just the Claude startup dialog.\n\n**Fix:** Replaced with `\"trust the files\"` (the actual Claude dialog text). Also replaced the\n\"bypass permissions\" match with a more specific substring.\n\n### 5. Dead code: duplicate `read_nonempty_file(output_file)` (minor)\n**Problem:** Two consecutive identical reads of `output_file` — the second was unreachable.\n\n**Fix:** Removed the second block.\n\n## Design principle (critical)\n\nThe claude-tmux wrapper must **never** modify the pipeline prompt. Its only addition to stdin\nis the output-capture instruction. Specifically:\n\n**DO NOT** prepend or append behavioral modifiers:\n- \"Do NOT run shell commands\"\n- \"Output ONLY JSON\"  \n- \"You MUST respond with raw JSON\"\n\nThe pipeline already sends those instructions. Adding them causes Claude to produce prose\ninstead of JSON or run commands the pipeline didn't request.\n\n**The correct pattern:**\n```python\nprompt += (\n    f\"\\n\\nWhen you are done, write your response to {output_file} \"\n    f\"using the Write tool. \"\n    f\"After the file is written, create an empty file at {done_sentinel} \"\n    f\"using the Write tool to signal completion.\"\n)\n```\n\n## Model aliases (claude-tmux)\n\n| Alias | Maps to | Notes |\n|-------|---------|-------|\n| `opus` | Claude Opus | Stable, no extended thinking |\n| `sonnet` | Claude Sonnet 4 | Good balance. No extended thinking. |\n| `fable` | Claude Fable 5 | Extended thinking 8-12 min. Separate usage cap from 5h quota. |\n| `best` | Claude Fable 5 | Same as `fable`. |\n| `haiku` | Claude Haiku | Fastest. |\n| `claude-*` | Any full model ID | Must match `claude-[A-Za-z0-9._-]+` pattern. |\n\n## Timeout guidance for adversarial pipeline\n\n| Model | --timeout (silence) | --hard-timeout | --cwd needed? |\n|-------|--------------------|----------------|---------------|\n| Fable 5 | 900-1800s (15-30 min) | 2400-3000s (40-50 min) | Yes |\n| Sonnet | 600s (10 min) | 1200-1800s (20-30 min) | Yes |\n| Opus | 600s (10 min) | 1200s (20 min) | Yes |\n\nFile v0.1.0:references/claude-tmux-fixes-2026-07-14.md\n\n# claude-tmux.py fixes (2026-07-14)\n\nThe claude-tmux wrapper (`/home/chpo/claude-tmux-wrapper/claude-tmux.py`) was fixed to behave exactly like `claude -p` from the outside — same stdin/stdout/stderr contract, same exit codes — but running Claude via tmux (5h sliding quota) instead of Agent SDK billing.\n\n## Bugs fixed by adversarial code loop (Codex DEV + GLM REVIEW)\n\n| Bug | Fix | Validated |\n|-----|-----|-----------|\n| `--hard-timeout` expired → `return 0` (partial output treated as success) | Now `return 3` (EXIT_TIMEOUT) after teardown | APPROVED 2026-07-14 |\n| `chpo@` hardcoded in pane-heuristic finish detection (only worked for one user) | Replaced with generic prompt-regex `[#$%>]\\s*\\Z` | APPROVED 2026-07-14 |\n| `output.txt` read before `done.sentinel` (partial output could end the run) | `done_sentinel` check placed BEFORE `hard_deadline` check in the wait loop | APPROVED 2026-07-14 |\n| Trust-dialog matched bare word `\"trust\"` (any \"trust\" in pane → unsolicited Enter) | Changed to `\"trust the files\"` (actual dialog text) | APPROVED 2026-07-14 |\n| Pane heuristic killed Claude prematurely on prompt appearance | Changed to deferred `pane_fallback_ready` flag — only used after normal wait expires | APPROVED 2026-07-14 |\n| Dead code: double `read_nonempty_file(output_file)` | Removed second (unreachable) block | APPROVED 2026-07-14 |\n\n## Prompt hygiene (critical)\n\nThe wrapper must NEVER modify the pipeline's prompt with behavioral instructions. Its only addition is:\n\n```python\nprompt += (\n    f\"\\n\\nWhen you are done, write your response to {output_file} \"\n    f\"using the Write tool. \"\n    f\"After the file is written, create an empty file at {done_sentinel} \"\n    f\"using the Write tool to signal completion.\"\n)\n```\n\nNo prepended \"Do NOT run shell commands\" or \"Output ONLY JSON\" — the pipeline already sends those. Adding them causes Claude to produce prose instead of JSON.\n\n## Key design decisions\n\n1. **done.sentinel is the authoritative completion signal.** The wait loop checks for it first, then hard_deadline, then pane heuristic. This prevents partial output from being returned as success.\n2. **Hard timeout returns 3 (EXIT_TIMEOUT).** The pipeline timeout branch (else clause) also returns 3. Automation consumers can distinguish clean completion (0) from truncation (3).\n3. **Pane heuristic is a last-resort fallback.** Only used after the normal wait loop expires AND `pane_fallback_ready` was set by seeing a shell prompt after a Claude response bullet (●).\n4. **Temporary directory uses PID-based naming** (not `mkdtemp`). Known limitation: predictable path, race condition possible. Not yet fixed.\n\nFile v0.1.0:references/claude-tmux-plan-challenge-compat.md\n\n# claude-tmux + adversarial-plan challenge phase compatibility\n\n## Problem\n\nThe adversarial-plan CHALLENGE phase pipes the plan.md to a reviewer command via stdin.\nWhen using claude-tmux, Claude receives the plan + \"Write to file...\" instruction.\nClaude then runs shell commands to explore the repo (as instructed by the persona)\nbut never reaches the JSON output writing stage. The `done.sentinel` file is never\ncreated, and the wrapper times out or falls back to pane scraping (which captures\nincomplete output → \"invalid JSON after retry\").\n\n## Root cause\n\nThe CHALLENGE phase's persona (plan-challenger.md) tells Claude to:\n1. Read the plan from stdin\n2. Explore the repository files on disk  \n3. Write JSON findings\n\nClaude runs step 2 (exploration) for too long — it reads files, runs `git diff`,\nand burns through context/time budget. By the time it tries to create the JSON\noutput file + done.sentinel, the hard-timeout fires or the Write tool fails\nbecause the session is being cleaned up.\n\nThis does NOT affect the adversarial-code-loop REVIEW phase, where the persona\nis different (critic.md), the input is a git diff, and Claude's exploration is\nmore bounded.\n\n## Validated on\n\n- 2026-07-14: adversarial-features plan, CHALLENGE phase with claude-tmux via\n  claude-tmux-wrapper v2 (done.sentinel mechanism). Same failure across 3 attempts.\n  Switching to GLM-5.2 (`pi -p --provider zai --model glm-5.2 --thinking high`)\n  completed the challenge in ~30s with valid JSON.\n\n## Workaround\n\nUse GLM-5.2 (via `pi`) for the CHALLENGE phase of adversarial-plan and the\nCHALLENGE/VERIFY phases of adversarial-spec. Keep claude-tmux for the \nadversarial-code-loop REVIEW phase, where it works reliably.\n\n## Permanent fix ideas\n\n1. Pre-compute the exploration context (file listing, git log) and include it in\n   the prompt so Claude doesn't need to run shell commands.\n2. Add `--max-turns 3` to claude-tmux to limit exploration before writing.\n3. Modify the plan-challenger persona to instruct: \"Write your JSON output file\n   BEFORE exploring the repo — exploration is optional, output is mandatory.\"\n\nFile v0.1.0:references/claude-tmux-prompt-hygiene.md\n\n# claude-tmux: Prompt Hygiene for Adversarial Pipeline Use\n\n**Validated 2026-07-14.** The claude-tmux wrapper exists to run Claude via tmux (stay on\n5h sliding quota) instead of `claude -p` (Agent SDK monthly cap). It must work as a\ntransparent stdin→stdout pipe — the adversarial pipeline sends the *exact* prompt it\nwants the model to receive, and the wrapper must pass it through unchanged.\n\n## Golden rule: the wrapper captures output, not behavior\n\nThe wrapper's job is to:\n1. Read stdin (contains the full prompt from the pipeline)\n2. Start a tmux session running `claude` with that prompt\n3. Wait for Claude to finish\n4. Collect Claude's output and write it to stdout\n\nThe wrapper must **NOT**:\n- ❌ Prepend instructions (\"Do NOT run shell commands\", \"You MUST output JSON\")\n- ❌ Append behavioral modifiers (\"Write the exact output that the task demands\")\n- ❌ Rephrase or reformat the pipeline's prompt in any way\n\nThe only append is the *output capture instruction*:\n```\n\"When you are done, write your response to {output_file}\nusing the Write tool. After the file is written, create an\nempty file at {done_sentinel} using the Write tool to signal\ncompletion.\"\n```\n\nThis is passive — it tells Claude *how* to deliver the result, not *what* to produce.\n\n## What goes wrong when you modify the prompt\n\n**Symptom:** CHALLENGE phase fails with `\"invalid JSON after retry\"`.\n**Cause:** The wrapper prepended `\"Do NOT run shell commands\"` or `\"You MUST respond\nwith ONLY JSON\"` after the pipeline already said `\"Output ONLY valid JSON\"`. Claude\nreceives two inconsistent instructions and either runs shell commands anyway (confused)\nor produces prose instead of JSON.\n\n**Symptom:** REVIEW phase produces empty findings even when BUILD committed code.\n**Cause:** The wrapper appended `\"Write the exact output that the task above demands\"`\nwhich overwrites the pipeline's instruction to review the git diff. Claude writes a\nsummary of the task instead of reviewing the code.\n\n## Implementation (v2 output mechanism)\n\nThe working version (342 lines, /home/chpo/claude-tmux-wrapper/claude-tmux.py) uses:\n\n```\ntmpdir = /tmp/claude-tmux-<pid>/\n├── output.txt       # Claude writes response here\n└── done.sentinel    # Claude creates this when done\n```\n\n```python\nprompt += (\n    f\"\\n\\nWhen you are done, write your response to {output_file} \"\n    f\"using the Write tool. \"\n    f\"After the file is written, create an empty file at {done_sentinel} \"\n    f\"using the Write tool to signal completion.\"\n)\n```\n\nNo other modification to `prompt`. The wait loop checks for `done_sentinel` existence,\nthen reads `output.txt`. Fallback to pane scraping only on timeout.\n\n## Restoration if deleted\n\nThe working copy lives at `/home/chpo/claude-tmux-wrapper/claude-tmux.py` (outside the\nskills directory, safe from git stash/checkout operations). If the skills directory\nauto-init or a stash conflict deletes it, restore with:\n\n```bash\ncp /home/chpo/claude-tmux-wrapper/claude-tmux.py \\\n  ~/.hermes/skills/autonomous-ai-agents/hermes-agent/scripts/claude-tmux.py\n```\n\nThe `autonomous-ai-agents` directory is NOT tracked by any skill repo and can be\nrecreated on demand.\n\n## When claude-tmux cannot be used\n\nThe plan CHALLENGE phase embeds the full plan text + full spec text in the prompt.\nFor specs with 12+ requirements and 27+ criteria (~240 lines), the prompt is too\nlarge for Claude Fable 5's extended thinking — the silence period exceeds 20 min\neven with `--timeout 1200`. Fall back to GLM-5.2 for the plan challenge:\n\n```bash\n--review-cmd \"pi -p --provider zai --model glm-5.2 --thinking high\"\n```\n\nClaude remains the best reviewer for CODE LOOP REVIEW phases (git diff, files on disk),\nwhere the prompt is under 1K tokens and extended thinking is 8-12 min.","readmeExcerpt":"Skill: adversarial-code-loop Owner: chpomob Summary: BUILD → REVIEW → (FIX → VERIFY)^N → ARBITER on isolated git branches. Git-native: each loop runs on its own branch, changes are committed, reviews inspect git diffs. Tags: latest:0.1.0 Version history: v0.1.0 | 2026-08-03T18:12:20.116Z | auto Adversarial Code Loop v4 is a major update making the workflow fully git-native. - Each code-review loop runs on its own iso","codeSnippets":[],"executableExamples":[{"language":"text","snippet":"PHASE 0 ──→ GIT SETUP   (detect/init repo, stash dirty tree, record branch-point,\n                          create loop/<feature>/<N>, bootstrap git identity, gitignore)\nPHASE 1 ──→ BUILD        (DEV writes code, orchestrator stages + commits \"build: ...\")\n                          [optional --build-cmd gate]\nPHASE 2 ──→ REVIEW       (model on git diff <branch-point>..HEAD → JSON findings)\nPHASE 3 ──→ FIX          (DEV addresses findings, orchestrator commits \"fix: ... (round N)\")\nPHASE 4 ──→ VERIFY       (model checks each finding resolved | rejected | disputed)\n   loop 3-4 until APPROVED or --max-loops reached\nPHASE 5 ──→ ARBITER      (optional; resolves disputes after max-loops)\n                          [optional --test-cmd gate]\nMERGE     ──→ squash-merge into parent + evidence tag (APPROVED / ARBITRATED)\n              or [REJECTED] marker commit, loop branch preserved (REJECT)"},{"language":"json","snippet":"{\n  \"findings\": [\n    {\"id\": \"A1\",\n     \"severity\": \"blocker|major|minor|nit\",\n     \"file\": \"path/to/file.rs\",\n     \"line\": 42,\n     \"summary\": \"Short title\",\n     \"evidence\": \"Why it matters, referencing real code in the diff\"}\n  ],\n  \"verdict\": \"REQUEST_CHANGES|APPROVE|REJECT\"\n}"},{"language":"json","snippet":"{\n  \"results\": [\n    {\"id\": \"A1\", \"status\": \"resolved|rejected|disputed\"}\n  ],\n  \"verdict\": \"APPROVE|REJECT\"\n}"},{"language":"bash","snippet":"# Basic — Codex DEV + GLM-5.2 REVIEW, default flags.\npython3 ~/.hermes/skills/adversarial-code-loop/scripts/adversarial_loop.py \\\n  --spec /tmp/spec.md --workdir /path/to/project\n\n# Claude-as-DEV via claude-tmux (Fable 5 / Opus). Use ABSOLUTE paths — `~` expands\n# relative to --workdir, not $HOME (pitfall #11). Extended thinking runs 8-12 min,\n# so push --timeout up and keep the inner --hard-timeout >= the loop timeout.\npython3 ~/.hermes/skills/adversarial-code-loop/scripts/adversarial_loop.py \\\n  --spec /tmp/spec.md --workdir /path/to/project \\\n  --dev-cmd \"python3 /path/to/claude-tmux.py --model best --timeout 900 --hard-timeout 2400 --max-turns 20\" \\\n  --timeout 2400\n\n# GLM-5.2 DEV + DeepSeek REVIEW (thinking high on both). No Claude quota needed.\npython3 ~/.hermes/skills/adversarial-code-loop/scripts/adversarial_loop.py \\\n  --spec /tmp/spec.md --workdir /path/to/project \\\n  --dev-cmd  \"pi -p --provider zai --model glm-5.2 --thinking high\" \\\n  --review-cmd \"pi -p --provider deepseek --model deepseek-v4-pro --thinking high\" \\\n  --max-loops 2 --no-arbiter --timeout 1200\n\n# With build + test gates and a named feature (Rust project).\npython3 ~/.hermes/skills/adversarial-code-loop/scripts/adversarial_loop.py \\\n  --spec /tmp/spec.md --workdir /path/to/project --feature peer-auth \\\n  --build-cmd \"cargo build\" --test-cmd \"cargo test\" \\\n  --max-loops 3 --timeout 1800\n\n# Arbiter on, no merge (human reviews the loop branch first).\npython3 ~/.hermes/skills/adversarial-code-loop/scripts/adversarial_loop.py \\\n  --spec /tmp/spec.md --workdir /path/to/project \\\n  --arbiter-cmd \"pi -p --provider gemini --model gemini-3-pro\" --no-merge\n\n# Resume after an interrupt (reads state.json under --out/<feature>/).\npython3 ~/.hermes/skills/adversarial-code-loop/scripts/adversarial_loop.py \\\n  --spec /tmp/spec.md --workdir /path/to/project --resume"},{"language":"bash","snippet":"cat ~/.hermes/skills/adversarial-code-loop/_retrospective/ISSUES.md"},{"language":"markdown","snippet":"### YYYY-MM-DD — Short title\n\n- **Model combo:** GLM/DeepSeek/Claude/Codex + (role)\n- **Symptom:** What went wrong\n- **Root cause:** Why it happened\n- **Fix/workaround:** How you worked around it\n- **Would fix in v5 by:** Concrete design change"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: adversarial-code-loop\ndescription: \"BUILD → REVIEW → (FIX → VERIFY)^N → ARBITER on isolated git branches. Git-native: each loop runs on its own branch, changes are committed, reviews inspect git diffs.\"\nversion: 4.2.0\nauthor: Hermes Agent\nlicense: 0BSD\nplatforms: [linux, macos]\nmetadata:\n  hermes:\n    tags: [adversarial, code-review, multi-model, sequential, loop, persona, git]\n    related_skills: [adversarial-code-review, triangle-code-review, claude-tmux-wrapper]\n---\n\n# Adversarial Code Loop v4\n\n**BUILD → REVIEW → (FIX → VERIFY)^N → ARBITER.** A sequential pipeline where one model\nwrites code, another critiques the git diff, the first fixes, the second validates, and\nan optional arbiter resolves the last disagreement. Every loop runs on its own git\nbranch; each BUILD/FIX is a commit; reviews inspect real git diffs; the result is squash-\nmerged into the parent branch (or marked `[REJECTED]`).\n\n> **Rule: the orchestrator never writes code directly.** This skill delegates code to\n> DEV/FIXER agents (codex, claude-tmux, pi). The orchestrator writes the spec, launches\n> the pipeline, and interprets the results. Never use `patch`/`write`/`bash` to edit code\n> inside a task covered by this skill — always go through the DEV role. If no DEV agent is\n> configured explicitly, use `pi` with the current model.\n\nBased on Multi-Persona adversarial debate (Smit et al., ICML 2024): each role gets a\ndistinct persona, which improves quality even when both roles share the same model.\n\n**When to use:** code that must be **reviewed by another model** before delivery\n(breaking the echo chamber), critical code (security, auth, money), and well-scoped\nmulti-file refactors (up to ~15 files with a structured spec). Not for simple questions,\ntrivial 1-file changes, or open-ended design exploration.\n\n## Installation\n\nRequires the `adversarial-common` sibling repo (shared engine). One-line install:\n\ncurl -fsSL https://raw.githubusercontent.com/chpomob/adversarial-code-loop/main/scripts/install.sh | bash\n\nor, from an existing checkout:\n\nbash scripts/install.sh\n\nBoth place adversarial-code-loop and adversarial-common side by side under `~/.hermes/skills` (override the target with `$1` or `$HERMES_HOME`).\n\n## Overview — what's new in v4\n\nv4 is **git-native**. Where v3 wrote files directly to the worktree and reviewed a stdin\nconcatenation of file contents, v4 isolates every loop on a dedicated branch and reviews\nreal diffs.\n\n| Concern | v3 | v4 |\n|---------|----|----|\n| Isolation | none — writes to live worktree | dedicated branch `loop/<feature>/<N>` |\n| Review input | concatenated file contents (stdin) | `git diff <branch-point>..HEAD` |\n| BUILD/FIX output | prose/JSON the orchestrator extracts | files committed by the model |\n| Recovery on failure | manual file salvage | `git reset`/`git checkout` to restore |\n| Merge | manual `git add -A` | squash-merge into parent branch |\n| Rejection | exit code only | `[REJECTED]` marker commit + branch preserved |\n| Resume | "},{"path":"README.md","content":"# adversarial-code-loop\n\n**BUILD → REVIEW → (FIX → VERIFY)^N → ARBITER.** A git-native adversarial development pipeline where one model writes code, another critiques the real `git diff`, the first fixes, the second validates, and an optional arbiter resolves deadlocks.\n\nFor Hermes Agent, Claude Code, Codex, or any LLM CLI.\n\n## How it works\n\nEvery loop runs on an isolated git branch (`loop/<feature>/<N>`):\n\n```\nPHASE 0 ──→ GIT SETUP   (branch, stash, identity, gitignore)\nPHASE 1 ──→ BUILD        (DEV model writes code, commits)\nPHASE 2 ──→ REVIEW       (CRITIC model inspects `git diff <branch>..HEAD`)\nPHASE 3 ──→ FIX          (DEV addresses findings, commits)\nPHASE 4 ──→ VERIFY       (CRITIC checks each finding resolved)\n   └── loop 3-4 until APPROVED or max-loops\nPHASE 5 ──→ ARBITER      (resolves last dispute, optional)\nMERGE     ──→ squash-merge into parent, or [REJECTED] marker\n```\n\n## Comparison\n\n| Feature | adversarial-code-loop | claude-wizard | opencode-spec-kit |\n|---------|----------------------|---------------|-------------------|\n| Git-native (reviews real diffs) | ✅ | ❌ | ❌ |\n| Multi-model (Codex DEV + Claude REVIEW) | ✅ | ❌ Single model | ❌ |\n| Per-step plan mode | ❌ (manual) | ❌ | ❌ |\n| Resume on interrupt | ✅ `--resume` from `state.json` | ❌ | ❌ |\n| Build/test gates | ✅ `--build-cmd` / `--test-cmd` | ❌ | ❌ |\n\n## Quick start\n\n```bash\npython3 scripts/adversarial_loop.py \\\n  --spec /path/to/spec.md \\\n  --workdir /path/to/project \\\n  --dev-cmd \"codex exec --sandbox workspace-write\" \\\n  --review-cmd \"pi -p --provider zai --model glm-5.2 --thinking high\"\n```\n\nSee `SKILL.md` for full CLI reference and 30+ validated pitfalls.\n\n## Dependencies\n\n- Python ≥ 3.11\n- Git ≥ 2.5\n- A DEV CLI (codex, pi, claude-tmux, …)\n- A REVIEW CLI (pi, claude-tmux, …)\n\nUses `adversarial-common` as the shared engine.\n\n## License\n\n0BSD — see [LICENSE](LICENSE)."},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7e26az9x7m8bgwfwg90q1wkh8bsqw0\",\n  \"slug\": \"adversarial-code-loop\",\n  \"version\": \"0.1.0\",\n  \"publishedAt\": 1785780740116\n}"},{"path":"references/batch-splitting-strategy.md","content":"# Batch Splitting Strategy for Adversarial Dev Loops\n\nHow to go from an adversarial-code-review report (17 findings) to fixed code via adversarial dev loops, without stepping on your own feet.\n\n## The Pipeline\n\n1. **adversarial-code-review** → finds bugs, classifies by severity, cross-validates\n2. **Group findings into batches** by file dependency (see below)\n3. **Write a spec per batch** covering the fixes, files, and test requirements\n4. **Run adversarial dev loops sequentially** — NEVER parallel on overlapping files\n5. **Commit after each batch** before starting the next\n\n## Why Sequential?\n\nThe DEV/FIXER role writes files to the workdir. Two simultaneous loops that touch the same file will overwrite each other. The review batches may touch disjoint files but the dev loop writes what the spec asks for — and specs often overlap on shared files (e.g. `routes_api.py`, `gui.py`).\n\n**Rule:** only parallelize when `git diff --stat` between batches shows zero file overlap. In practice, this almost never happens — just run sequentially.\n\n## Batch Grouping Rules\n\n1. **Tightly coupled bugs → same batch.** If fixing bug A enables bug B (e.g. callback fix enables refresh-loop fix), batch them together.\n2. **Same file modified → same batch if possible.** Two bugs in `routes_api.py` should be one batch.\n3. **Independent fixes → separate batches.** Config atomicity (routes_api.py) and version sync (__init__.py) can be separate if they don't touch the same function.\n4. **Size: 3-6 bugs per batch max.** More than 6 and the spec gets too long; the DEV may skip items.\n\n## Validated Example (pz-save-manager, 2026-06-16)\n\n17 adversarial-review findings, 10 fixed across 3 batches:\n\n| Batch | Bugs | Files | Cycles | Files touched |\n|-------|------|-------|--------|---------------|\n| 1 | A2/B1, B2, B3, C1, C2 | 5 | 1 | gui.py, routes_api.py, watcher.py, index.html, test_gui.py |\n| 2 | B5, B6, B7, B8 | 4 | 1 | routes_api.py, backup.py, index.html, __init__.py |\n| 3 | A4 | 1 | 2 | gui.py |\n\nBatch 1 and 2 both touched `routes_api.py` and `index.html` → must be sequential. Batch 3 touched `gui.py` (also touched by batch 1) → must be sequential.\n\n**Total: ~30 min for 3 batches, 4 cycles, all APPROVED, Codex DEV + Claude Opus REVIEW.**\n\n## Pitfalls\n\n- **Don't batch security hardening with functional fixes.** The review's A1 (loopback gate) is a cross-cutting concern touching every route — it deserves its own batch with careful testing.\n- **The adversarial-code-review cross-review was one-directional** (A reviewed B only). Single-reviewer findings (A1-A9) have lower confidence. Prioritize cross-validated (A2/B1) and consensus (B2-B8) findings first.\n- **Disputed findings (A7/B4) need human adjudication** before entering a dev loop. Don't automate a fix for something the reviewers disagreed on."},{"path":"references/bug-fix-after-review-workflow.md","content":"# Bug Fix After Review — Workflow Pattern\n\nWhen a code review produces 5+ findings across multiple modules,\norganize fixes into small adversarial-loop-sized specs rather than\na single large spec.\n\n## The Pattern\n\n1. **Taxonomy pass**: Group findings by module/severity. Each group\n   becomes one spec (e.g. F1a = wifi_csi data races, F1b = ble_rssi\n   data races, etc.).\n\n2. **Batch by risk**:\n   - Lot 1 (HIGH/CRITICAL): data races, memory safety, init bugs\n   - Lot 2 (MEDIUM): dead code, edge cases, missing init\n   - Lot 3 (LOW): cosmetic, performance, style\n\n3. **Each spec targets 1-3 files max** and adds zero or very few tests.\n   The review already validated the tests — the fix just needs to\n   compile and pass.\n\n4. **Launch order**: most critical first, simplest first. This maximizes\n   the chance that every fix gets done even if quota runs out.\n\n## Example: OmniSense Bug Fix Session\n\n```\nLot 1 — Data races (4 specs, 4 files total)\n  F1a  wifi_csi atomics\n  F1b  ble_rssi atomics\n  F1c  subghz atomics\n  F1d  fusion volatile -> _Atomic\n\nLot 2 — Logic bugs (3 specs, 4 files)\n  F2a  config SD init in setup()\n  F2b  VHCI stale-response race\n  F2c  call fusion_set_band_snr()\n\nLot 3 — Edge cases (2 specs, 2 files)\n  F3a  millis() wraparound\n  F3b  csi_doppler min subcarriers\n```\n\n## When FIX Times Out\n\nIf the FIX phase times out (common with Codex sandbox at 300s\non integration specs):\n\n1. Kill the process\n2. `git diff --stat` — verify the BUILD wrote the core changes\n3. Read `02_review.json` — check if the findings are critical\n4. Apply critical fixes manually\n5. `make all && pio run` — if it compiles and tests pass, commit\n6. Document skipped findings as technical debt\n\nIf Codex wrote files during BUILD (sandbox mode), they're always\nrecoverable via `git diff --stat` even if FIX never ran."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"BUILD → REVIEW → (FIX → VERIFY)^N → ARBITER on isolated git branches. Git-native: each loop runs on its own branch, changes are committed, reviews inspect git diffs. Skill: adversarial-code-loop Owner: chpomob Summary: BUILD → REVIEW → (FIX → VERIFY)^N → ARBITER on isolated git branches. Git-native: each loop runs on its own branch, changes are committed, reviews inspect git diffs. Tags: latest:0.1.0 Version history: v0.1.0 | 2026-08-03T18:12:20.116Z | auto Adversarial Code Loop v4 is a major update making the workflow fully git-native. - Each code-review loop runs on its own iso","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1706,"uniquenessScore":47,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T13:35:10.043Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T13:35:10.043Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T16:06:24.296Z","emptyReason":null},"items":[{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-10T18:48:31.762Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}